commit 64f0a6b2ea8d9b79580df8ed14d8b69b84e4870d
parent 89b8136e0ec9de9bfb911c82a111d43c21aecb02
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 9 Jul 2026 00:21:37 -0400
Ask chat: drill into a hit — read the surrounding transcript on demand
Pinned search results were fixed to their ~240-char matched snippets, too
thin to answer "why did they say that / what surrounded it." Add on-demand
depth two ways, both sharing one windowing mechanism (fetch transcript →
window cues around a moment → merge as snippets into the grounding a video
actually feeds the answer):
- User-driven "Load context" in the pinned panel — deterministic, no AI call,
works in strict AND expand mode on every provider.
- AI fetch_context tool — expand mode + native transports only; the model
pulls context itself when a snippet is too thin. Scripted-fallback providers
degrade gracefully (no tool; user-expand covers them).
The answer phase reads only the grounding block, so fetched context is merged
back into the video's snippets to survive into the answer + citations. Windows
are bounded (±45s, ≤60 cues, ≤30 excerpts/video) so multi-hour transcripts are
never dumped; enriched excerpts persist with the conversation. Pagination was
considered and skipped (ranking front-loads relevance; expand-mode re-search
covers "more").
New common/lib/transcriptWindow.ts (windowCues/cuesToSnippets/mergeSnippets,
reused both sides). fetch_context registered conditionally across the three
native providers with generalized parsers. +7 unit (transcriptWindow, parsers),
+3 ask-chat e2e (AI tool, user-expand w/ reload, scripted degrade).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
18 files changed, 1011 insertions(+), 151 deletions(-)
diff --git a/common/lib/aiHandoff.ts b/common/lib/aiHandoff.ts
@@ -52,7 +52,7 @@ export type HandoffSummaryRef = {
siteTitle?: string;
};
-function hms(s: number): string {
+export function hms(s: number): string {
const n = Math.max(0, Math.floor(s));
const h = Math.floor(n / 3600);
const m = Math.floor((n % 3600) / 60);
diff --git a/common/lib/transcriptWindow.test.ts b/common/lib/transcriptWindow.test.ts
@@ -0,0 +1,84 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { windowCues, cuesToSnippets, mergeSnippets } from "./transcriptWindow";
+import type { Cue } from "./vtt";
+
+function cue(start: number, text = `t${start}`): Cue {
+ return { start, end: start + 2, text };
+}
+
+test("windowCues keeps cues within [center-before, center+after]", () => {
+ const cues = [cue(0), cue(30), cue(60), cue(90), cue(120)];
+ const win = windowCues(cues, 60, { before: 45, after: 45 });
+ assert.deepEqual(
+ win.map((c) => c.start),
+ [30, 60, 90],
+ );
+});
+
+test("windowCues handles a center at 0s", () => {
+ const cues = [cue(0), cue(10), cue(50), cue(100)];
+ const win = windowCues(cues, 0, { before: 45, after: 45 });
+ assert.deepEqual(
+ win.map((c) => c.start),
+ [0, 10],
+ );
+});
+
+test("windowCues caps to the maxCues closest, restoring chronological order", () => {
+ const cues = Array.from({ length: 20 }, (_, i) => cue(i * 5)); // 0,5,…,95
+ const win = windowCues(cues, 50, { before: 100, after: 100, maxCues: 3 });
+ // Closest to 50 are 50,45,55 → returned sorted.
+ assert.deepEqual(
+ win.map((c) => c.start),
+ [45, 50, 55],
+ );
+});
+
+test("cuesToSnippets formats clock, collapses whitespace, caps to 240 chars", () => {
+ const long = "a ".repeat(200); // 400 chars pre-slice
+ const snips = cuesToSnippets([cue(65, "hello world"), cue(0, " "), cue(3600, long)]);
+ // Empty cue dropped.
+ assert.equal(snips.length, 2);
+ assert.deepEqual(snips[0], { clock: "1:05", seconds: 65, text: "hello world" });
+ assert.equal(snips[1].clock, "1:00:00");
+ assert.ok(snips[1].text.length <= 240);
+});
+
+test("mergeSnippets dedupes by seconds, keeps existing, sorts, caps", () => {
+ const existing = [
+ { seconds: 100, text: "hit-a" },
+ { seconds: 20, text: "hit-b" },
+ ];
+ const incoming = [
+ { seconds: 20, text: "dupe-should-not-replace" },
+ { seconds: 10, text: "ctx-1" },
+ { seconds: 30, text: "ctx-2" },
+ ];
+ const merged = mergeSnippets(existing, incoming, 10);
+ assert.deepEqual(
+ merged.map((s) => s.seconds),
+ [10, 20, 30, 100],
+ );
+ // Existing snippet at 20 wins over the incoming duplicate.
+ assert.equal(merged.find((s) => s.seconds === 20)?.text, "hit-b");
+});
+
+test("mergeSnippets never drops existing hits and stops adding at cap", () => {
+ const existing = [
+ { seconds: 1, text: "a" },
+ { seconds: 2, text: "b" },
+ ];
+ const incoming = [
+ { seconds: 3, text: "c" },
+ { seconds: 4, text: "d" },
+ { seconds: 5, text: "e" },
+ ];
+ const merged = mergeSnippets(existing, incoming, 3);
+ // Both existing kept; only one incoming added to reach cap 3.
+ assert.equal(merged.length, 3);
+ assert.deepEqual(
+ merged.map((s) => s.seconds),
+ [1, 2, 3],
+ );
+});
diff --git a/common/lib/transcriptWindow.ts b/common/lib/transcriptWindow.ts
@@ -0,0 +1,77 @@
+import type { Cue } from "./vtt";
+import { hms } from "./aiHandoff";
+
+// Turning a hit into its surrounding transcript. Both the /ask chat's user-driven
+// "expand this hit" and the AI's fetch_context tool share these helpers: fetch a
+// full transcript (client-side, IDB-cached), take a bounded window of cues around
+// a timestamp, and merge them into a video's snippets (the citable excerpts). All
+// pure so they unit-test without I/O.
+
+// Same shape as HandoffSnippet / RetrievedVideo's snippet, so a windowed slice
+// drops straight into the grounding.
+export type WindowSnippet = { clock: string; seconds: number; text: string };
+
+// Match askRetrieval's per-snippet slice so windowed lines stay the same size as
+// search hits.
+const SNIPPET_MAX_CHARS = 240;
+
+// Cues whose start falls within [center-before, center+after] seconds. When more
+// than `maxCues` qualify, keep the `maxCues` closest to the center (a token guard
+// for very dense stretches) but return them back in chronological order.
+export function windowCues(
+ cues: Cue[],
+ centerSeconds: number,
+ opts: { before?: number; after?: number; maxCues?: number } = {},
+): Cue[] {
+ const before = opts.before ?? 45;
+ const after = opts.after ?? 45;
+ const maxCues = opts.maxCues ?? 60;
+ const lo = centerSeconds - before;
+ const hi = centerSeconds + after;
+ const inWindow = cues.filter((c) => c.start >= lo && c.start <= hi);
+ if (inWindow.length <= maxCues) return inWindow;
+ return inWindow
+ .slice()
+ .sort(
+ (a, b) =>
+ Math.abs(a.start - centerSeconds) - Math.abs(b.start - centerSeconds),
+ )
+ .slice(0, maxCues)
+ .sort((a, b) => a.start - b.start);
+}
+
+// Render cues as snippet objects (clock + seconds + collapsed, capped text).
+// Empty cues are dropped.
+export function cuesToSnippets(cues: Cue[]): WindowSnippet[] {
+ const out: WindowSnippet[] = [];
+ for (const c of cues) {
+ const text = c.text.trim().replace(/\s+/g, " ").slice(0, SNIPPET_MAX_CHARS);
+ if (!text) continue;
+ out.push({
+ clock: hms(c.start),
+ seconds: Math.max(0, Math.floor(c.start)),
+ text,
+ });
+ }
+ return out;
+}
+
+// Merge freshly-windowed snippets into a video's existing ones: keep every
+// existing snippet (the original hits are the most relevant), append incoming
+// ones (deduped by second) until `cap`, and return sorted by timestamp. Never
+// drops existing hits, so enrichment only ever adds context.
+export function mergeSnippets<T extends { seconds: number }>(
+ existing: T[],
+ incoming: T[],
+ cap: number,
+): T[] {
+ const seen = new Set(existing.map((s) => s.seconds));
+ const merged = existing.slice();
+ for (const s of incoming) {
+ if (merged.length >= cap) break;
+ if (seen.has(s.seconds)) continue;
+ seen.add(s.seconds);
+ merged.push(s);
+ }
+ return merged.sort((a, b) => a.seconds - b.seconds);
+}
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Drill into a result — read the transcript around any hit.** Pinned search results used to be fixed to their matched snippets (~240 characters), which is often too little to answer "*why* did they say that / what surrounded it." Now the surrounding transcript can be pulled in on demand, two ways. In the pinned panel, expand a video and click a timestamp (or **context**) to add the neighbouring transcript to what the assistant reads — deterministic, no extra AI call, and it works in strict mode and on every provider. And in expand mode on tool-capable providers, the assistant can do this itself via a new **fetch_context** tool when a snippet is too thin, showing a *reading* step in the live pipeline. Windows are bounded (±45s, capped cues, ≤30 excerpts per video) so full multi-hour transcripts are never dumped, and enriched excerpts persist with the conversation. See `common/lib/transcriptWindow.ts`, `export/app/lib/nativeTools/*`, `export/app/lib/searchAgent.ts`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,PipelineStatus.tsx}`, and `export/e2e/ask-chat.spec.ts`.
- **Hand a search's results straight to the "Ask AI" chat.** The search results header gains an **Ask AI about these results** button (next to *Copy for AI*) that opens the chat grounded in *exactly* the videos you found, instead of the assistant deciding its own search. By default it answers **only** from those results (fast and predictable); a per-chat toggle — *Answer only from these results* — lets the assistant also search the archive, using your results as a starting point. The pinned set is shown with the search that produced it, survives reloads, and can be detached (**Clear**) or replaced with a new hand-off. See `common/lib/aiHandoff.ts`, `common/components/TranscriptSearch.tsx`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,AskChat.tsx}`, `export/app/lib/searchAgent.ts`, and `export/e2e/{ask-chat,query-tree}.spec.ts`.
## [0.7.5] - 2026-07-07
diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx
@@ -117,8 +117,10 @@ export default function AskChat() {
pinned={s.pinned}
strictGrounding={s.strictGrounding}
busy={busy}
+ expanding={s.expanding}
onSetStrict={s.setStrictGrounding}
onClear={s.clearPinned}
+ onExpandVideo={s.expandPinnedVideo}
/>
)}
diff --git a/export/app/ask/MessageBubble.tsx b/export/app/ask/MessageBubble.tsx
@@ -76,11 +76,17 @@ export function MessageBubble({
</div>
)}
- {!isUser && message.phase === "done" && (message.searchSteps?.length ?? 0) > 0 && (
- <p className="font-mono text-xs text-muted-foreground/70">
- Searched: {message.searchSteps!.map((s) => s.query).join(" · ")}
- </p>
- )}
+ {!isUser &&
+ message.phase === "done" &&
+ (message.searchSteps?.filter((s) => s.kind !== "fetch").length ?? 0) > 0 && (
+ <p className="font-mono text-xs text-muted-foreground/70">
+ Searched:{" "}
+ {message
+ .searchSteps!.filter((s) => s.kind !== "fetch")
+ .map((s) => s.query)
+ .join(" · ")}
+ </p>
+ )}
{!isUser && message.phase === "done" && message.content && (
<div className="flex items-center">
diff --git a/export/app/ask/PinnedResultsPanel.tsx b/export/app/ask/PinnedResultsPanel.tsx
@@ -1,24 +1,35 @@
"use client";
import { useState } from "react";
-import { ChevronDownIcon, PinIcon, XIcon } from "lucide-react";
+import {
+ ChevronDownIcon,
+ Loader2Icon,
+ PinIcon,
+ PlusIcon,
+ XIcon,
+} from "lucide-react";
import type { SearchHandoff } from "yt-dlp-transcript-common/lib/aiHandoff";
// Shows the search results handed off from the search page as the chat's pinned
// grounding: what they are, whether the assistant may look beyond them (the
-// strict/expand toggle), and a way to detach them.
+// strict/expand toggle), a way to read more transcript around any hit, and a way
+// to detach them.
export function PinnedResultsPanel({
pinned,
strictGrounding,
busy,
+ expanding,
onSetStrict,
onClear,
+ onExpandVideo,
}: {
pinned: SearchHandoff;
strictGrounding: boolean;
busy: boolean;
+ expanding: Record<string, boolean>;
onSetStrict: (on: boolean) => void;
onClear: () => void;
+ onExpandVideo: (key: string, aroundSeconds?: number) => void;
}) {
const [showList, setShowList] = useState(false);
const n = pinned.videos.length;
@@ -78,15 +89,63 @@ export function PinnedResultsPanel({
{showList ? "Hide" : "Show"} the {n} video{n === 1 ? "" : "s"}
</button>
{showList && (
- <ul className="mt-2 flex flex-col gap-1 border-t border-border pt-2">
- {pinned.videos.map((v) => (
- <li key={v.key} className="truncate text-xs text-muted-foreground">
- <span className="text-foreground">{v.title}</span>
- {v.channel ? ` — ${v.channel}` : ""}
- </li>
- ))}
+ <ul className="mt-2 flex flex-col gap-3 border-t border-border pt-2">
+ {pinned.videos.map((v) => {
+ const loading = !!expanding[v.key];
+ return (
+ <li key={v.key} className="flex flex-col gap-1 text-xs">
+ <div className="flex items-start gap-2">
+ <div className="min-w-0 flex-1 truncate text-muted-foreground">
+ <span className="text-foreground">{v.title}</span>
+ {v.channel ? ` — ${v.channel}` : ""}
+ <span className="text-muted-foreground/60">
+ {" "}
+ · {v.snippets.length} excerpt
+ {v.snippets.length === 1 ? "" : "s"}
+ </span>
+ </div>
+ <button
+ type="button"
+ onClick={() => onExpandVideo(v.key)}
+ disabled={busy || loading}
+ title="Read the transcript around the top hit and add it to what the assistant reads"
+ className="inline-flex shrink-0 items-center gap-1 rounded-md border border-border px-1.5 py-0.5 text-muted-foreground transition-colors hover:text-foreground disabled:opacity-50"
+ >
+ {loading ? (
+ <Loader2Icon className="size-3 animate-spin motion-reduce:animate-none" />
+ ) : (
+ <PlusIcon className="size-3" />
+ )}
+ context
+ </button>
+ </div>
+ {v.snippets.length > 0 && (
+ <ul className="flex flex-col gap-0.5 border-l border-border pl-2 font-mono text-muted-foreground/80">
+ {v.snippets.map((sn) => (
+ <li key={sn.seconds} className="flex gap-2">
+ <button
+ type="button"
+ onClick={() => onExpandVideo(v.key, sn.seconds)}
+ disabled={busy || loading}
+ title="Read the transcript around this moment"
+ className="shrink-0 text-brand transition-colors hover:underline disabled:opacity-50"
+ >
+ {sn.clock}
+ </button>
+ <span className="truncate">{sn.text}</span>
+ </li>
+ ))}
+ </ul>
+ )}
+ </li>
+ );
+ })}
</ul>
)}
+ <p className="mt-2 text-xs text-muted-foreground/70">
+ Click a timestamp (or “context”) to add the surrounding transcript to
+ what the assistant reads.
+ </p>
</div>
</div>
);
diff --git a/export/app/ask/PipelineStatus.tsx b/export/app/ask/PipelineStatus.tsx
@@ -31,25 +31,32 @@ export function PipelineStatus({
</div>
)}
- {steps.map((s, i) => (
- <div
- key={i}
- className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none"
- >
- {s.count === undefined ? (
- <Loader2Icon className="size-3.5 shrink-0 animate-spin text-brand motion-reduce:animate-none" />
- ) : (
- <CheckIcon className="size-3.5 shrink-0 text-brand" />
- )}
- <span className="shrink-0">searching</span>
- <span className="truncate text-foreground">“{s.query}”</span>
- {s.count !== undefined && (
- <span className="shrink-0 text-muted-foreground/70">
- · {s.count} result{s.count === 1 ? "" : "s"}
+ {steps.map((s, i) => {
+ const isFetch = s.kind === "fetch";
+ return (
+ <div
+ key={i}
+ className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none"
+ >
+ {s.count === undefined ? (
+ <Loader2Icon className="size-3.5 shrink-0 animate-spin text-brand motion-reduce:animate-none" />
+ ) : (
+ <CheckIcon className="size-3.5 shrink-0 text-brand" />
+ )}
+ <span className="shrink-0">{isFetch ? "reading" : "searching"}</span>
+ <span className="truncate text-foreground">
+ {isFetch ? s.query : `“${s.query}”`}
</span>
- )}
- </div>
- ))}
+ {s.count !== undefined && (
+ <span className="shrink-0 text-muted-foreground/70">
+ {isFetch
+ ? `· ${s.count} line${s.count === 1 ? "" : "s"}`
+ : `· ${s.count} result${s.count === 1 ? "" : "s"}`}
+ </span>
+ )}
+ </div>
+ );
+ })}
{phase === "answering" && (
<div className="flex items-center gap-2 animate-in fade-in slide-in-from-top-1 motion-reduce:animate-none">
diff --git a/export/app/ask/useAskChat.ts b/export/app/ask/useAskChat.ts
@@ -6,6 +6,12 @@ import {
AI_HANDOFF_KEY,
type SearchHandoff,
} from "yt-dlp-transcript-common/lib/aiHandoff";
+import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache";
+import {
+ cuesToSnippets,
+ mergeSnippets,
+ windowCues,
+} from "yt-dlp-transcript-common/lib/transcriptWindow";
import { PROVIDERS, type Provider } from "../lib/askProvider";
import { runAskTurn, type AgentEvent, type AgentMode } from "../lib/searchAgent";
import type { RetrievedVideo } from "../lib/askRetrieval";
@@ -61,6 +67,8 @@ export function useAskChat() {
// Pinned mode only: answer strictly from the pinned results (skip gather) vs.
// use them as a starting point the AI may expand with its own searches.
const [strictGrounding, setStrictGroundingState] = useState(true);
+ // Per-video key → true while its "Load context" fetch is in flight.
+ const [expanding, setExpanding] = useState<Record<string, boolean>>({});
const [input, setInput] = useState("");
const [busy, setBusy] = useState(false);
const abortRef = useRef<AbortController | null>(null);
@@ -269,6 +277,28 @@ export function useAskChat() {
return { ...m, searchSteps: steps };
});
break;
+ case "fetch_start":
+ patchAt(assistantIndex, (m) => ({
+ ...m,
+ phase: "gathering",
+ searchSteps: [
+ ...(m.searchSteps ?? []),
+ { query: e.label, kind: "fetch" },
+ ],
+ }));
+ break;
+ case "fetch_done":
+ patchAt(assistantIndex, (m) => {
+ const steps = (m.searchSteps ?? []).slice();
+ for (let i = steps.length - 1; i >= 0; i--) {
+ if (steps[i].kind === "fetch" && steps[i].count === undefined) {
+ steps[i] = { ...steps[i], count: e.count };
+ break;
+ }
+ }
+ return { ...m, searchSteps: steps };
+ });
+ break;
case "answer_start":
// Store grounding NOW (before streaming) so a failed stream can be
// retried without re-searching, and the excerpts persist.
@@ -448,6 +478,54 @@ export function useAskChat() {
const setStrictGrounding = (on: boolean) => setStrictGroundingState(on);
+ // User-driven "expand this hit": read the surrounding transcript for a pinned
+ // video (client-side, IDB-cached) and merge it into that video's snippets, so
+ // the assistant reads the fuller context on the next question. Works in strict
+ // AND expand mode, on every provider — no LLM call. Persisted with the convo.
+ const expandPinnedVideo = useCallback(
+ async (key: string, aroundSeconds?: number) => {
+ const pin = pinnedRef.current;
+ const v = pin?.videos.find((x) => x.key === key);
+ if (!v) return;
+ const center =
+ typeof aroundSeconds === "number"
+ ? aroundSeconds
+ : v.snippets[0]?.seconds ?? 0;
+ setExpanding((e) => ({ ...e, [key]: true }));
+ try {
+ const detail = await fetchTranscript(key);
+ const snips = cuesToSnippets(
+ windowCues(detail.cues ?? [], center, {
+ before: 45,
+ after: 45,
+ maxCues: 60,
+ }),
+ );
+ setPinned((prev) =>
+ prev
+ ? {
+ ...prev,
+ videos: prev.videos.map((x) =>
+ x.key === key
+ ? { ...x, snippets: mergeSnippets(x.snippets, snips, 30) }
+ : x,
+ ),
+ }
+ : prev,
+ );
+ } catch {
+ /* transcript unavailable — leave the pin untouched */
+ } finally {
+ setExpanding((e) => {
+ const next = { ...e };
+ delete next[key];
+ return next;
+ });
+ }
+ },
+ [],
+ );
+
// The effective context that will be sent next turn, serialized for the panel:
// the override (if any) prepended to the replayed conversation.
const contextText = useMemo(() => {
@@ -509,6 +587,8 @@ export function useAskChat() {
strictGrounding,
setStrictGrounding,
clearPinned,
+ expandPinnedVideo,
+ expanding,
};
}
diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts
@@ -14,9 +14,11 @@ import type { ChatMessage } from "./askProvider";
import { buildContext, type RetrievedVideo } from "./askRetrieval";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
-// One search the agent ran this turn (shown live in the pipeline UI). `count`
-// is undefined while the search is in flight, then set to the match count.
-export type SearchStep = { query: string; count?: number };
+// One retrieval step the agent ran this turn (shown live in the pipeline UI).
+// `kind` is "search" (default) or "fetch" (reading more of a pinned video's
+// transcript); for a fetch step `query` holds the video title. `count` is
+// undefined while in flight, then the match/line count.
+export type SearchStep = { query: string; count?: number; kind?: "search" | "fetch" };
// Top-level stage of a pending assistant turn, for the status indicator.
export type AssistantPhase =
@@ -162,6 +164,7 @@ export function gatherSystemPrompt(
aliases: SearchAlias[],
mode: "native" | "scripted",
budget: number,
+ canFetch = false,
): string {
const base =
"You are helping answer a question about a video-transcript archive. " +
@@ -172,10 +175,17 @@ export function gatherSystemPrompt(
"Only search when you need transcript evidence: if the latest message just " +
"asks to reformat, summarise, translate, or expand on the previous answer, " +
`do not search. You may run at most ${budget} searches.`;
+ const fetchLine = canFetch
+ ? " When a snippet is too short to answer confidently, call the " +
+ "fetch_context tool with the video's ref (and optionally a timestamp in " +
+ "seconds) to read the surrounding transcript before answering."
+ : "";
const tail =
mode === "native"
- ? "Call the search_transcripts tool to search. Call the finish tool as " +
- "soon as you have enough excerpts, or immediately if no search is needed."
+ ? "Call the search_transcripts tool to search." +
+ fetchLine +
+ " Call the finish tool as soon as you have enough excerpts, or " +
+ "immediately if no search is needed."
: "Reply with EXACTLY one line and nothing else: either " +
"`SEARCH: <query>` to run a search, or `DONE` when you have enough " +
"excerpts (or need no search).";
diff --git a/export/app/lib/nativeTools/anthropic.ts b/export/app/lib/nativeTools/anthropic.ts
@@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider";
import {
abortError,
EMPTY_PARAMS,
+ FETCH_BUDGET,
+ FETCH_PARAMS,
+ FETCH_TOOL_DESCRIPTION,
+ FETCH_TOOL_NAME,
FINISH_TOOL_DESCRIPTION,
FINISH_TOOL_NAME,
+ type ParsedToolCall,
postJson,
SEARCH_PARAMS,
SEARCH_TOOL_DESCRIPTION,
@@ -14,38 +19,53 @@ import {
type NativeGatherContext,
} from "./shared";
-const TOOLS = [
- {
- name: SEARCH_TOOL_NAME,
- description: SEARCH_TOOL_DESCRIPTION,
- input_schema: SEARCH_PARAMS,
- },
- {
+// The fetch_context tool is offered only when the turn can read transcripts
+// (expand mode with a pinned set); search + finish are always present.
+function buildTools(includeFetch: boolean) {
+ const tools: { name: string; description: string; input_schema: unknown }[] = [
+ {
+ name: SEARCH_TOOL_NAME,
+ description: SEARCH_TOOL_DESCRIPTION,
+ input_schema: SEARCH_PARAMS,
+ },
+ ];
+ if (includeFetch) {
+ tools.push({
+ name: FETCH_TOOL_NAME,
+ description: FETCH_TOOL_DESCRIPTION,
+ input_schema: FETCH_PARAMS,
+ });
+ }
+ tools.push({
name: FINISH_TOOL_NAME,
description: FINISH_TOOL_DESCRIPTION,
input_schema: EMPTY_PARAMS,
- },
-];
+ });
+ return tools;
+}
// Pure: extract tool_use blocks from an Anthropic message response.
-export function parseAnthropicToolUses(
- json: unknown,
-): { id: string; name: string; query: string }[] {
+export function parseAnthropicToolUses(json: unknown): ParsedToolCall[] {
const content = (json as { content?: unknown }).content;
if (!Array.isArray(content)) return [];
- const out: { id: string; name: string; query: string }[] = [];
+ const out: ParsedToolCall[] = [];
for (const block of content) {
const b = block as {
type?: string;
id?: string;
name?: string;
- input?: { query?: unknown };
+ input?: { query?: unknown; video?: unknown; aroundSeconds?: unknown };
};
if (b.type === "tool_use" && typeof b.name === "string") {
out.push({
id: b.id ?? "",
name: b.name,
query: typeof b.input?.query === "string" ? b.input.query : "",
+ video: typeof b.input?.video === "string" ? b.input.video : "",
+ aroundSeconds:
+ typeof b.input?.aroundSeconds === "number"
+ ? b.input.aroundSeconds
+ : undefined,
});
}
}
@@ -53,13 +73,17 @@ export function parseAnthropicToolUses(
}
export async function anthropicGather(ctx: NativeGatherContext): Promise<void> {
+ const canFetch = !!ctx.runFetchContext;
+ const tools = buildTools(canFetch);
const messages: { role: string; content: unknown }[] = [
...ctx.history.map((m) => ({ role: m.role, content: m.content })),
{ role: "user", content: ctx.question },
];
let searchesRun = 0;
+ let fetchesRun = 0;
+ const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0);
- for (let round = 0; round <= ctx.budget + 1; round++) {
+ for (let round = 0; round <= maxRounds; round++) {
if (ctx.signal?.aborted) throw abortError();
const json = await postJson(
"https://api.anthropic.com/v1/messages",
@@ -72,7 +96,7 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> {
model: ctx.model || PROVIDERS.anthropic.defaultModel,
max_tokens: 512,
system: ctx.system,
- tools: TOOLS,
+ tools,
messages,
},
"Anthropic",
@@ -80,11 +104,14 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> {
);
const calls = parseAnthropicToolUses(json);
- const searches = calls.filter(
+ const wantsSearch = calls.some(
(c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "",
);
- // No search requested → the model finished (or answered directly). Done.
- if (searches.length === 0) return;
+ const wantsFetch =
+ canFetch &&
+ calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== "");
+ // No actionable tool call → the model finished (or answered directly). Done.
+ if (!wantsSearch && !wantsFetch) return;
// Replay the assistant's tool_use turn verbatim, then answer each tool_use.
messages.push({
@@ -93,30 +120,35 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> {
});
const toolResults: unknown[] = [];
for (const call of calls) {
+ let content: string;
if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") {
- const content =
- searchesRun < ctx.budget
- ? await ((): Promise<string> => {
- searchesRun += 1;
- return ctx.runSearch(call.query);
- })()
- : "Search budget reached — answer with the excerpts gathered so far.";
- toolResults.push({
- type: "tool_result",
- tool_use_id: call.id,
- content,
- });
+ if (searchesRun < ctx.budget) {
+ searchesRun += 1;
+ content = await ctx.runSearch(call.query);
+ } else {
+ content = "Search budget reached — answer with the excerpts gathered so far.";
+ }
+ } else if (
+ canFetch &&
+ call.name === FETCH_TOOL_NAME &&
+ call.video.trim() !== ""
+ ) {
+ if (fetchesRun < FETCH_BUDGET) {
+ fetchesRun += 1;
+ content = await ctx.runFetchContext!(call.video, call.aroundSeconds);
+ } else {
+ content = "Reached the limit on transcript reads this turn.";
+ }
} else {
// finish (or any other tool): acknowledge so the block is satisfied.
- toolResults.push({
- type: "tool_result",
- tool_use_id: call.id,
- content: "Acknowledged.",
- });
+ content = "Acknowledged.";
}
+ toolResults.push({
+ type: "tool_result",
+ tool_use_id: call.id,
+ content,
+ });
}
messages.push({ role: "user", content: toolResults });
-
- if (searchesRun >= ctx.budget) return;
}
}
diff --git a/export/app/lib/nativeTools/gemini.ts b/export/app/lib/nativeTools/gemini.ts
@@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider";
import {
abortError,
EMPTY_PARAMS,
+ FETCH_BUDGET,
+ FETCH_PARAMS,
+ FETCH_TOOL_DESCRIPTION,
+ FETCH_TOOL_NAME,
FINISH_TOOL_DESCRIPTION,
FINISH_TOOL_NAME,
+ type ParsedToolCall,
postJson,
SEARCH_PARAMS,
SEARCH_TOOL_DESCRIPTION,
@@ -14,41 +19,61 @@ import {
type NativeGatherContext,
} from "./shared";
-const TOOLS = [
- {
- functionDeclarations: [
- {
- name: SEARCH_TOOL_NAME,
- description: SEARCH_TOOL_DESCRIPTION,
- parameters: SEARCH_PARAMS,
- },
- {
- name: FINISH_TOOL_NAME,
- description: FINISH_TOOL_DESCRIPTION,
- parameters: EMPTY_PARAMS,
- },
- ],
- },
-];
+function buildTools(includeFetch: boolean) {
+ const functionDeclarations: {
+ name: string;
+ description: string;
+ parameters: unknown;
+ }[] = [
+ {
+ name: SEARCH_TOOL_NAME,
+ description: SEARCH_TOOL_DESCRIPTION,
+ parameters: SEARCH_PARAMS,
+ },
+ ];
+ if (includeFetch) {
+ functionDeclarations.push({
+ name: FETCH_TOOL_NAME,
+ description: FETCH_TOOL_DESCRIPTION,
+ parameters: FETCH_PARAMS,
+ });
+ }
+ functionDeclarations.push({
+ name: FINISH_TOOL_NAME,
+ description: FINISH_TOOL_DESCRIPTION,
+ parameters: EMPTY_PARAMS,
+ });
+ return [{ functionDeclarations }];
+}
// Pure: extract functionCall parts from a Gemini generateContent response.
-export function parseGeminiFunctionCalls(
- json: unknown,
-): { name: string; query: string }[] {
+export function parseGeminiFunctionCalls(json: unknown): ParsedToolCall[] {
const parts =
(
json as {
candidates?: { content?: { parts?: unknown[] } }[];
}
).candidates?.[0]?.content?.parts ?? [];
- const out: { name: string; query: string }[] = [];
+ const out: ParsedToolCall[] = [];
for (const part of parts) {
- const fc = (part as { functionCall?: { name?: string; args?: { query?: unknown } } })
- .functionCall;
+ const fc = (
+ part as {
+ functionCall?: {
+ name?: string;
+ args?: { query?: unknown; video?: unknown; aroundSeconds?: unknown };
+ };
+ }
+ ).functionCall;
if (fc?.name) {
out.push({
+ id: "",
name: fc.name,
query: typeof fc.args?.query === "string" ? fc.args.query : "",
+ video: typeof fc.args?.video === "string" ? fc.args.video : "",
+ aroundSeconds:
+ typeof fc.args?.aroundSeconds === "number"
+ ? fc.args.aroundSeconds
+ : undefined,
});
}
}
@@ -56,6 +81,8 @@ export function parseGeminiFunctionCalls(
}
export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
+ const canFetch = !!ctx.runFetchContext;
+ const tools = buildTools(canFetch);
const model = ctx.model || PROVIDERS.gemini.defaultModel;
const url =
`https://generativelanguage.googleapis.com/v1beta/models/` +
@@ -68,15 +95,17 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
{ role: "user", parts: [{ text: ctx.question }] },
];
let searchesRun = 0;
+ let fetchesRun = 0;
+ const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0);
- for (let round = 0; round <= ctx.budget + 1; round++) {
+ for (let round = 0; round <= maxRounds; round++) {
if (ctx.signal?.aborted) throw abortError();
const json = await postJson(
url,
{},
{
system_instruction: { parts: [{ text: ctx.system }] },
- tools: TOOLS,
+ tools,
contents,
},
"Gemini",
@@ -84,16 +113,29 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
);
const calls = parseGeminiFunctionCalls(json);
- const searches = calls.filter(
+ const wantsSearch = calls.some(
(c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "",
);
- if (searches.length === 0) return; // finish or a plain answer → done
+ const wantsFetch =
+ canFetch &&
+ calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== "");
+ if (!wantsSearch && !wantsFetch) return; // finish or a plain answer → done
// Replay the model's functionCall turn, then send functionResponse parts.
const modelParts = calls.map((c) => ({
functionCall: {
name: c.name,
- args: c.name === SEARCH_TOOL_NAME ? { query: c.query } : {},
+ args:
+ c.name === SEARCH_TOOL_NAME
+ ? { query: c.query }
+ : c.name === FETCH_TOOL_NAME
+ ? {
+ video: c.video,
+ ...(c.aroundSeconds != null
+ ? { aroundSeconds: c.aroundSeconds }
+ : {}),
+ }
+ : {},
},
}));
contents.push({ role: "model", parts: modelParts });
@@ -102,13 +144,23 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
for (const call of calls) {
let result: string;
if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") {
- result =
- searchesRun < ctx.budget
- ? await (() => {
- searchesRun += 1;
- return ctx.runSearch(call.query);
- })()
- : "Search budget reached — answer with the excerpts gathered so far.";
+ if (searchesRun < ctx.budget) {
+ searchesRun += 1;
+ result = await ctx.runSearch(call.query);
+ } else {
+ result = "Search budget reached — answer with the excerpts gathered so far.";
+ }
+ } else if (
+ canFetch &&
+ call.name === FETCH_TOOL_NAME &&
+ call.video.trim() !== ""
+ ) {
+ if (fetchesRun < FETCH_BUDGET) {
+ fetchesRun += 1;
+ result = await ctx.runFetchContext!(call.video, call.aroundSeconds);
+ } else {
+ result = "Reached the limit on transcript reads this turn.";
+ }
} else {
result = "Acknowledged.";
}
@@ -118,7 +170,5 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
}
// Function responses are sent back as a user-role turn.
contents.push({ role: "user", parts: responseParts });
-
- if (searchesRun >= ctx.budget) return;
}
}
diff --git a/export/app/lib/nativeTools/nativeTools.test.ts b/export/app/lib/nativeTools/nativeTools.test.ts
@@ -13,8 +13,8 @@ test("parseAnthropicToolUses extracts tool_use blocks", () => {
],
};
assert.deepEqual(parseAnthropicToolUses(json), [
- { id: "tu_1", name: "search_transcripts", query: "platner" },
- { id: "tu_2", name: "finish", query: "" },
+ { id: "tu_1", name: "search_transcripts", query: "platner", video: "", aroundSeconds: undefined },
+ { id: "tu_2", name: "finish", query: "", video: "", aroundSeconds: undefined },
]);
assert.deepEqual(parseAnthropicToolUses({ content: "no tools" }), []);
});
@@ -36,8 +36,8 @@ test("parseOpenAIToolCalls parses JSON arguments", () => {
],
};
assert.deepEqual(parseOpenAIToolCalls(json), [
- { id: "call_1", name: "search_transcripts", query: "senate campaign" },
- { id: "call_2", name: "finish", query: "" },
+ { id: "call_1", name: "search_transcripts", query: "senate campaign", video: "", aroundSeconds: undefined },
+ { id: "call_2", name: "finish", query: "", video: "", aroundSeconds: undefined },
]);
// Malformed arguments degrade to an empty query, not a throw.
const bad = {
@@ -55,7 +55,7 @@ test("parseOpenAIToolCalls parses JSON arguments", () => {
],
};
assert.deepEqual(parseOpenAIToolCalls(bad), [
- { id: "x", name: "search_transcripts", query: "" },
+ { id: "x", name: "search_transcripts", query: "", video: "", aroundSeconds: undefined },
]);
assert.deepEqual(parseOpenAIToolCalls({ choices: [{ message: {} }] }), []);
});
@@ -74,7 +74,7 @@ test("parseGeminiFunctionCalls extracts functionCall parts", () => {
],
};
assert.deepEqual(parseGeminiFunctionCalls(json), [
- { name: "search_transcripts", query: "timeline" },
+ { id: "", name: "search_transcripts", query: "timeline", video: "", aroundSeconds: undefined },
]);
assert.deepEqual(parseGeminiFunctionCalls({ candidates: [] }), []);
});
diff --git a/export/app/lib/nativeTools/openai.ts b/export/app/lib/nativeTools/openai.ts
@@ -5,8 +5,13 @@ import { PROVIDERS } from "../askProvider";
import {
abortError,
EMPTY_PARAMS,
+ FETCH_BUDGET,
+ FETCH_PARAMS,
+ FETCH_TOOL_DESCRIPTION,
+ FETCH_TOOL_NAME,
FINISH_TOOL_DESCRIPTION,
FINISH_TOOL_NAME,
+ type ParsedToolCall,
postJson,
SEARCH_PARAMS,
SEARCH_TOOL_DESCRIPTION,
@@ -14,24 +19,40 @@ import {
type NativeGatherContext,
} from "./shared";
-const TOOLS = [
- {
- type: "function",
- function: {
- name: SEARCH_TOOL_NAME,
- description: SEARCH_TOOL_DESCRIPTION,
- parameters: SEARCH_PARAMS,
+function buildTools(includeFetch: boolean) {
+ const tools: {
+ type: "function";
+ function: { name: string; description: string; parameters: unknown };
+ }[] = [
+ {
+ type: "function",
+ function: {
+ name: SEARCH_TOOL_NAME,
+ description: SEARCH_TOOL_DESCRIPTION,
+ parameters: SEARCH_PARAMS,
+ },
},
- },
- {
+ ];
+ if (includeFetch) {
+ tools.push({
+ type: "function",
+ function: {
+ name: FETCH_TOOL_NAME,
+ description: FETCH_TOOL_DESCRIPTION,
+ parameters: FETCH_PARAMS,
+ },
+ });
+ }
+ tools.push({
type: "function",
function: {
name: FINISH_TOOL_NAME,
description: FINISH_TOOL_DESCRIPTION,
parameters: EMPTY_PARAMS,
},
- },
-];
+ });
+ return tools;
+}
type RawToolCall = {
id?: string;
@@ -39,34 +60,40 @@ type RawToolCall = {
};
// Pure: extract tool calls from an OpenAI chat-completion response.
-export function parseOpenAIToolCalls(
- json: unknown,
-): { id: string; name: string; query: string }[] {
+export function parseOpenAIToolCalls(json: unknown): ParsedToolCall[] {
const msg = (json as { choices?: { message?: { tool_calls?: unknown } }[] })
.choices?.[0]?.message;
const calls = (msg as { tool_calls?: unknown })?.tool_calls;
if (!Array.isArray(calls)) return [];
return (calls as RawToolCall[]).map((c) => {
let query = "";
+ let video = "";
+ let aroundSeconds: number | undefined;
try {
const args = JSON.parse(c.function?.arguments ?? "{}");
if (typeof args.query === "string") query = args.query;
+ if (typeof args.video === "string") video = args.video;
+ if (typeof args.aroundSeconds === "number") aroundSeconds = args.aroundSeconds;
} catch {
- /* malformed args → empty query */
+ /* malformed args → empty */
}
- return { id: c.id ?? "", name: c.function?.name ?? "", query };
+ return { id: c.id ?? "", name: c.function?.name ?? "", query, video, aroundSeconds };
});
}
export async function openaiGather(ctx: NativeGatherContext): Promise<void> {
+ const canFetch = !!ctx.runFetchContext;
+ const tools = buildTools(canFetch);
const messages: unknown[] = [
{ role: "system", content: ctx.system },
...ctx.history.map((m) => ({ role: m.role, content: m.content })),
{ role: "user", content: ctx.question },
];
let searchesRun = 0;
+ let fetchesRun = 0;
+ const maxRounds = ctx.budget + 1 + (canFetch ? FETCH_BUDGET : 0);
- for (let round = 0; round <= ctx.budget + 1; round++) {
+ for (let round = 0; round <= maxRounds; round++) {
if (ctx.signal?.aborted) throw abortError();
const json = await postJson(
"https://api.openai.com/v1/chat/completions",
@@ -74,7 +101,7 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> {
{
model: ctx.model || PROVIDERS.openai.defaultModel,
max_tokens: 512,
- tools: TOOLS,
+ tools,
tool_choice: "auto",
messages,
},
@@ -85,11 +112,14 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> {
const rawMsg = (json as { choices?: { message?: unknown }[] }).choices?.[0]
?.message;
const calls = parseOpenAIToolCalls(json);
- const searches = calls.filter(
+ const wantsSearch = calls.some(
(c) => c.name === SEARCH_TOOL_NAME && c.query.trim() !== "",
);
- // No tool call (or only a plain answer) → done gathering.
- if (calls.length === 0 || searches.length === 0) return;
+ const wantsFetch =
+ canFetch &&
+ calls.some((c) => c.name === FETCH_TOOL_NAME && c.video.trim() !== "");
+ // No actionable tool call → done gathering.
+ if (!wantsSearch && !wantsFetch) return;
// Replay the assistant message with its tool_calls, then answer EVERY call
// (OpenAI requires a tool response for each tool_call id).
@@ -97,19 +127,27 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> {
for (const call of calls) {
let content: string;
if (call.name === SEARCH_TOOL_NAME && call.query.trim() !== "") {
- content =
- searchesRun < ctx.budget
- ? await (() => {
- searchesRun += 1;
- return ctx.runSearch(call.query);
- })()
- : "Search budget reached — answer with the excerpts gathered so far.";
+ if (searchesRun < ctx.budget) {
+ searchesRun += 1;
+ content = await ctx.runSearch(call.query);
+ } else {
+ content = "Search budget reached — answer with the excerpts gathered so far.";
+ }
+ } else if (
+ canFetch &&
+ call.name === FETCH_TOOL_NAME &&
+ call.video.trim() !== ""
+ ) {
+ if (fetchesRun < FETCH_BUDGET) {
+ fetchesRun += 1;
+ content = await ctx.runFetchContext!(call.video, call.aroundSeconds);
+ } else {
+ content = "Reached the limit on transcript reads this turn.";
+ }
} else {
content = "Acknowledged.";
}
messages.push({ role: "tool", tool_call_id: call.id, content });
}
-
- if (searchesRun >= ctx.budget) return;
}
}
diff --git a/export/app/lib/nativeTools/parsers.test.ts b/export/app/lib/nativeTools/parsers.test.ts
@@ -0,0 +1,90 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { parseAnthropicToolUses } from "./anthropic";
+import { parseOpenAIToolCalls } from "./openai";
+import { parseGeminiFunctionCalls } from "./gemini";
+import { FETCH_TOOL_NAME, SEARCH_TOOL_NAME } from "./shared";
+
+test("parseAnthropicToolUses reads search and fetch_context calls", () => {
+ const calls = parseAnthropicToolUses({
+ content: [
+ { type: "text", text: "thinking" },
+ { type: "tool_use", id: "a", name: SEARCH_TOOL_NAME, input: { query: "platner" } },
+ {
+ type: "tool_use",
+ id: "b",
+ name: FETCH_TOOL_NAME,
+ input: { video: "chan/xyz", aroundSeconds: 125 },
+ },
+ ],
+ });
+ assert.equal(calls.length, 2);
+ assert.deepEqual(
+ { ...calls[0] },
+ { id: "a", name: SEARCH_TOOL_NAME, query: "platner", video: "", aroundSeconds: undefined },
+ );
+ assert.deepEqual(
+ { ...calls[1] },
+ { id: "b", name: FETCH_TOOL_NAME, query: "", video: "chan/xyz", aroundSeconds: 125 },
+ );
+});
+
+test("parseOpenAIToolCalls reads search and fetch_context calls", () => {
+ const calls = parseOpenAIToolCalls({
+ choices: [
+ {
+ message: {
+ tool_calls: [
+ {
+ id: "1",
+ function: { name: SEARCH_TOOL_NAME, arguments: JSON.stringify({ query: "platner" }) },
+ },
+ {
+ id: "2",
+ function: {
+ name: FETCH_TOOL_NAME,
+ arguments: JSON.stringify({ video: "chan/xyz", aroundSeconds: 60 }),
+ },
+ },
+ ],
+ },
+ },
+ ],
+ });
+ assert.equal(calls.length, 2);
+ assert.equal(calls[0].name, SEARCH_TOOL_NAME);
+ assert.equal(calls[0].query, "platner");
+ assert.equal(calls[1].name, FETCH_TOOL_NAME);
+ assert.equal(calls[1].video, "chan/xyz");
+ assert.equal(calls[1].aroundSeconds, 60);
+});
+
+test("parseOpenAIToolCalls tolerates malformed arguments", () => {
+ const calls = parseOpenAIToolCalls({
+ choices: [
+ { message: { tool_calls: [{ id: "1", function: { name: SEARCH_TOOL_NAME, arguments: "{bad" } }] } },
+ ],
+ });
+ assert.equal(calls[0].query, "");
+ assert.equal(calls[0].video, "");
+});
+
+test("parseGeminiFunctionCalls reads search and fetch_context calls", () => {
+ const calls = parseGeminiFunctionCalls({
+ candidates: [
+ {
+ content: {
+ parts: [
+ { functionCall: { name: SEARCH_TOOL_NAME, args: { query: "platner" } } },
+ { functionCall: { name: FETCH_TOOL_NAME, args: { video: "chan/xyz", aroundSeconds: 30 } } },
+ ],
+ },
+ },
+ ],
+ });
+ assert.equal(calls.length, 2);
+ assert.equal(calls[0].query, "platner");
+ assert.equal(calls[1].name, FETCH_TOOL_NAME);
+ assert.equal(calls[1].video, "chan/xyz");
+ assert.equal(calls[1].aroundSeconds, 30);
+});
diff --git a/export/app/lib/nativeTools/shared.ts b/export/app/lib/nativeTools/shared.ts
@@ -19,17 +19,32 @@ export type NativeGatherContext = {
budget: number;
// Runs one search and returns the results text to feed back to the model.
runSearch: (query: string) => Promise<string>;
+ // Reads more of a pinned video's transcript around a moment and returns the
+ // window text. Present only in expand mode with a pinned set (native only);
+ // when undefined the fetch_context tool is not offered.
+ runFetchContext?: (video: string, aroundSeconds?: number) => Promise<string>;
signal?: AbortSignal;
};
export const SEARCH_TOOL_NAME = "search_transcripts";
+export const FETCH_TOOL_NAME = "fetch_context";
export const FINISH_TOOL_NAME = "finish";
+// Bounds transcript reads per turn (independent of the search budget) so the
+// user's key isn't spent on unbounded fetching.
+export const FETCH_BUDGET = 4;
+
export const SEARCH_TOOL_DESCRIPTION =
"Search the video-transcript archive for excerpts relevant to a query. " +
"Returns matching videos with timestamped snippet lines. Use focused keyword " +
"or name queries; call it again to refine based on what you find.";
+export const FETCH_TOOL_DESCRIPTION =
+ "Read more of a specific pinned video's transcript around a moment, to get the " +
+ "context surrounding a snippet. Pass the video's ref (shown with each pinned " +
+ "result) and optionally a timestamp in seconds to centre on. The returned " +
+ "lines are added to your citable excerpts for that video.";
+
export const FINISH_TOOL_DESCRIPTION =
"Call this when you have gathered enough excerpts (or none are needed) and " +
"are ready to answer.";
@@ -47,8 +62,36 @@ export const SEARCH_PARAMS = {
required: ["query"],
} as const;
+// JSON-Schema for the fetch_context tool's input.
+export const FETCH_PARAMS = {
+ type: "object",
+ properties: {
+ video: {
+ type: "string",
+ description: "The video ref to read, exactly as shown with the pinned results.",
+ },
+ aroundSeconds: {
+ type: "number",
+ description:
+ "Centre the excerpt on this timestamp (seconds). Optional; defaults to " +
+ "the video's first matched moment.",
+ },
+ },
+ required: ["video"],
+} as const;
+
export const EMPTY_PARAMS = { type: "object", properties: {} } as const;
+// Shape returned by every provider's tool-call parser: `query` for search,
+// `video`/`aroundSeconds` for fetch_context (all optional, filled per tool).
+export type ParsedToolCall = {
+ id: string;
+ name: string;
+ query: string;
+ video: string;
+ aroundSeconds?: number;
+};
+
// Thrown when a native tool-calling request fails in a way that suggests the
// model/endpoint doesn't support tools (HTTP 400/404). The agent catches this
// and retries the turn with the scripted transport.
diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts
@@ -32,12 +32,22 @@ import {
} from "./nativeTools/shared";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts";
+import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache";
+import {
+ cuesToSnippets,
+ mergeSnippets,
+ windowCues,
+} from "yt-dlp-transcript-common/lib/transcriptWindow";
+import { hms } from "yt-dlp-transcript-common/lib/aiHandoff";
export type AgentMode = "auto" | "native" | "scripted";
export type AgentEvent =
| { type: "search_start"; query: string }
| { type: "search_done"; query: string; count: number }
+ // The model read more of a pinned video's transcript (fetch_context tool).
+ | { type: "fetch_start"; ref: string; label: string }
+ | { type: "fetch_done"; ref: string; count: number }
// Fired once gather is done, BEFORE the answer streams — carries the grounding
// so the UI can persist it and a failed answer stream can be retried without
// re-searching.
@@ -50,6 +60,10 @@ export const DEFAULT_BUDGET = 4;
// Per-search result cap and overall context cap (bounds tokens sent to the model).
const PER_SEARCH_LIMIT = 8;
const MAX_CONTEXT_VIDEOS = 15;
+// fetch_context windowing: ± seconds around the moment, a hard cue cap, and the
+// max snippets a video may accumulate (so enrichment can't blow the budget).
+const FETCH_WINDOW = { before: 45, after: 45, maxCues: 60 };
+const SNIPPETS_PER_VIDEO_CAP = 30;
// ─── transport selection ───
@@ -151,6 +165,26 @@ export function formatResultsForModel(videos: RetrievedVideo[]): string {
.join("\n\n");
}
+// Tell the model about the pinned results it's grounded in (expand mode), each
+// with the `ref` it passes to fetch_context and the moments that matched — so it
+// can reason about them and read more around any moment.
+export function buildSeedDigest(videos: RetrievedVideo[]): string {
+ const lines = videos.map((v) => {
+ const times = v.snippets
+ .slice(0, 6)
+ .map((s) => s.clock)
+ .join(", ");
+ const site = v.siteTitle ? ` (${v.siteTitle})` : "";
+ return `- ref "${v.key}": "${v.title}" — ${v.channel}${site}${times ? `; matched at ${times}` : ""}`;
+ });
+ return (
+ "PINNED RESULTS — the user handed you these search results as your starting " +
+ "grounding. You may search the archive for more, and you may read more of " +
+ "any of these around a moment:\n" +
+ lines.join("\n")
+ );
+}
+
// ─── the turn ───
export type RunAskTurnOptions = {
@@ -262,6 +296,46 @@ export async function runAskTurn(
return formatResultsForModel(r.videos);
};
+ // Expand mode with a pinned set: let the model read more of a grounding video's
+ // transcript around a moment. Windows the cues client-side (the channel page is
+ // already warm from the search), merges them into that video's snippets — so the
+ // deeper context reaches the answer + citations, not just the gather loop.
+ const canFetchThisTurn = seedVideos.length > 0;
+ const runFetchContext = async (
+ rawRef: string,
+ aroundSeconds?: number,
+ ): Promise<string> => {
+ const ref = rawRef.trim();
+ const v = videos.get(ref);
+ if (!v) return `No pinned video with ref "${ref}".`;
+ const center =
+ typeof aroundSeconds === "number"
+ ? aroundSeconds
+ : v.snippets[0]?.seconds ?? 0;
+ onEvent({ type: "fetch_start", ref: v.key, label: v.title });
+ let cues;
+ try {
+ const detail = await fetchTranscript(v.key);
+ cues = detail.cues ?? [];
+ } catch {
+ onEvent({ type: "fetch_done", ref: v.key, count: 0 });
+ return `Couldn't load the transcript for "${v.title}".`;
+ }
+ const snips = cuesToSnippets(windowCues(cues, center, FETCH_WINDOW));
+ const before = v.snippets.length;
+ v.snippets = mergeSnippets(v.snippets, snips, SNIPPETS_PER_VIDEO_CAP);
+ const added = v.snippets.length - before;
+ onEvent({ type: "fetch_done", ref: v.key, count: snips.length });
+ if (snips.length === 0) {
+ return `No transcript lines found near ${hms(center)} in "${v.title}".`;
+ }
+ const body = snips.map((s) => `[${s.clock}] ${s.text}`).join("\n");
+ return (
+ `Transcript excerpt from "${v.title}" around ${hms(center)} ` +
+ `(${added} new line${added === 1 ? "" : "s"} added to your citable excerpts):\n${body}`
+ );
+ };
+
const gatherCtx: NativeGatherContext = {
apiKey,
model,
@@ -274,9 +348,13 @@ export async function runAskTurn(
};
const doGather = async (kind: "native" | "scripted"): Promise<void> => {
+ const canFetch = kind === "native" && canFetchThisTurn;
+ let system = gatherSystemPrompt(aliases, kind, budget, canFetch);
+ if (canFetchThisTurn) system += `\n\n${buildSeedDigest(seedVideos)}`;
const ctx: NativeGatherContext = {
...gatherCtx,
- system: gatherSystemPrompt(aliases, kind, budget),
+ system,
+ runFetchContext: canFetch ? runFetchContext : undefined,
};
if (kind === "scripted") return scriptedGather(ctx, provider);
switch (provider) {
diff --git a/export/e2e/ask-chat.spec.ts b/export/e2e/ask-chat.spec.ts
@@ -264,11 +264,214 @@ test.describe("ask chat", () => {
}, HANDOFF);
}
- async function keyIn(page: Page) {
- await page.getByRole("button", { name: "Scripted" }).click();
+ async function keyIn(page: Page, mode: "Scripted" | "Native tools" = "Scripted") {
+ await page.getByRole("button", { name: mode }).click();
await page.locator('input[placeholder^="sk-ant"]').fill("sk-ant-test");
}
+ // A hand-off whose video key resolves against the transcript fixtures
+ // (installRoutes), so fetch_context / "Load context" can pull real cues. The
+ // fixture cues sit at 5s, 50s, 100s; a window around the 5s hit pulls in the
+ // 50s "beta line" — the signal that a fetch enriched the grounding.
+ const REAL_KEY = "test-channel/vid-transcript-only";
+ const FETCH_HANDOFF = {
+ label: "alpha",
+ videos: [
+ {
+ key: REAL_KEY,
+ videoId: "vid-transcript-only",
+ title: "Transcript only",
+ channel: "Test Channel",
+ uploadDate: "20200101",
+ url: "https://x/v1",
+ snippets: [
+ { clock: "0:05", seconds: 5, text: "transcript-only video — alpha line" },
+ ],
+ },
+ ],
+ totalVideos: 1,
+ truncated: false,
+ };
+
+ // Seed once — guarded so a reload doesn't re-seed (which would clobber the
+ // enriched pin restored from localStorage).
+ async function seedFetchHandoff(page: Page) {
+ await page.addInitScript((h) => {
+ if (!sessionStorage.getItem("seeded_once")) {
+ sessionStorage.setItem("ytdlp-tb:ai:handoff", JSON.stringify(h));
+ sessionStorage.setItem("seeded_once", "1");
+ }
+ }, FETCH_HANDOFF);
+ }
+
+ test("expand mode: the model reads more transcript via fetch_context", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ let sawFetchTool = false;
+ let answerSawWindow = false;
+ let round = 0;
+ await page.route("https://api.anthropic.com/**", async (route) => {
+ if (route.request().method() === "OPTIONS") {
+ await route.fulfill({ status: 204, headers: CORS });
+ return;
+ }
+ const body = route.request().postDataJSON() as {
+ system?: string;
+ tools?: { name?: string }[];
+ messages?: { role: string; content: unknown }[];
+ };
+ const system = body.system ?? "";
+ if (system.includes("Markdown")) {
+ answerSawWindow = JSON.stringify(body.messages ?? []).includes("beta line");
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse("Fetched answer [1]"),
+ });
+ return;
+ }
+ // Native gather turn.
+ if (Array.isArray(body.tools)) {
+ if (body.tools.some((t) => t.name === "fetch_context")) sawFetchTool = true;
+ const msgs = body.messages ?? [];
+ const lastUser = [...msgs].reverse().find((m) => m.role === "user");
+ const isToolResult = Array.isArray(lastUser?.content);
+ round += 1;
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "application/json" },
+ body: isToolResult
+ ? toolUse("finish", {})
+ : toolUse("fetch_context", { video: REAL_KEY, aroundSeconds: 5 }),
+ });
+ return;
+ }
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse("DONE"),
+ });
+ });
+ await seedFetchHandoff(page);
+ await page.goto("/ask/");
+ await keyIn(page, "Native tools");
+
+ await expect(page.getByText(/Grounded in 1 result/)).toBeVisible();
+ // Expand mode → the fetch_context tool is offered and used.
+ await page.getByLabel("Answer only from these results").uncheck();
+ await ask(page, "what surrounds the alpha moment");
+
+ await expect(page.getByText(/Fetched answer/)).toBeVisible();
+ expect(sawFetchTool).toBe(true);
+ // The windowed line (50s "beta line") reached the answer's grounding.
+ expect(answerSawWindow).toBe(true);
+ expect(round).toBeGreaterThanOrEqual(2); // fetch round + finish round
+ });
+
+ test("user 'Load context' enriches the pin and persists (strict, no search)", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ let gatherCalls = 0;
+ let answerSawWindow = false;
+ await page.route("https://api.anthropic.com/**", async (route) => {
+ if (route.request().method() === "OPTIONS") {
+ await route.fulfill({ status: 204, headers: CORS });
+ return;
+ }
+ const body = route.request().postDataJSON() as {
+ system?: string;
+ messages?: { role: string; content: unknown }[];
+ };
+ const system = body.system ?? "";
+ if (system.includes("Markdown")) {
+ answerSawWindow = JSON.stringify(body.messages ?? []).includes("beta line");
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse("Answer over expanded excerpts [1]"),
+ });
+ return;
+ }
+ gatherCalls += 1; // strict grounding → this must never fire
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse("SEARCH: alpha"),
+ });
+ });
+ await seedFetchHandoff(page);
+ await page.goto("/ask/");
+ await keyIn(page);
+
+ await expect(page.getByText(/Grounded in 1 result/)).toBeVisible();
+ await page.getByRole("button", { name: /Show the 1 video/ }).click();
+ // Starts with the single matched excerpt.
+ await expect(page.getByText(/1 excerpt/)).toBeVisible();
+ await page.getByRole("button", { name: "context" }).click();
+ // Windowing pulled in the neighbouring cue → the count grows.
+ await expect(page.getByText(/2 excerpts/)).toBeVisible();
+
+ // Persists across a reload (stored in the conversation). The key is only
+ // saved on send, so re-enter it after reloading.
+ await page.waitForTimeout(600);
+ await page.reload();
+ await keyIn(page);
+ await page.getByRole("button", { name: /Show the 1 video/ }).click();
+ await expect(page.getByText(/2 excerpts/)).toBeVisible();
+
+ // Strict answer carries the enriched excerpts, with no gather search.
+ await ask(page, "summarize these");
+ await expect(page.getByText(/Answer over expanded excerpts/)).toBeVisible();
+ expect(gatherCalls).toBe(0);
+ expect(answerSawWindow).toBe(true);
+ });
+
+ test("scripted transport degrades gracefully: no fetch tool, still answers", async ({
+ page,
+ }) => {
+ await installRoutes(page);
+ let fetchToolEverOffered = false;
+ await page.route("https://api.anthropic.com/**", async (route) => {
+ if (route.request().method() === "OPTIONS") {
+ await route.fulfill({ status: 204, headers: CORS });
+ return;
+ }
+ const body = route.request().postDataJSON() as {
+ system?: string;
+ tools?: { name?: string }[];
+ messages?: { role: string; content: unknown }[];
+ };
+ if (Array.isArray(body.tools) && body.tools.some((t) => t.name === "fetch_context")) {
+ fetchToolEverOffered = true;
+ }
+ const system = body.system ?? "";
+ const msgs = body.messages ?? [];
+ const lastUser = [...msgs].reverse().find((m) => m.role === "user");
+ const lastText = typeof lastUser?.content === "string" ? lastUser.content : "";
+ let text: string;
+ if (system.includes("Markdown")) text = "Scripted answer [1]";
+ else if (lastText.includes('Results for "')) text = "DONE";
+ else text = "SEARCH: alpha";
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse(text),
+ });
+ });
+ await seedFetchHandoff(page);
+ await page.goto("/ask/");
+ await keyIn(page, "Scripted");
+
+ await expect(page.getByText(/Grounded in 1 result/)).toBeVisible();
+ await page.getByLabel("Answer only from these results").uncheck();
+ await ask(page, "what surrounds the alpha moment");
+
+ await expect(page.getByText(/Scripted answer/)).toBeVisible();
+ expect(fetchToolEverOffered).toBe(false);
+ });
+
test("pinned strict grounding answers from the handed-off results, no search", async ({
page,
}) => {