commit 150f3df28779adea1554fe6af71f194d4364cd93
parent 3f7b09b9d5972169ac64a9a91cc33d9ffc401d4c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 16 Jul 2026 16:19:36 -0400
Ask chat: fix context ballooning ("trapped in context")
Stop re-embedding every prior turn's excerpts in every turn's replayed
history (quadratic growth that stalled long/grounded conversations).
Excerpts are now carried forward once via a deduped cumulative pool
(collectPriorPool) merged into the current turn's grounding, while
buildApiMessages replays only bare questions + answers. This turn's fresh
videos are prioritised over the old pool when the set is capped.
Also:
- Distinguish a provider "context length exceeded" 400 from a
tools-unavailable 400 (ContextTooLargeError + isContextLengthError) so it
surfaces an actionable message instead of a scripted-retry loop.
- Persist only `sources` per turn (not the grounded text ×3); rebuild
grounding on demand. Handle QuotaExceededError by trimming oldest turns +
warning, instead of silently losing newer turns on reload.
- Normalize a non-terminal phase on restore so a reload mid-answer shows a
Retry rather than spinning forever.
- Hold the Context panel's serialized text stable while streaming (no
per-token textarea remount).
Unit: +collectPriorPool, +isContextLengthError/ContextTooLargeError; updated
buildApiMessages test to bare-replay. 37 export + 59 common green; tsc clean.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
9 files changed, 293 insertions(+), 54 deletions(-)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,5 +1,8 @@
# Changelog
+## [Unreleased]
+- **The "Ask AI" chat no longer balloons its own context.** A long conversation — especially one grounded in a big set of search results — used to re-send *every* previous turn's excerpts inside *every* new turn, so the context grew quadratically until answers stalled or a provider rejected the request. Excerpts are now carried forward once, in a deduplicated pool (each video's excerpts merged across turns and sent a single time), while the replayed history is just the questions and answers. Three related fixes ride along: a "context length exceeded" error from your provider is now recognised and shown as a clear, actionable message instead of being mistaken for "this model doesn't support tools" (which pointlessly retried the oversized request); a conversation that outgrows your browser's storage quota now trims its oldest turns and warns you, instead of silently failing to save so newer turns vanished on reload; and reloading the page mid-answer no longer leaves a message spinning forever — the interrupted turn shows a **Retry**. See `export/app/lib/{askConversation,searchAgent,askProvider}.ts`, `export/app/lib/nativeTools/shared.ts`, and `export/app/ask/{useAskChat.ts,AskChat.tsx}`.
+
## [0.7.6] - 2026-07-16
- **Drill into a result — read the transcript around any hit.** Pinned search results used to be fixed to their matched snippets (~240 characters), which is often too little to answer "*why* did they say that / what surrounded it." Now the surrounding transcript can be pulled in on demand, two ways. In the pinned panel, expand a video and click a timestamp (or **context**) to add the neighbouring transcript to what the assistant reads — deterministic, no extra AI call, and it works in strict mode and on every provider. And in expand mode on tool-capable providers, the assistant can do this itself via a new **fetch_context** tool when a snippet is too thin, showing a *reading* step in the live pipeline. Windows are bounded (±45s, capped cues, ≤30 excerpts per video) so full multi-hour transcripts are never dumped, and enriched excerpts persist with the conversation. See `common/lib/transcriptWindow.ts`, `export/app/lib/nativeTools/*`, `export/app/lib/searchAgent.ts`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,PipelineStatus.tsx}`, and `export/e2e/ask-chat.spec.ts`.
- **Hand a search's results straight to the "Ask AI" chat.** The search results header gains an **Ask AI about these results** button (next to *Copy for AI*) that opens the chat grounded in *exactly* the videos you found, instead of the assistant deciding its own search. By default it answers **only** from those results (fast and predictable); a per-chat toggle — *Answer only from these results* — lets the assistant also search the archive, using your results as a starting point. The pinned set is shown with the search that produced it, survives reloads, and can be detached (**Clear**) or replaced with a new hand-off. See `common/lib/aiHandoff.ts`, `common/components/TranscriptSearch.tsx`, `export/app/ask/{useAskChat.ts,PinnedResultsPanel.tsx,AskChat.tsx}`, `export/app/lib/searchAgent.ts`, and `export/e2e/{ask-chat,query-tree}.spec.ts`.
diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx
@@ -92,6 +92,13 @@ export default function AskChat() {
</p>
)}
+ {s.storageWarning && (
+ <p className="text-xs text-warning">
+ This conversation grew too large to save fully — older turns won't
+ be restored if you reload. Start a new chat to reset.
+ </p>
+ )}
+
{(messages.length > 0 || s.contextOverride || s.pinned) && (
<div className="flex items-center justify-between">
<span className="text-xs text-muted-foreground/70">
diff --git a/export/app/ask/useAskChat.ts b/export/app/ask/useAskChat.ts
@@ -41,6 +41,37 @@ type PersistedConvo = {
strictGrounding?: boolean;
};
+const NON_TERMINAL_PHASES = new Set(["gathering", "answering", "streaming"]);
+
+// A turn that was mid-flight when the tab closed/reloaded can't resume — mark it
+// interrupted (shows a Retry) instead of restoring a phase that spins forever.
+function normalizeRestoredMessage(m: UiMessage): UiMessage {
+ if (m.role === "assistant" && m.phase && NON_TERMINAL_PHASES.has(m.phase)) {
+ return {
+ ...m,
+ phase: "error",
+ error: true,
+ content: m.content || "This answer was interrupted. Retry to continue.",
+ };
+ }
+ return m;
+}
+
+// Write the conversation, catching a storage-quota overflow. Returns true if it
+// persisted, false if the quota was exceeded (so the caller can trim + retry).
+function writeConvo(payload: PersistedConvo): boolean {
+ try {
+ localStorage.setItem(K_CONVO, JSON.stringify(payload));
+ return true;
+ } catch (e) {
+ const quota =
+ e instanceof DOMException &&
+ (e.name === "QuotaExceededError" ||
+ e.name === "NS_ERROR_DOM_QUOTA_REACHED");
+ return !quota; // non-quota errors: treat as "done" (storage unavailable)
+ }
+}
+
// State + turn orchestration for the /ask chat. Owns the conversation (persisted
// across reloads), the retrieval agent turns (gather → answer), a Retry path that
// reuses already-gathered excerpts, and a human-editable context override.
@@ -69,6 +100,10 @@ export function useAskChat() {
const [strictGrounding, setStrictGroundingState] = useState(true);
// Per-video key → true while its "Load context" fetch is in flight.
const [expanding, setExpanding] = useState<Record<string, boolean>>({});
+ // Set when the conversation outgrew the storage quota and older turns had to be
+ // dropped from what's saved — surfaced so a reload's missing history isn't a
+ // silent surprise.
+ const [storageWarning, setStorageWarning] = useState(false);
const [input, setInput] = useState("");
const [busy, setBusy] = useState(false);
const abortRef = useRef<AbortController | null>(null);
@@ -124,7 +159,9 @@ export function useAskChat() {
const rawConvo = localStorage.getItem(K_CONVO);
if (rawConvo) {
const parsed = JSON.parse(rawConvo) as PersistedConvo;
- if (Array.isArray(parsed.messages)) setMessages(parsed.messages);
+ if (Array.isArray(parsed.messages)) {
+ setMessages(parsed.messages.map(normalizeRestoredMessage));
+ }
if (typeof parsed.contextOverride === "string") {
setContextOverride(parsed.contextOverride);
}
@@ -151,14 +188,28 @@ export function useAskChat() {
try {
if (messages.length === 0 && !contextOverride && !pinned) {
localStorage.removeItem(K_CONVO);
- } else {
- const payload: PersistedConvo = {
- messages,
- contextOverride,
- pinned,
- strictGrounding,
- };
- localStorage.setItem(K_CONVO, JSON.stringify(payload));
+ setStorageWarning(false);
+ return;
+ }
+ // Persist, trimming the oldest turn-pairs if the blob exceeds the quota,
+ // so a long conversation keeps its most recent history rather than
+ // silently failing to save anything.
+ let msgs = messages;
+ for (;;) {
+ if (writeConvo({ messages: msgs, contextOverride, pinned, strictGrounding })) {
+ setStorageWarning(msgs.length < messages.length);
+ break;
+ }
+ if (msgs.length <= 2) {
+ try {
+ localStorage.removeItem(K_CONVO);
+ } catch {
+ /* ignore */
+ }
+ setStorageWarning(true);
+ break;
+ }
+ msgs = msgs.slice(2);
}
} catch {
/* ignore */
@@ -246,7 +297,7 @@ export function useAskChat() {
};
seedVideos?: RetrievedVideo[];
}) => {
- const { question, prior, userIndex, assistantIndex } = args;
+ const { question, prior, assistantIndex } = args;
setBusy(true);
const ac = new AbortController();
abortRef.current = ac;
@@ -300,17 +351,14 @@ export function useAskChat() {
});
break;
case "answer_start":
- // Store grounding NOW (before streaming) so a failed stream can be
- // retried without re-searching, and the excerpts persist.
+ // Store the grounding videos NOW (before streaming) so a failed
+ // stream can be retried without re-searching, and follow-ups can
+ // reuse the excerpts via the cumulative pool. The full grounded text
+ // is rebuilt from `sources` on demand — not persisted.
patchAt(assistantIndex, (m) => ({
...m,
phase: "answering",
sources: e.videos,
- groundedContent: e.groundedContent,
- }));
- patchAt(userIndex, (m) => ({
- ...m,
- groundedContent: e.groundedContent,
}));
break;
case "delta":
@@ -345,7 +393,6 @@ export function useAskChat() {
}
: undefined,
});
- patchAt(userIndex, (m) => ({ ...m, groundedContent: result.groundedContent }));
patchAt(assistantIndex, (m) => ({
...m,
content: m.content || result.answer,
@@ -437,9 +484,15 @@ export function useAskChat() {
const failed = current[assistantIndex];
if (!userMsg || userMsg.role !== "user" || !failed) return;
const prior = current.slice(0, userIndex);
- const precomputedGrounding = failed.groundedContent
- ? { videos: failed.sources, groundedContent: failed.groundedContent }
- : undefined;
+ // Reuse the grounding captured before the answer stream failed — rebuilt
+ // from the stored `sources` (no re-search).
+ const precomputedGrounding =
+ failed.sources && failed.sources.length
+ ? {
+ videos: failed.sources,
+ groundedContent: buildGroundedContent(userMsg.content, failed.sources),
+ }
+ : undefined;
// Reset the failed message to a pending state (keep captured grounding).
patchAt(assistantIndex, (m) => ({
...m,
@@ -527,11 +580,17 @@ export function useAskChat() {
);
// The effective context that will be sent next turn, serialized for the panel:
- // the override (if any) prepended to the replayed conversation.
+ // the override (if any) prepended to the replayed conversation. Held stable
+ // while a turn streams — recomputing it per token thrashed the panel's
+ // (uncontrolled) textarea and blew away any in-progress edit.
+ const contextTextRef = useRef("");
const contextText = useMemo(() => {
+ if (busy) return contextTextRef.current;
const base = contextOverride ? parseContext(contextOverride) : [];
- return serializeContext([...base, ...buildApiMessages(messages)]);
- }, [contextOverride, messages]);
+ const text = serializeContext([...base, ...buildApiMessages(messages)]);
+ contextTextRef.current = text;
+ return text;
+ }, [contextOverride, messages, busy]);
// Apply an edited context as the new starting point: it becomes the override
// (prepended to future turns) and the visible conversation resets to it.
@@ -577,6 +636,7 @@ export function useAskChat() {
stop,
reset,
retry,
+ storageWarning,
// context editing
contextText,
contextOverride,
diff --git a/export/app/lib/askConversation.test.ts b/export/app/lib/askConversation.test.ts
@@ -4,6 +4,7 @@ import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import {
buildApiMessages,
buildGroundedContent,
+ collectPriorPool,
renderAliasGlossary,
gatherSystemPrompt,
answerSystemPrompt,
@@ -36,31 +37,59 @@ function video(over: Partial<RetrievedVideo> = {}): RetrievedVideo {
};
}
-test("buildApiMessages replays prior user turns with their grounded content", () => {
+test("buildApiMessages replays prior turns as BARE text (no re-embedded excerpts)", () => {
const prior: UiMessage[] = [
- {
- role: "user",
- content: "tell me about Platner",
- groundedContent: "tell me about Platner\n\n---\n[1] excerpt about platner",
- },
+ { role: "user", content: "tell me about Platner" },
{ role: "assistant", content: "He is ...", sources: [video()] },
];
const msgs = buildApiMessages(prior);
assert.equal(msgs.length, 2);
- // The excerpts from turn 1 survive into the replayed history (the core fix).
- assert.match(msgs[0].content, /excerpt about platner/);
+ // The user turn replays as the bare question — excerpts are carried forward via
+ // the cumulative pool, not re-embedded per turn (the "trapped in context" fix).
+ assert.equal(msgs[0].content, "tell me about Platner");
+ assert.doesNotMatch(msgs[0].content, /excerpt|\[1\]/);
assert.equal(msgs[1].content, "He is ...");
});
test("buildApiMessages drops error turns and empty pending assistants", () => {
const prior: UiMessage[] = [
- { role: "user", content: "q", groundedContent: "q" },
+ { role: "user", content: "q" },
{ role: "assistant", content: "boom", error: true },
- { role: "assistant", content: "" }, // pending, no grounded content
+ { role: "assistant", content: "" }, // pending, no content yet
];
assert.equal(buildApiMessages(prior).length, 1);
});
+test("collectPriorPool dedupes videos by key in first-seen order, merging snippets", () => {
+ const prior: UiMessage[] = [
+ {
+ role: "assistant",
+ content: "a1",
+ sources: [
+ video({ key: "a", snippets: [{ clock: "0:05", seconds: 5, text: "five" }] }),
+ video({ key: "b", title: "Second", snippets: [{ clock: "0:10", seconds: 10, text: "ten" }] }),
+ ],
+ },
+ // A later turn expands video "a" with a new snippet + repeats "b".
+ {
+ role: "assistant",
+ content: "a2",
+ sources: [
+ video({ key: "a", snippets: [{ clock: "0:50", seconds: 50, text: "fifty" }] }),
+ ],
+ },
+ { role: "assistant", content: "boom", error: true, sources: [video({ key: "z" })] },
+ ];
+ const pool = collectPriorPool(prior);
+ // First-seen order: a, then b. Error turn's "z" is skipped.
+ assert.deepEqual(pool.map((v) => v.key), ["a", "b"]);
+ // Video "a" merged both turns' snippets (deduped/sorted by seconds).
+ assert.deepEqual(
+ pool[0].snippets.map((sn) => sn.seconds),
+ [5, 50],
+ );
+});
+
test("buildGroundedContent appends numbered excerpts, or just the question", () => {
assert.equal(buildGroundedContent("hi", []), "hi");
const grounded = buildGroundedContent("who?", [video()]);
diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts
@@ -12,6 +12,7 @@
import type { ChatMessage } from "./askProvider";
import { buildContext, type RetrievedVideo } from "./askRetrieval";
+import { mergeSnippets } from "yt-dlp-transcript-common/lib/transcriptWindow";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
// One retrieval step the agent ran this turn (shown live in the pipeline UI).
@@ -32,10 +33,9 @@ export type AssistantPhase =
export type UiMessage = {
role: "user" | "assistant";
content: string;
- // User turns: the exact string sent to the model (question + this turn's
- // excerpts). Replayed verbatim so earlier grounding persists across turns.
- groundedContent?: string;
- // Assistant turns:
+ // Assistant turns: the videos whose excerpts grounded this answer. This is the
+ // persisted grounding — the full grounded text is rebuilt from it on demand
+ // (retry, follow-up pool) rather than stored, keeping the saved blob small.
sources?: RetrievedVideo[];
searchSteps?: SearchStep[];
phase?: AssistantPhase;
@@ -43,18 +43,48 @@ export type UiMessage = {
error?: boolean;
};
-// Replay completed prior turns for the model. User turns use their persisted
-// grounded content (with excerpts); assistant turns use their text. Error turns
-// are dropped. The CURRENT turn is appended by the caller (bare question for the
-// gather phase, grounded question for the answer phase).
+// Replay completed prior turns for the model as BARE text — user turns are the
+// question only, assistant turns are the answer. Excerpts are deliberately NOT
+// re-embedded here: re-sending every prior turn's excerpts in every turn's
+// history grew the context quadratically (the "trapped in context" bug). Instead
+// the excerpts a follow-up may reuse are carried forward once, via the cumulative
+// pool (`collectPriorPool`) that the caller folds into THIS turn's grounding.
+// Error/empty turns are dropped. The current turn is appended by the caller.
export function buildApiMessages(prior: UiMessage[]): ChatMessage[] {
return prior
- .filter((m) => !m.error && (m.content.trim() !== "" || m.groundedContent))
- .map((m) => ({
- role: m.role,
- content:
- m.role === "user" ? m.groundedContent ?? m.content : m.content,
- }));
+ .filter((m) => !m.error && m.content.trim() !== "")
+ .map((m) => ({ role: m.role, content: m.content }));
+}
+
+// The cumulative excerpt pool for a conversation: every video cited in an earlier
+// assistant turn, deduped by `key` in first-seen order, with snippets merged
+// across turns (a video expanded later keeps its fuller excerpts). Seeded into a
+// new turn's grounding so a reformat/summarise follow-up still sees earlier
+// excerpts — but each excerpt is sent ONCE per turn, not re-embedded per prior
+// turn. Pure + testable.
+export function collectPriorPool(
+ prior: UiMessage[],
+ perVideoCap = 30,
+): RetrievedVideo[] {
+ const order: string[] = [];
+ const byKey = new Map<string, RetrievedVideo>();
+ for (const m of prior) {
+ if (m.role !== "assistant" || m.error || !m.sources) continue;
+ for (const v of m.sources) {
+ const existing = byKey.get(v.key);
+ if (!existing) {
+ order.push(v.key);
+ byKey.set(v.key, { ...v, snippets: v.snippets.slice() });
+ } else {
+ existing.snippets = mergeSnippets(
+ existing.snippets,
+ v.snippets,
+ perVideoCap,
+ );
+ }
+ }
+ }
+ return order.map((k) => byKey.get(k)!);
}
// ─── editable context (view / prune / forward to a new session) ───
diff --git a/export/app/lib/askProvider.test.ts b/export/app/lib/askProvider.test.ts
@@ -0,0 +1,47 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { ContextTooLargeError, isContextLengthError } from "./askProvider";
+
+test("isContextLengthError detects provider context-length phrasings", () => {
+ // Anthropic
+ assert.equal(
+ isContextLengthError(
+ '{"type":"error","error":{"type":"invalid_request_error","message":"prompt is too long: 250000 tokens > 200000 maximum"}}',
+ ),
+ true,
+ );
+ // OpenAI
+ assert.equal(
+ isContextLengthError(
+ '{"error":{"message":"This model\'s maximum context length is 128000 tokens.","code":"context_length_exceeded"}}',
+ ),
+ true,
+ );
+ // Gemini
+ assert.equal(
+ isContextLengthError(
+ '{"error":{"code":400,"message":"The input token count (1200000) exceeds the maximum number of tokens allowed"}}',
+ ),
+ true,
+ );
+});
+
+test("isContextLengthError ignores unrelated 400s", () => {
+ assert.equal(
+ isContextLengthError('{"error":{"message":"invalid api key"}}'),
+ false,
+ );
+ assert.equal(
+ isContextLengthError('{"error":{"message":"tools are not supported by this model"}}'),
+ false,
+ );
+ assert.equal(isContextLengthError(""), false);
+});
+
+test("ContextTooLargeError carries an actionable message and stable name", () => {
+ const e = new ContextTooLargeError("anthropic request failed (400).");
+ assert.equal(e.name, "ContextTooLargeError");
+ assert.ok(e instanceof Error);
+ assert.match(e.message, /too large/i);
+ assert.match(e.message, /new chat/i);
+});
diff --git a/export/app/lib/askProvider.ts b/export/app/lib/askProvider.ts
@@ -114,6 +114,39 @@ async function* sseLines(
}
}
+// Thrown when a request fails because the conversation exceeds the model's
+// context window (a provider 400 with a token/context-length signature). Kept
+// distinct from a generic error — and from ToolsUnavailableError — so the chat
+// surfaces an actionable message instead of retrying the oversized request.
+export class ContextTooLargeError extends Error {
+ constructor(providerMsg?: string) {
+ super(
+ "The conversation is too large for this model's context window. Start a " +
+ "new chat, or open the Context panel and trim earlier turns, to continue." +
+ (providerMsg ? `\n\n(${providerMsg})` : ""),
+ );
+ this.name = "ContextTooLargeError";
+ }
+}
+
+// Heuristically detect a "context length exceeded" error from a provider's 400
+// response body. Covers Anthropic ("prompt is too long"), OpenAI
+// ("context_length_exceeded" / "maximum context length"), and Gemini ("input
+// token count … exceeds"). Case-insensitive; matches the common phrasings.
+export function isContextLengthError(detail: string): boolean {
+ const d = detail.toLowerCase();
+ return (
+ d.includes("context_length_exceeded") ||
+ d.includes("context length") ||
+ d.includes("maximum context") ||
+ d.includes("prompt is too long") ||
+ d.includes("too many tokens") ||
+ d.includes("input token count") ||
+ d.includes("reduce the length") ||
+ (d.includes("token") && d.includes("exceed"))
+ );
+}
+
async function ensureOk(res: Response, provider: string): Promise<void> {
if (res.ok) return;
let detail = "";
@@ -122,9 +155,11 @@ async function ensureOk(res: Response, provider: string): Promise<void> {
} catch {
/* ignore */
}
- throw new Error(
- `${provider} request failed (${res.status}). ${detail.slice(0, 300)}`,
- );
+ const msg = `${provider} request failed (${res.status}). ${detail.slice(0, 300)}`;
+ if (res.status === 400 && isContextLengthError(detail)) {
+ throw new ContextTooLargeError(msg);
+ }
+ throw new Error(msg);
}
function emit(full: string[], chunk: string, onDelta?: (c: string) => void): void {
diff --git a/export/app/lib/nativeTools/shared.ts b/export/app/lib/nativeTools/shared.ts
@@ -6,6 +6,7 @@
// provider; this module holds the common types, tool text, and POST helper.
import type { ChatMessage } from "../askProvider";
+import { ContextTooLargeError, isContextLengthError } from "../askProvider";
export type NativeGatherContext = {
apiKey: string;
@@ -126,6 +127,12 @@ export async function postJson(
/* ignore */
}
const msg = `${provider} request failed (${res.status}). ${detail.slice(0, 300)}`;
+ // A 400 caused by an over-large context must NOT be mistaken for "this model
+ // doesn't support tools" (which would pointlessly retry the same oversized
+ // request via the scripted transport). Disambiguate before falling back.
+ if (res.status === 400 && isContextLengthError(detail)) {
+ throw new ContextTooLargeError(msg);
+ }
if (res.status === 400 || res.status === 404) {
throw new ToolsUnavailableError(msg);
}
diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts
@@ -19,6 +19,7 @@ import {
answerSystemPrompt,
buildApiMessages,
buildGroundedContent,
+ collectPriorPool,
gatherSystemPrompt,
type UiMessage,
} from "./askConversation";
@@ -271,10 +272,21 @@ export async function runAskTurn(
);
const videos = new Map<string, RetrievedVideo>();
+ // Keys gathered THIS turn (pinned seed + fresh search hits), as opposed to the
+ // carried-forward pool. Used to prioritise this turn's relevant videos when the
+ // merged set is capped, so a new search isn't starved by an old pool.
+ const freshKeys = new Set<string>();
+ // Carry forward the cumulative excerpt pool from earlier turns so a follow-up
+ // (reformat/summarise/expand) still sees those excerpts — sent once, in this
+ // turn's grounding, rather than re-embedded in every replayed history turn.
+ for (const v of collectPriorPool(prior)) videos.set(v.key, v);
// Pinned-expand: start from the handed-off results, then let the model add to
// them. Deduped by key; still capped below at MAX_CONTEXT_VIDEOS.
const seedVideos = opts.seedVideos ?? [];
- for (const v of seedVideos) videos.set(v.key, v);
+ for (const v of seedVideos) {
+ videos.set(v.key, v);
+ freshKeys.add(v.key);
+ }
const queries: string[] = [];
let truncated = false;
@@ -289,7 +301,10 @@ export async function runAskTurn(
signal,
limit: PER_SEARCH_LIMIT,
});
- for (const v of r.videos) videos.set(v.key, v);
+ for (const v of r.videos) {
+ videos.set(v.key, v);
+ freshKeys.add(v.key);
+ }
if (r.truncated) truncated = true;
queries.push(query);
onEvent({ type: "search_done", query, count: r.videos.length });
@@ -388,7 +403,13 @@ export async function runAskTurn(
await runSearch(question);
}
- const finalVideos = [...videos.values()].slice(0, MAX_CONTEXT_VIDEOS);
+ // Cap the merged set, but keep this turn's fresh videos ahead of the older
+ // carried-forward pool so a new search's results survive the cut.
+ const all = [...videos.values()];
+ const finalVideos = [
+ ...all.filter((v) => freshKeys.has(v.key)),
+ ...all.filter((v) => !freshKeys.has(v.key)),
+ ].slice(0, MAX_CONTEXT_VIDEOS);
const groundedContent = buildGroundedContent(question, finalVideos);
onEvent({ type: "answer_start", videos: finalVideos, groundedContent });