commit 832b236b751407694eded8a48f85c7de4c8f3d1b
parent 5d8ca107a1680e413f16c5215a7d60d961778698
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 18 Jul 2026 19:38:31 -0400
Ask chat: fix silent answer truncation + add a debug export
Long answers (esp. reports on free Gemini) truncated mid-sentence because the
answer was capped at 2048 output tokens and no provider read its finish reason.
- Capture each provider's finish reason: askAnthropic (message_delta
stop_reason), askOpenAI (finish_reason + usage), askGemini (candidates
finishReason + promptFeedback.blockReason + usageMetadata); normFinish()
maps them to length/safety/stop/other. Threaded via an onDebug sink on
AskOptions + NativeGatherContext (gather captured in postJson).
- Raise the answer cap 2048 -> 8192 (DEFAULT_ANSWER_TOKENS) and add a
user-tunable "Max answer length" setting (persisted, clamped).
- Surface it: a "cut off at the output limit" notice when finishReason=length,
and a "blocked (reason)" message for an empty safety-blocked Gemini answer
(instead of a blank bubble). UiMessage gains finishReason/blockReason.
- Debug export: a runtime ring buffer of recent provider calls + a pure
askDebug assembler (chat state + trace), with Download/Copy JSON buttons in
provider settings. The API key is never included (redactKeys backstop).
Verified: tsc clean (export + common); +5 unit (normFinish, askDebug redaction)
= 49 export + 60 common; full e2e 111/111 (+truncation notice, +redacted export
download); no new lint errors.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
16 files changed, 623 insertions(+), 25 deletions(-)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **"Ask AI" chat — long answers no longer silently truncate, plus a debug export.** Long answers (especially reports on the free Gemini tier) used to get cut off mid-sentence with no indication, because the answer was hard-capped at 2048 output tokens and the app ignored the provider's "why did it stop" signal. Now: the answer budget defaults to **8192** and there's a **Max answer length** setting to push it higher; when a model *does* hit its output limit the answer shows a clear **"⚠ Cut off at the model's output limit"** notice (with a nudge to raise the limit or use Report mode); and a Gemini response that comes back empty because it was safety-blocked now says so instead of rendering a blank bubble. For troubleshooting, provider settings gain **Download / Copy debug JSON** — a redacted snapshot of the conversation (messages, phases, finish reasons, report, settings) plus the most recent raw API calls (request, response, finish reason, token usage); **your API key is never included**. See `export/app/lib/{askProvider,searchAgent,askDebug}.ts`, `export/app/lib/nativeTools/shared.ts`, and `export/app/ask/{useAskChat,ProviderSettings,MessageBubble}.tsx`.
- **"Ask AI" chat — clickable citations and smaller niceties.** The `[1]`, `[2]`… citation markers in an answer are now **clickable**: click one to jump straight to that source in the list below (a smooth in-page scroll with a brief highlight — no new tab, no navigation). Also: a suggestion chip now **focuses the composer** when you pick it (so you can tweak and press Enter), and the answer **Copy** button now says "Copy failed" when the browser blocks clipboard access (e.g. on an insecure origin) instead of silently doing nothing. See `common/components/Markdown.tsx` (an optional `linkComponent`), `export/app/ask/{MessageBubble,Composer,AskChat}.tsx`.
- **"Ask AI" chat — an opt-in Report mode for long research sessions.** Turn on **Report mode** (in the new **Report** panel; tool-capable providers only) and the assistant maintains a persistent Markdown **report** document — a running canvas it updates via an `update_report` tool as you keep chatting, upserting well-titled sections that persist across turns. Crucially, while it's on the conversation is **compacted into the report** instead of replaying every prior question and answer: each turn sends a single *"report so far"* summary plus the usual deduplicated excerpt pool, so a long back-and-forth stays bounded in tokens instead of growing every turn. The report renders live in the panel (with an *updating…* shimmer while a turn writes to it) and is saved with the conversation, so it survives reloads; *New chat* clears it. On the Scripted transport there's no tool to drive it, so the toggle is disabled with a hint. See `export/app/lib/nativeTools/{shared,anthropic,openai,gemini}.ts` (the `update_report` tool), `export/app/lib/askConversation.ts` (`applyReportPatch`), `export/app/lib/searchAgent.ts` (compaction + executor), `export/app/ask/{useAskChat.ts,ReportPanel.tsx,AskChat.tsx}`, and `export/e2e/ask-chat.spec.ts`.
- **"Ask AI" now grounds in your *current* search automatically.** With search and chat sharing one workspace, you no longer click "Ask AI about these results" to hand a frozen copy of your results to the chat — the chat reads the **live** search directly. Run a search, switch to Chat, and it's already grounded in exactly those results (with the same *answer only from these / may also search* toggle); change the search and the grounding follows. No active search → the chat searches on its own, as before. A **Detach** control lets you ask a free-form question without the current search grounding it (and a **Ground in my search** button re-attaches). The old "Ask AI about these results" button and its one-shot hand-off are retired. See `common/components/SearchSessionContext.tsx` (`liveGrounding`), `export/app/ask/useAskChat.ts`, `export/app/ask/{AskChat,PinnedResultsPanel}.tsx`, and `common/components/SearchResults.tsx`.
diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx
@@ -92,6 +92,9 @@ export default function AskChat() {
searchMode={s.searchMode}
setSearchMode={s.setSearchMode}
persistKey={s.persistKey}
+ maxAnswerTokens={s.maxAnswerTokens}
+ setMaxAnswerTokens={s.setMaxAnswerTokens}
+ buildDebugJson={s.buildDebugJson}
/>
{corpusError && (
diff --git a/export/app/ask/MessageBubble.tsx b/export/app/ask/MessageBubble.tsx
@@ -165,9 +165,25 @@ export function MessageBubble({
</div>
)}
- {emptyAnswer && (
- <p className="text-sm italic text-muted-foreground/80">
- No answer was returned — retry, or rephrase your question.
+ {emptyAnswer &&
+ (message.finishReason === "safety" || message.blockReason ? (
+ <p className="text-sm text-warning">
+ The model returned no answer — blocked
+ {message.blockReason ? ` (${message.blockReason})` : " for safety"}.
+ Try rephrasing.
+ </p>
+ ) : (
+ <p className="text-sm italic text-muted-foreground/80">
+ No answer was returned — retry, or rephrase your question.
+ </p>
+ ))}
+
+ {!isUser && !!message.content && message.finishReason === "length" && (
+ <p className="text-xs text-warning/90">
+ ⚠ Cut off at the model's output limit. Raise{" "}
+ <span className="font-medium">Max answer length</span> in settings, or
+ turn on <span className="font-medium">Report mode</span> for long
+ reports.
</p>
)}
diff --git a/export/app/ask/ProviderSettings.tsx b/export/app/ask/ProviderSettings.tsx
@@ -24,6 +24,9 @@ type Props = {
searchMode: AgentMode;
setSearchMode: (m: AgentMode) => void;
persistKey: (p: Provider, key: string, mdl: string, rememberOn: boolean) => void;
+ maxAnswerTokens: number;
+ setMaxAnswerTokens: (n: number) => void;
+ buildDebugJson: () => string;
};
export function ProviderSettings(props: Props) {
@@ -41,12 +44,38 @@ export function ProviderSettings(props: Props) {
searchMode,
setSearchMode,
persistKey,
+ maxAnswerTokens,
+ setMaxAnswerTokens,
+ buildDebugJson,
} = props;
const info = PROVIDERS[provider];
// Seed the panel open when there's no key yet, then hand control to the user.
// Must NOT be recomputed from `apiKey` on every render — otherwise typing the
// first character of the key flips it closed and yanks the panel away mid-entry.
const [open, setOpen] = useState(!apiKey);
+ const [debugCopied, setDebugCopied] = useState(false);
+
+ const downloadDebug = () => {
+ const blob = new Blob([buildDebugJson()], { type: "application/json" });
+ const url = URL.createObjectURL(blob);
+ const a = document.createElement("a");
+ a.href = url;
+ a.download = `ask-debug-${Date.now()}.json`;
+ document.body.appendChild(a);
+ a.click();
+ a.remove();
+ URL.revokeObjectURL(url);
+ };
+ const copyDebug = async () => {
+ try {
+ if (!navigator.clipboard) throw new Error("clipboard unavailable");
+ await navigator.clipboard.writeText(buildDebugJson());
+ setDebugCopied(true);
+ setTimeout(() => setDebugCopied(false), 1500);
+ } catch {
+ /* clipboard blocked — the Download button still works */
+ }
+ };
return (
<details
@@ -150,6 +179,50 @@ export function ProviderSettings(props: Props) {
</span>
</div>
+ {/* Answer length: raise it so long reports don't truncate at the cap. */}
+ <label className="flex flex-col gap-1 text-xs text-muted-foreground">
+ Max answer length (tokens)
+ <input
+ type="number"
+ min={256}
+ max={32768}
+ step={512}
+ value={maxAnswerTokens}
+ onChange={(e) => setMaxAnswerTokens(Number(e.target.value))}
+ className="w-32 rounded-md border border-border bg-background px-2 py-1.5 text-sm text-foreground"
+ />
+ <span className="text-muted-foreground/70">
+ How long an answer can be before the model cuts it off. Gemini,
+ OpenAI, and Claude all support ~8k; raise it for big reports.
+ </span>
+ </label>
+
+ {/* Debug export: a redacted snapshot for troubleshooting. */}
+ <div className="flex flex-col gap-1 text-xs text-muted-foreground">
+ <span>Debug</span>
+ <div className="flex flex-wrap items-center gap-2">
+ <button
+ type="button"
+ onClick={downloadDebug}
+ className="rounded-md border border-border px-2.5 py-1 text-xs text-muted-foreground transition-colors hover:text-foreground"
+ >
+ Download debug JSON
+ </button>
+ <button
+ type="button"
+ onClick={() => void copyDebug()}
+ className="rounded-md border border-border px-2.5 py-1 text-xs text-muted-foreground transition-colors hover:text-foreground"
+ >
+ {debugCopied ? "Copied" : "Copy debug JSON"}
+ </button>
+ </div>
+ <span className="text-muted-foreground/70">
+ A redacted snapshot of this chat plus the most recent API calls
+ (finish reasons + payloads) for troubleshooting. Your API key is
+ never included.
+ </span>
+ </div>
+
<p className="text-xs text-muted-foreground/80">
Your key is stored only{" "}
{remember ? "in this browser" : "for this page session"} and is sent
diff --git a/export/app/ask/useAskChat.ts b/export/app/ask/useAskChat.ts
@@ -10,14 +10,16 @@ import {
mergeSnippets,
windowCues,
} from "yt-dlp-transcript-common/lib/transcriptWindow";
-import { PROVIDERS, type Provider } from "../lib/askProvider";
+import { PROVIDERS, type DebugCall, type Provider } from "../lib/askProvider";
import {
+ DEFAULT_ANSWER_TOKENS,
runAskTurn,
supportsNativeTools,
type AgentEvent,
type AgentMode,
} from "../lib/searchAgent";
import type { RetrievedVideo } from "../lib/askRetrieval";
+import { buildDebugExport } from "../lib/askDebug";
import {
buildApiMessages,
buildGroundedContent,
@@ -32,9 +34,16 @@ const K_REMEMBER = "ytdlp-tb:ai:remember";
const K_SEARCHMODE = "ytdlp-tb:ai:searchmode";
const K_MARKDOWN = "ytdlp-tb:ai:md";
const K_CONVO = "ytdlp-tb:ai:conversation";
+const K_MAXANSWER = "ytdlp-tb:ai:maxanswer";
const keyFor = (p: Provider) => `ytdlp-tb:ai:key:${p}`;
const modelFor = (p: Provider) => `ytdlp-tb:ai:model:${p}`;
+// Bounds for the user-set "Max answer length" (output tokens).
+const MAX_ANSWER_MIN = 256;
+const MAX_ANSWER_MAX = 32768;
+// How many recent provider calls to retain for the debug export (ring buffer).
+const DEBUG_TRACE_CAP = 8;
+
type Snip = { clock: string; seconds: number; text: string };
type PersistedConvo = {
@@ -126,6 +135,12 @@ export function useAskChat() {
// True while a turn is writing to the report (drives the panel shimmer);
// cleared when the turn ends.
const [reportUpdating, setReportUpdating] = useState(false);
+ // Max output tokens for the answer phase (user-tunable; default 8192). Raising
+ // it lets long reports finish instead of truncating at the model's output cap.
+ const [maxAnswerTokens, setMaxAnswerTokensState] = useState(DEFAULT_ANSWER_TOKENS);
+ // Ring buffer of the most recent provider calls (request/response/finishReason),
+ // for the debug export. Runtime-only (not persisted); trimmed to DEBUG_TRACE_CAP.
+ const debugTraceRef = useRef<DebugCall[]>([]);
// The effective pinned grounding = the live search, overlaid with any per-video
// "Load context" enrichments, unless the user detached.
@@ -175,6 +190,15 @@ export function useAskChat() {
reportRef.current = report;
const reportModeRef = useRef(reportMode);
reportModeRef.current = reportMode;
+ const maxAnswerTokensRef = useRef(maxAnswerTokens);
+ maxAnswerTokensRef.current = maxAnswerTokens;
+
+ // Append a captured provider call to the debug ring buffer (keep the last N).
+ const pushDebug = useCallback((rec: DebugCall) => {
+ const next = [...debugTraceRef.current, rec];
+ debugTraceRef.current =
+ next.length > DEBUG_TRACE_CAP ? next.slice(-DEBUG_TRACE_CAP) : next;
+ }, []);
// Restore saved preferences + (if remembered) the provider's key/model, plus a
// persisted conversation.
@@ -194,6 +218,10 @@ export function useAskChat() {
setSearchModeState(savedMode);
}
setMarkdownOnState(localStorage.getItem(K_MARKDOWN) !== "0");
+ const savedMax = Number(localStorage.getItem(K_MAXANSWER));
+ if (Number.isFinite(savedMax) && savedMax >= MAX_ANSWER_MIN) {
+ setMaxAnswerTokensState(Math.min(MAX_ANSWER_MAX, savedMax));
+ }
loadProviderCreds(p, rememberSaved);
const rawConvo = localStorage.getItem(K_CONVO);
if (rawConvo) {
@@ -349,6 +377,54 @@ export function useAskChat() {
}
}
+ function setMaxAnswerTokens(n: number) {
+ const clamped = Math.max(
+ MAX_ANSWER_MIN,
+ Math.min(MAX_ANSWER_MAX, Math.round(n) || DEFAULT_ANSWER_TOKENS),
+ );
+ setMaxAnswerTokensState(clamped);
+ try {
+ localStorage.setItem(K_MAXANSWER, String(clamped));
+ } catch {
+ /* ignore */
+ }
+ }
+
+ // Assemble a redacted debug snapshot (chat state + recent raw provider calls)
+ // for offline debugging, and hand it back as a pretty JSON string. The API key
+ // is never included.
+ const buildDebugJson = useCallback((): string => {
+ return buildDebugExport({
+ provider,
+ model: model.trim() || PROVIDERS[provider].defaultModel,
+ searchMode,
+ markdownOn,
+ reportMode,
+ strictGrounding,
+ detached,
+ maxAnswerTokens,
+ messages: messagesRef.current,
+ report,
+ grounding: pinned
+ ? { label: pinned.label, videoCount: pinned.videos.length, truncated: pinned.truncated }
+ : null,
+ contextOverride,
+ trace: debugTraceRef.current,
+ });
+ }, [
+ provider,
+ model,
+ searchMode,
+ markdownOn,
+ reportMode,
+ strictGrounding,
+ detached,
+ maxAnswerTokens,
+ report,
+ pinned,
+ contextOverride,
+ ]);
+
// Report mode is only meaningful on a tool-capable transport: Scripted has no
// tool to maintain the report, so the toggle is disabled there. (Persistence is
// handled by the conversation-persist effect, which lists reportMode as a dep.)
@@ -485,6 +561,8 @@ export function useAskChat() {
seedVideos: args.seedVideos,
report: reportRef.current,
reportMode: effectiveReportMode,
+ maxAnswerTokens: maxAnswerTokensRef.current,
+ onDebug: pushDebug,
precomputedGrounding: args.precomputedGrounding?.groundedContent
? {
videos: args.precomputedGrounding.videos ?? [],
@@ -502,6 +580,8 @@ export function useAskChat() {
content: m.content || result.answer,
sources: result.videos,
truncated: result.truncated,
+ finishReason: result.finishReason,
+ blockReason: result.blockReason,
error: false,
phase: "done",
model: model.trim() || PROVIDERS[provider].defaultModel,
@@ -534,7 +614,7 @@ export function useAskChat() {
abortRef.current = null;
}
},
- [provider, apiKey, model, searchMode, summaries, aliases, reportModeAvailable],
+ [provider, apiKey, model, searchMode, summaries, aliases, reportModeAvailable, pushDebug],
);
const send = useCallback(async () => {
@@ -765,6 +845,10 @@ export function useAskChat() {
setReportMode,
reportModeAvailable,
reportUpdating,
+ // answer length + debug export
+ maxAnswerTokens,
+ setMaxAnswerTokens,
+ buildDebugJson,
// data
summariesReady,
corpusError,
diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts
@@ -44,6 +44,11 @@ export type UiMessage = {
// Assistant turns: which model produced this answer (shown as a per-message
// attribution, since the provider/model can change mid-conversation).
model?: string;
+ // Assistant turns: the answer's normalized finish reason ("length" = cut off at
+ // the output limit; "safety" = blocked) so the UI can flag it. `blockReason` is
+ // the provider's raw block/safety string when present.
+ finishReason?: string;
+ blockReason?: string;
};
// Replay completed prior turns for the model as BARE text — user turns are the
diff --git a/export/app/lib/askDebug.test.ts b/export/app/lib/askDebug.test.ts
@@ -0,0 +1,93 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { buildDebugExport, redactKeys, type DebugExportInput } from "./askDebug";
+import type { UiMessage } from "./askConversation";
+import type { DebugCall } from "./askProvider";
+
+function input(over: Partial<DebugExportInput> = {}): DebugExportInput {
+ const messages: UiMessage[] = [
+ { role: "user", content: "write a report" },
+ {
+ role: "assistant",
+ content: "Here is the report…",
+ phase: "done",
+ truncated: true,
+ finishReason: "length",
+ model: "gemini-2.5-flash",
+ },
+ ];
+ const trace: DebugCall[] = [
+ {
+ phase: "answer",
+ provider: "gemini",
+ model: "gemini-2.5-flash",
+ request: { system: "answer prompt", messages: [], maxTokens: 8192 },
+ response: "Here is the report…",
+ finishReason: "length",
+ rawFinish: "MAX_TOKENS",
+ ms: 1200,
+ },
+ ];
+ return {
+ provider: "gemini",
+ model: "gemini-2.5-flash",
+ searchMode: "auto",
+ markdownOn: true,
+ reportMode: false,
+ strictGrounding: true,
+ detached: false,
+ maxAnswerTokens: 8192,
+ messages,
+ report: "",
+ grounding: { label: "wife", videoCount: 12, truncated: true },
+ contextOverride: null,
+ trace,
+ now: 0,
+ ...over,
+ };
+}
+
+test("buildDebugExport captures state, finishReason, and the call trace", () => {
+ const json = buildDebugExport(input());
+ const obj = JSON.parse(json);
+ assert.equal(obj.meta.provider, "gemini");
+ assert.equal(obj.meta.model, "gemini-2.5-flash");
+ assert.equal(obj.meta.maxAnswerTokens, 8192);
+ assert.equal(obj.meta.exportedAt, "1970-01-01T00:00:00.000Z"); // now:0
+ assert.equal(obj.messages[1].finishReason, "length");
+ assert.equal(obj.messages[1].truncated, true);
+ assert.equal(obj.grounding.videoCount, 12);
+ assert.equal(obj.trace[0].finishReason, "length");
+ assert.equal(obj.trace[0].rawFinish, "MAX_TOKENS");
+});
+
+test("buildDebugExport redacts anything key-shaped (never leaks an API key)", () => {
+ // Even if a key sneaks into a message or the trace, it must be scrubbed.
+ const messages: UiMessage[] = [
+ { role: "user", content: "my key is sk-ant-abcdefghijklmnop1234567890" },
+ ];
+ const trace: DebugCall[] = [
+ {
+ phase: "answer",
+ provider: "openai",
+ request: { note: "Bearer sk-proj-ABCDEFGHIJKLMNOP1234" },
+ response: "AIzaSyA-1234567890abcdefghijklmnopqrstuv",
+ ms: 1,
+ },
+ ];
+ const json = buildDebugExport(input({ messages, trace }));
+ assert.doesNotMatch(json, /sk-ant-[A-Za-z0-9]/);
+ assert.doesNotMatch(json, /sk-proj-[A-Za-z0-9]/);
+ assert.doesNotMatch(json, /AIzaSy[A-Za-z0-9]/);
+ assert.match(json, /\[REDACTED\]/);
+});
+
+test("redactKeys scrubs the three provider key shapes", () => {
+ assert.equal(redactKeys("k=sk-ant-0123456789abcdef end"), "k=[REDACTED] end");
+ assert.equal(redactKeys("k=sk-0123456789abcdef0123 end"), "k=[REDACTED] end");
+ assert.equal(
+ redactKeys("k=AIzaSyABCDEFGHIJ0123456789 end"),
+ "k=[REDACTED] end",
+ );
+ assert.equal(redactKeys("no key here"), "no key here");
+});
diff --git a/export/app/lib/askDebug.ts b/export/app/lib/askDebug.ts
@@ -0,0 +1,62 @@
+// Assembles a redacted debug snapshot of the /ask chat — the conversation state
+// plus a trace of recent raw provider calls (request / response / finish reason)
+// — as a pretty JSON string for offline debugging (truncation, empty answers,
+// "trapped in context"). Pure + unit-testable. An API key is NEVER included: the
+// inputs don't carry one, and `redactKeys` scrubs anything key-shaped as a
+// defense-in-depth backstop.
+
+import type { UiMessage } from "./askConversation";
+import type { DebugCall } from "./askProvider";
+
+export type DebugExportInput = {
+ provider: string;
+ model: string;
+ searchMode: string;
+ markdownOn: boolean;
+ reportMode: boolean;
+ strictGrounding: boolean;
+ detached: boolean;
+ maxAnswerTokens: number;
+ messages: UiMessage[];
+ report: string;
+ grounding: { label: string; videoCount: number; truncated: boolean } | null;
+ contextOverride: string | null;
+ trace: DebugCall[];
+ now?: number; // injectable timestamp for tests
+};
+
+// Provider API-key shapes: Anthropic (sk-ant-…), OpenAI (sk-…), Gemini (AIza…).
+const KEY_PATTERNS = [
+ /sk-ant-[A-Za-z0-9_-]{10,}/g,
+ /sk-[A-Za-z0-9_-]{16,}/g,
+ /AIza[A-Za-z0-9_-]{10,}/g,
+];
+
+export function redactKeys(s: string): string {
+ return KEY_PATTERNS.reduce((acc, re) => acc.replace(re, "[REDACTED]"), s);
+}
+
+export function buildDebugExport(input: DebugExportInput): string {
+ const obj = {
+ meta: {
+ kind: "ytdlp-transcript-browser/ask-debug",
+ version: 1,
+ exportedAt: new Date(input.now ?? Date.now()).toISOString(),
+ provider: input.provider,
+ model: input.model,
+ searchMode: input.searchMode,
+ markdownOn: input.markdownOn,
+ reportMode: input.reportMode,
+ strictGrounding: input.strictGrounding,
+ detached: input.detached,
+ maxAnswerTokens: input.maxAnswerTokens,
+ },
+ grounding: input.grounding,
+ contextOverride: input.contextOverride,
+ report: input.report,
+ messages: input.messages,
+ // Recent raw provider calls (finish reasons, payloads). No API key.
+ trace: input.trace,
+ };
+ return redactKeys(JSON.stringify(obj, null, 2));
+}
diff --git a/export/app/lib/askProvider.test.ts b/export/app/lib/askProvider.test.ts
@@ -1,6 +1,28 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { ContextTooLargeError, isContextLengthError } from "./askProvider";
+import {
+ ContextTooLargeError,
+ isContextLengthError,
+ normFinish,
+} from "./askProvider";
+
+test("normFinish maps each provider's raw finish string to a normalized reason", () => {
+ // length (truncation): Anthropic max_tokens, OpenAI length, Gemini MAX_TOKENS
+ assert.equal(normFinish("max_tokens"), "length");
+ assert.equal(normFinish("length"), "length");
+ assert.equal(normFinish("MAX_TOKENS"), "length");
+ // safety / blocked
+ assert.equal(normFinish("content_filter"), "safety");
+ assert.equal(normFinish("SAFETY"), "safety");
+ assert.equal(normFinish("RECITATION"), "safety");
+ // normal completion
+ assert.equal(normFinish("end_turn"), "stop");
+ assert.equal(normFinish("stop"), "stop");
+ assert.equal(normFinish("STOP"), "stop");
+ assert.equal(normFinish("tool_use"), "stop");
+ // unknown
+ assert.equal(normFinish("something_else"), "other");
+});
test("isContextLengthError detects provider context-length phrasings", () => {
// Anthropic
diff --git a/export/app/lib/askProvider.ts b/export/app/lib/askProvider.ts
@@ -52,6 +52,47 @@ export const PROVIDERS: Record<Provider, ProviderInfo> = {
},
};
+// Normalized provider finish reason. `length` = truncated at the output cap
+// (Anthropic max_tokens / OpenAI length / Gemini MAX_TOKENS); `safety` = blocked
+// (content_filter / SAFETY / RECITATION); `stop` = a normal completion.
+export type FinishReason = "length" | "stop" | "safety" | "other";
+
+// One captured provider API call, for the debug export. `request`/`response` are
+// the raw-ish payloads (system + messages / accumulated text for answers; the
+// full body / parsed JSON for gather rounds). An API key is NEVER included.
+export type DebugCall = {
+ phase: "gather" | "answer";
+ provider: string;
+ model?: string;
+ request: unknown;
+ response: unknown;
+ finishReason?: FinishReason;
+ rawFinish?: string;
+ blockReason?: string;
+ usage?: unknown;
+ ms: number;
+};
+
+// Map a provider's raw finish/stop string to our normalized FinishReason.
+export function normFinish(raw: string): FinishReason {
+ const r = raw.toLowerCase();
+ if (r === "max_tokens" || r === "length") return "length";
+ if (
+ r === "safety" ||
+ r === "recitation" ||
+ r === "content_filter" ||
+ r === "prohibited_content" ||
+ r === "blocklist" ||
+ r === "spii"
+ ) {
+ return "safety";
+ }
+ if (r === "end_turn" || r === "stop" || r === "stop_sequence" || r === "tool_use") {
+ return "stop";
+ }
+ return "other";
+}
+
export type AskOptions = {
provider: Provider;
apiKey: string;
@@ -61,6 +102,9 @@ export type AskOptions = {
maxTokens?: number;
signal?: AbortSignal;
onDelta?: (chunk: string) => void;
+ // Optional capture of the completed call (finish reason, usage, payloads) for
+ // the debug export. Never receives the API key.
+ onDebug?: (rec: DebugCall) => void;
};
// Stream a completion from the chosen provider, invoking onDelta for each text
@@ -171,6 +215,8 @@ function emit(full: string[], chunk: string, onDelta?: (c: string) => void): voi
// ─── Anthropic ───
async function askAnthropic(opts: AskOptions): Promise<string> {
+ const t0 = Date.now();
+ const model = opts.model || PROVIDERS.anthropic.defaultModel;
const res = await fetch("https://api.anthropic.com/v1/messages", {
method: "POST",
signal: opts.signal,
@@ -181,7 +227,7 @@ async function askAnthropic(opts: AskOptions): Promise<string> {
"anthropic-dangerous-direct-browser-access": "true",
},
body: JSON.stringify({
- model: opts.model || PROVIDERS.anthropic.defaultModel,
+ model,
max_tokens: opts.maxTokens ?? 1024,
system: opts.system,
messages: opts.messages.map((m) => ({ role: m.role, content: m.content })),
@@ -190,6 +236,9 @@ async function askAnthropic(opts: AskOptions): Promise<string> {
});
await ensureOk(res, "Anthropic");
const full: string[] = [];
+ let finishReason: FinishReason | undefined;
+ let rawFinish: string | undefined;
+ let usage: unknown;
for await (const data of sseLines(res, opts.signal)) {
if (!data || data === "[DONE]") continue;
let evt: unknown;
@@ -200,18 +249,40 @@ async function askAnthropic(opts: AskOptions): Promise<string> {
}
const e = evt as {
type?: string;
- delta?: { type?: string; text?: string };
+ delta?: { type?: string; text?: string; stop_reason?: string };
+ usage?: unknown;
};
if (e.type === "content_block_delta" && e.delta?.type === "text_delta") {
emit(full, e.delta.text ?? "", opts.onDelta);
+ } else if (e.type === "message_delta") {
+ // The terminal event carries stop_reason ("max_tokens" = truncated) + usage.
+ if (e.delta?.stop_reason) {
+ rawFinish = e.delta.stop_reason;
+ finishReason = normFinish(rawFinish);
+ }
+ if (e.usage) usage = e.usage;
}
}
- return full.join("");
+ const text = full.join("");
+ opts.onDebug?.({
+ phase: "answer",
+ provider: "anthropic",
+ model,
+ request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens },
+ response: text,
+ finishReason,
+ rawFinish,
+ usage,
+ ms: Date.now() - t0,
+ });
+ return text;
}
// ─── OpenAI ───
async function askOpenAI(opts: AskOptions): Promise<string> {
+ const t0 = Date.now();
+ const model = opts.model || PROVIDERS.openai.defaultModel;
const res = await fetch("https://api.openai.com/v1/chat/completions", {
method: "POST",
signal: opts.signal,
@@ -220,8 +291,9 @@ async function askOpenAI(opts: AskOptions): Promise<string> {
authorization: `Bearer ${opts.apiKey}`,
},
body: JSON.stringify({
- model: opts.model || PROVIDERS.openai.defaultModel,
+ model,
stream: true,
+ stream_options: { include_usage: true },
// Only cap when asked (the scripted search loop passes a small value);
// otherwise let the provider default so answers aren't truncated.
...(opts.maxTokens ? { max_tokens: opts.maxTokens } : {}),
@@ -233,6 +305,9 @@ async function askOpenAI(opts: AskOptions): Promise<string> {
});
await ensureOk(res, "OpenAI");
const full: string[] = [];
+ let finishReason: FinishReason | undefined;
+ let rawFinish: string | undefined;
+ let usage: unknown;
for await (const data of sseLines(res, opts.signal)) {
if (!data || data === "[DONE]") continue;
let evt: unknown;
@@ -241,16 +316,38 @@ async function askOpenAI(opts: AskOptions): Promise<string> {
} catch {
continue;
}
- const e = evt as { choices?: { delta?: { content?: string } }[] };
+ const e = evt as {
+ choices?: { delta?: { content?: string }; finish_reason?: string | null }[];
+ usage?: unknown;
+ };
const chunk = e.choices?.[0]?.delta?.content;
if (chunk) emit(full, chunk, opts.onDelta);
+ const fr = e.choices?.[0]?.finish_reason;
+ if (fr) {
+ rawFinish = fr;
+ finishReason = normFinish(fr);
+ }
+ if (e.usage) usage = e.usage;
}
- return full.join("");
+ const text = full.join("");
+ opts.onDebug?.({
+ phase: "answer",
+ provider: "openai",
+ model,
+ request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens },
+ response: text,
+ finishReason,
+ rawFinish,
+ usage,
+ ms: Date.now() - t0,
+ });
+ return text;
}
// ─── Google Gemini ───
async function askGemini(opts: AskOptions): Promise<string> {
+ const t0 = Date.now();
const model = opts.model || PROVIDERS.gemini.defaultModel;
const url =
`https://generativelanguage.googleapis.com/v1beta/models/` +
@@ -273,6 +370,10 @@ async function askGemini(opts: AskOptions): Promise<string> {
});
await ensureOk(res, "Gemini");
const full: string[] = [];
+ let finishReason: FinishReason | undefined;
+ let rawFinish: string | undefined;
+ let blockReason: string | undefined;
+ let usage: unknown;
for await (const data of sseLines(res, opts.signal)) {
if (!data) continue;
let evt: unknown;
@@ -282,10 +383,37 @@ async function askGemini(opts: AskOptions): Promise<string> {
continue;
}
const e = evt as {
- candidates?: { content?: { parts?: { text?: string }[] } }[];
+ candidates?: {
+ content?: { parts?: { text?: string }[] };
+ finishReason?: string;
+ }[];
+ promptFeedback?: { blockReason?: string };
+ usageMetadata?: unknown;
};
- const parts = e.candidates?.[0]?.content?.parts ?? [];
+ const cand = e.candidates?.[0];
+ const parts = cand?.content?.parts ?? [];
for (const p of parts) emit(full, p.text ?? "", opts.onDelta);
+ // finishReason MAX_TOKENS = truncated; SAFETY/RECITATION = blocked (often
+ // with no parts at all → an empty answer, which we must surface, not swallow).
+ if (cand?.finishReason) {
+ rawFinish = cand.finishReason;
+ finishReason = normFinish(cand.finishReason);
+ }
+ if (e.promptFeedback?.blockReason) blockReason = e.promptFeedback.blockReason;
+ if (e.usageMetadata) usage = e.usageMetadata;
}
- return full.join("");
+ const text = full.join("");
+ opts.onDebug?.({
+ phase: "answer",
+ provider: "gemini",
+ model,
+ request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens },
+ response: text,
+ finishReason,
+ rawFinish,
+ blockReason,
+ usage,
+ ms: Date.now() - t0,
+ });
+ return text;
}
diff --git a/export/app/lib/nativeTools/anthropic.ts b/export/app/lib/nativeTools/anthropic.ts
@@ -131,6 +131,7 @@ export async function anthropicGather(ctx: NativeGatherContext): Promise<void> {
},
"Anthropic",
ctx.signal,
+ { onDebug: ctx.onDebug, model: ctx.model },
);
const calls = parseAnthropicToolUses(json);
diff --git a/export/app/lib/nativeTools/gemini.ts b/export/app/lib/nativeTools/gemini.ts
@@ -139,6 +139,7 @@ export async function geminiGather(ctx: NativeGatherContext): Promise<void> {
},
"Gemini",
ctx.signal,
+ { onDebug: ctx.onDebug, model: ctx.model },
);
const calls = parseGeminiFunctionCalls(json);
diff --git a/export/app/lib/nativeTools/openai.ts b/export/app/lib/nativeTools/openai.ts
@@ -139,6 +139,7 @@ export async function openaiGather(ctx: NativeGatherContext): Promise<void> {
},
"OpenAI",
ctx.signal,
+ { onDebug: ctx.onDebug, model: ctx.model },
);
const rawMsg = (json as { choices?: { message?: unknown }[] }).choices?.[0]
diff --git a/export/app/lib/nativeTools/shared.ts b/export/app/lib/nativeTools/shared.ts
@@ -5,7 +5,7 @@
// search budget is reached). Only the request/response shapes differ per
// provider; this module holds the common types, tool text, and POST helper.
-import type { ChatMessage } from "../askProvider";
+import type { ChatMessage, DebugCall } from "../askProvider";
import { ContextTooLargeError, isContextLengthError } from "../askProvider";
export type NativeGatherContext = {
@@ -27,6 +27,9 @@ export type NativeGatherContext = {
// Upserts a section of the running report document (report mode, native only).
// When undefined the update_report tool is not offered.
runUpdateReport?: (section: string, content: string) => Promise<string>;
+ // Optional capture of each gather round (request + raw response) for the debug
+ // export. Never receives the API key.
+ onDebug?: (rec: DebugCall) => void;
signal?: AbortSignal;
};
@@ -144,8 +147,10 @@ export async function postJson(
body: unknown,
provider: string,
signal?: AbortSignal,
+ debug?: { onDebug?: (rec: DebugCall) => void; model?: string },
): Promise<unknown> {
if (signal?.aborted) throw abortError();
+ const t0 = Date.now();
const res = await fetch(url, {
method: "POST",
signal,
@@ -171,5 +176,14 @@ export async function postJson(
}
throw new Error(msg);
}
- return res.json();
+ const json = await res.json();
+ debug?.onDebug?.({
+ phase: "gather",
+ provider,
+ model: debug.model,
+ request: body,
+ response: json,
+ ms: Date.now() - t0,
+ });
+ return json;
}
diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts
@@ -12,6 +12,8 @@ import {
askOnce,
askStream,
type ChatMessage,
+ type DebugCall,
+ type FinishReason,
type Provider,
} from "./askProvider";
import { retrieve, type RetrievedVideo } from "./askRetrieval";
@@ -61,6 +63,10 @@ export type AgentEvent =
// Default search budget per turn. Bounds round-trips (and spend on the user's
// key). Round-trips ≈ searches + 1 (answer), + gather decision turns.
export const DEFAULT_BUDGET = 4;
+// Default max output tokens for the ANSWER phase. 2048 was too small for report-
+// style answers (they truncated mid-sentence); 8192 fits comfortably within
+// Gemini 2.5 Flash / GPT-4o / Claude output limits. User-overridable per turn.
+export const DEFAULT_ANSWER_TOKENS = 8192;
// Per-search result cap and overall context cap (bounds tokens sent to the model).
const PER_SEARCH_LIMIT = 8;
const MAX_CONTEXT_VIDEOS = 15;
@@ -140,6 +146,7 @@ async function scriptedGather(
messages: convo,
maxTokens: 120,
signal: ctx.signal,
+ onDebug: ctx.onDebug,
});
const decision = parseScriptedDecision(reply);
if (decision.kind === "done") return;
@@ -234,6 +241,11 @@ export type RunAskTurnOptions = {
// and the update_report tool is offered so the model keeps the report current.
report?: string;
reportMode?: boolean;
+ // Max output tokens for the answer phase (default DEFAULT_ANSWER_TOKENS).
+ maxAnswerTokens?: number;
+ // Optional capture of each provider call (finish reason, payloads) for the
+ // debug export. Never receives the API key.
+ onDebug?: (rec: DebugCall) => void;
};
export type AskTurnResult = {
@@ -244,6 +256,10 @@ export type AskTurnResult = {
// The report after this turn (possibly updated via update_report). Echoes the
// input report unchanged when report mode is off or nothing was written.
report: string;
+ // The answer's normalized finish reason and (if blocked) the raw block reason,
+ // so the UI can flag truncation / a safety block instead of a silent stop.
+ finishReason?: FinishReason;
+ blockReason?: string;
};
export async function runAskTurn(
@@ -263,10 +279,24 @@ export async function runAskTurn(
} = opts;
const budget = opts.budget ?? DEFAULT_BUDGET;
const reportMode = opts.reportMode === true;
+ const answerTokens = opts.maxAnswerTokens ?? DEFAULT_ANSWER_TOKENS;
// The running report — mutated in place by the update_report executor below and
// returned in the result so the caller can persist it.
let report = opts.report ?? "";
+ // Capture the answer phase's finish reason (+ any block reason) so the result
+ // can flag truncation/blocking instead of a silent stop; also forward every
+ // provider call to the debug sink for the export.
+ let answerFinish: FinishReason | undefined;
+ let answerBlock: string | undefined;
+ const onDebug = (rec: DebugCall) => {
+ if (rec.phase === "answer") {
+ answerFinish = rec.finishReason;
+ answerBlock = rec.blockReason;
+ }
+ opts.onDebug?.(rec);
+ };
+
// Report mode COMPACTION: instead of replaying every prior Q&A turn as text,
// send a single "report so far" assistant message (empty report → no history at
// all). The cumulative excerpt pool still flows into the grounding unchanged —
@@ -296,16 +326,20 @@ export async function runAskTurn(
model,
system: answerSystemPrompt(aliases),
messages: [...history, { role: "user", content: groundedContent }],
- maxTokens: 2048,
+ maxTokens: answerTokens,
signal,
onDelta: (chunk) => onEvent({ type: "delta", text: chunk }),
+ onDebug,
});
return {
answer,
videos: pv,
groundedContent,
- truncated: opts.precomputedGrounding.truncated ?? false,
+ truncated:
+ (opts.precomputedGrounding.truncated ?? false) || answerFinish === "length",
report,
+ finishReason: answerFinish,
+ blockReason: answerBlock,
};
}
@@ -414,6 +448,7 @@ export async function runAskTurn(
question,
budget,
runSearch,
+ onDebug,
signal,
};
@@ -480,10 +515,19 @@ export async function runAskTurn(
model,
system: answerSystemPrompt(aliases),
messages: [...answerHistory, { role: "user", content: groundedContent }],
- maxTokens: 2048,
+ maxTokens: answerTokens,
signal,
onDelta: (chunk) => onEvent({ type: "delta", text: chunk }),
+ onDebug,
});
- return { answer, videos: finalVideos, groundedContent, truncated, report };
+ return {
+ answer,
+ videos: finalVideos,
+ groundedContent,
+ truncated: truncated || answerFinish === "length",
+ report,
+ finishReason: answerFinish,
+ blockReason: answerBlock,
+ };
}
diff --git a/export/e2e/ask-chat.spec.ts b/export/e2e/ask-chat.spec.ts
@@ -1,3 +1,4 @@
+import { readFileSync } from "node:fs";
import { expect, test, type Page } from "@playwright/test";
import { CHANNEL_SLUG, VIDEO_TRANSCRIPT_ONLY } from "./fixtures/data";
import { installRoutes } from "./helpers";
@@ -79,13 +80,21 @@ const CORS = {
"access-control-allow-methods": "*",
};
-function sse(text: string): string {
- return (
+function sse(text: string, stopReason?: string): string {
+ const events = [
`data: ${JSON.stringify({
type: "content_block_delta",
delta: { type: "text_delta", text },
- })}\n\n` + `data: ${JSON.stringify({ type: "message_stop" })}\n\n`
- );
+ })}\n\n`,
+ ];
+ // The terminal message_delta carries stop_reason ("max_tokens" = truncated).
+ if (stopReason) {
+ events.push(
+ `data: ${JSON.stringify({ type: "message_delta", delta: { stop_reason: stopReason } })}\n\n`,
+ );
+ }
+ events.push(`data: ${JSON.stringify({ type: "message_stop" })}\n\n`);
+ return events.join("");
}
function toolUse(name: string, input: Record<string, unknown>) {
@@ -135,6 +144,16 @@ async function mockAnthropic(page: Page) {
let text: string;
if (system.includes("Markdown")) {
text = "Here is the answer with a list:\n\n- point one [1]\n- point two";
+ // Truncation drill: a question containing "truncate" gets a cut-off answer
+ // (terminal stop_reason "max_tokens"), so tests can exercise the notice.
+ if (/truncate/i.test(lastText)) {
+ await route.fulfill({
+ status: 200,
+ headers: { ...CORS, "content-type": "text/event-stream" },
+ body: sse(text, "max_tokens"),
+ });
+ return;
+ }
} else if (lastText.includes('Results for "')) {
text = "DONE"; // we've searched once — stop
} else if (/timeline|format/i.test(lastText)) {
@@ -203,6 +222,37 @@ test.describe("ask chat", () => {
expect(page.url()).not.toContain("#");
});
+ test("a truncated answer (stop_reason max_tokens) shows the cut-off notice", async ({
+ page,
+ }) => {
+ await setup(page);
+ // The mock returns a cut-off answer (stop_reason "max_tokens") for a question
+ // containing "truncate".
+ await ask(page, "tell me about alpha and truncate it");
+ await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible();
+ await expect(page.getByText(/Cut off at the model/)).toBeVisible();
+ });
+
+ test("Export debug JSON downloads a redacted snapshot (no API key)", async ({
+ page,
+ }) => {
+ await setup(page);
+ await ask(page, "tell me about the alpha discussion");
+ await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible();
+ // The settings panel stays open after keying in; download the debug JSON.
+ const [download] = await Promise.all([
+ page.waitForEvent("download"),
+ page.getByRole("button", { name: "Download debug JSON" }).click(),
+ ]);
+ const path = await download.path();
+ const json = readFileSync(path, "utf8");
+ const obj = JSON.parse(json);
+ expect(obj.meta.provider).toBeTruthy();
+ expect(obj.messages.length).toBeGreaterThan(0);
+ // The key we typed must never appear in the export.
+ expect(json).not.toContain("sk-ant-test");
+ });
+
test("a reformat follow-up reuses context without a junk search", async ({
page,
}) => {