commit f18a3e0d531b9f930791878ef283f4eea65ead5b
parent c2591f8bcf4847863f706f56a6cac124bd0747ff
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 16 Jul 2026 16:39:57 -0400
Ask chat: pass the whole search to the AI via tiered grounding
The search->chat handoff capped at 20 videos, so the chat literally couldn't
see the rest of a result set. Now the whole set is handed off (up to 100) and
grounded in tiers:
- buildSearchHandoff carries every match (default cap 100), full excerpts for
the top ~12 and a single index snippet for the tail — bounded payload, whole
set available.
- buildTieredGrounding (askConversation) renders an index of ALL videos
(title/channel/date/hit count) + full excerpts for the top 12, with shared
[n] numbering; buildGroundedContent switches to it above 12 videos. The model
can fetch_context (or the user can "Load context") to pull excerpts for any
indexed video.
- Cap buildSeedDigest at 40 refs with an overflow note (seed sets can now be
large in expand mode).
Verified: tsc clean (export + common); +2 unit; 16 export + 7 aiHandoff unit;
13/13 ask-chat e2e green.
Note: the "chat reads a live shared active search" half of Phase 1 depends on
the Phase 0 shell and is not included here.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
6 files changed, 107 insertions(+), 10 deletions(-)
diff --git a/common/lib/aiHandoff.test.ts b/common/lib/aiHandoff.test.ts
@@ -61,6 +61,25 @@ test("buildSearchHandoff caps videos and hits, flagging truncation", () => {
assert.equal(h.truncated, true); // both video cap and per-video hit overflow
});
+test("buildSearchHandoff passes the whole set and tiers snippets (top full, tail indexed)", () => {
+ const groups = Array.from({ length: 20 }, (_, i) =>
+ group(`chan/${i}`, Array.from({ length: 5 }, (_, j) => hit(j, `h${j}`))),
+ );
+ const h = buildSearchHandoff(groups, () => undefined, ["q"], {
+ fullExcerptVideos: 3,
+ maxHitsPerVideo: 5,
+ });
+ // Default maxVideos no longer caps at 20 — the whole set is passed.
+ assert.equal(h.videos.length, 20);
+ // Top tier keeps full excerpts; the tail is indexed with a single snippet.
+ assert.equal(h.videos[0].snippets.length, 5);
+ assert.equal(h.videos[2].snippets.length, 5);
+ assert.equal(h.videos[3].snippets.length, 1);
+ assert.equal(h.videos[19].snippets.length, 1);
+ // The intentional tail trim is not flagged as truncation.
+ assert.equal(h.truncated, false);
+});
+
test("buildSearchHandoff marks truncated when the search itself was capped", () => {
const h = buildSearchHandoff([group("chan/aaa", [hit(1, "a")])], lookup, ["q"], {
searchCapped: true,
diff --git a/common/lib/aiHandoff.ts b/common/lib/aiHandoff.ts
@@ -62,9 +62,12 @@ export function hms(s: number): string {
return h > 0 ? `${h}:${mm}:${sss}` : `${m}:${sss}`;
}
-// Turn a completed search's result groups into a SearchHandoff. Bounded (default
-// 20 videos, 8 snippets each) so the grounding stays a reasonable token size;
-// `truncated` records whether anything was dropped.
+// Turn a completed search's result groups into a SearchHandoff. Passes the whole
+// result set (up to `maxVideos`) so the chat can present every match, but tiers
+// the payload: the top `fullExcerptVideos` carry full excerpts (`maxHitsPerVideo`
+// each) while the rest carry a single top snippet — enough for the chat's index,
+// with the AI able to fetch more on demand. `truncated` records whether the video
+// list itself was capped.
export function buildSearchHandoff(
groups: HandoffGroup[],
lookup: (slug: string) => HandoffSummaryRef | undefined,
@@ -72,18 +75,23 @@ export function buildSearchHandoff(
opts: {
maxVideos?: number;
maxHitsPerVideo?: number;
+ fullExcerptVideos?: number;
searchCapped?: boolean;
} = {},
): SearchHandoff {
- const maxVideos = opts.maxVideos ?? 20;
+ const maxVideos = opts.maxVideos ?? 100;
const maxHits = opts.maxHitsPerVideo ?? 8;
+ const fullExcerptVideos = opts.fullExcerptVideos ?? 12;
const shown = groups.slice(0, maxVideos);
let hitOverflow = false;
- const videos: HandoffVideo[] = shown.map((g) => {
+ const videos: HandoffVideo[] = shown.map((g, i) => {
const ref = lookup(g.slug);
- const hits = g.hits.slice(0, maxHits);
- if (g.hits.length > hits.length) hitOverflow = true;
+ const perVideoCap = i < fullExcerptVideos ? maxHits : 1;
+ const hits = g.hits.slice(0, perVideoCap);
+ // Only count dropped hits on the full-excerpt tier as "truncation" — the tail
+ // is indexed by design, and the AI/user can pull more on demand.
+ if (i < fullExcerptVideos && g.hits.length > hits.length) hitOverflow = true;
return {
key: g.slug,
videoId: ref?.id ?? g.slug,
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Handing a search to "Ask AI" now passes the *whole* result set, not just the first 20.** Previously only the top 20 videos reached the chat, so it literally couldn't see the rest. Now every match is handed off (up to a generous cap) and presented in tiers: the model gets a compact **index of all matching videos** (title, channel, date, hit count) plus **full excerpts for the top ~12** — and it can pull excerpts for any other result on demand via the existing *fetch_context* tool, or you can click **Load context** on any of them. This keeps the payload bounded while letting the assistant reason over the complete set. See `common/lib/aiHandoff.ts` (tiered `buildSearchHandoff`) and `export/app/lib/askConversation.ts` (`buildTieredGrounding`).
- **"Ask AI" chat — regenerate, edit & resend, and per-answer attribution.** You can now **Regenerate** a completed answer (reusing the excerpts it already gathered), **Edit** any earlier question to pull it back into the composer and re-ask from that point, and see **which model** produced each answer — useful when you switch providers mid-conversation. The Context panel now shows a rough **token estimate** (not just a character count) as a cost cue, a hand-typed model name survives switching providers and back within a session, and a failed "Load context" fetch now says so on that result instead of the spinner just quietly stopping. See `export/app/ask/{MessageBubble,AskChat,PinnedResultsPanel,ContextPanel,useAskChat}.tsx` and `export/app/lib/askConversation.ts`.
- **"Ask AI" chat polish — fewer dead ends, clearer feedback.** Several first-use rough edges are fixed: the provider settings panel no longer collapses out from under you the moment you start typing your API key; **Enter** now sends (with **Shift+Enter** for a new line); a message that stops or fails part-way through streaming now keeps its actions (Copy, Retry) and still shows what was searched, instead of freezing as raw text with no way forward; a request that errors mid-answer now shows *what* went wrong appended to whatever streamed, rather than silently dropping the error; an empty answer says so (with a Retry) instead of rendering nothing; the pre-answer "writing…" indicator no longer double-renders with an empty bubble; a fast double-press can no longer fire two turns at once; and the streaming answer is announced to screen readers. See `export/app/ask/{ProviderSettings,Composer,MessageBubble,useAskChat}.tsx`.
- **The "Ask AI" chat no longer balloons its own context.** A long conversation — especially one grounded in a big set of search results — used to re-send *every* previous turn's excerpts inside *every* new turn, so the context grew quadratically until answers stalled or a provider rejected the request. Excerpts are now carried forward once, in a deduplicated pool (each video's excerpts merged across turns and sent a single time), while the replayed history is just the questions and answers. Three related fixes ride along: a "context length exceeded" error from your provider is now recognised and shown as a clear, actionable message instead of being mistaken for "this model doesn't support tools" (which pointlessly retried the oversized request); a conversation that outgrows your browser's storage quota now trims its oldest turns and warns you, instead of silently failing to save so newer turns vanished on reload; and reloading the page mid-answer no longer leaves a message spinning forever — the interrupted turn shows a **Retry**. See `export/app/lib/{askConversation,searchAgent,askProvider}.ts`, `export/app/lib/nativeTools/shared.ts`, and `export/app/ask/{useAskChat.ts,AskChat.tsx}`.
diff --git a/export/app/lib/askConversation.test.ts b/export/app/lib/askConversation.test.ts
@@ -97,6 +97,24 @@ test("buildGroundedContent appends numbered excerpts, or just the question", ()
assert.match(grounded, /\[1\] "Platner interview" — Rekieta/);
});
+test("buildGroundedContent tiers a large set: index of all + excerpts for the top 12", () => {
+ const few = [video({ key: "k0", title: "T0" })];
+ assert.doesNotMatch(buildGroundedContent("q", few), /matching videos/);
+
+ const many = Array.from({ length: 15 }, (_, i) =>
+ video({ key: `k${i}`, title: `T${i}`, channel: "C", uploadDate: "20200102" }),
+ );
+ const tiered = buildGroundedContent("q", many);
+ // Index lists EVERY video (with a formatted date), numbered.
+ assert.match(tiered, /All 15 matching videos/);
+ assert.match(tiered, /\[15\] "T14" — C · 2020-01-02/);
+ // Full excerpts only for the top 12, and the excerpt numbering matches the index.
+ assert.match(tiered, /Full excerpts for the top 12/);
+ assert.match(tiered, /\[1\] "T0" — C\n {4}\[0:05\] he said/);
+ // A tail video (13th+) appears in the index but NOT as a full excerpt block.
+ assert.doesNotMatch(tiered, /\[13\] "T12" — C\n {4}\[/);
+});
+
test("renderAliasGlossary directs searching the full term + carries the note", () => {
const g = renderAliasGlossary(ALIASES);
// Directive to search the full term, not a fragment.
diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts
@@ -135,14 +135,57 @@ export function parseContext(text: string): ChatMessage[] {
return out;
}
+// Above this many grounding videos we switch to a tiered layout: an index of
+// EVERY video plus full excerpts for only the top ones. Keeps a large result set
+// bounded in tokens while still telling the model every match exists (it can
+// fetch_context — or the user can "Load context" — to pull excerpts for any).
+export const EXCERPT_VIDEOS_CAP = 12;
+
+function formatUploadDate(d: string): string {
+ const m = /^(\d{4})(\d{2})(\d{2})/.exec(d);
+ return m ? `${m[1]}-${m[2]}-${m[3]}` : d;
+}
+
+// A one-line index entry for a video: number, title, channel, date, hit count.
+function videoIndexLine(v: RetrievedVideo, n: number): string {
+ const meta = [v.channel, formatUploadDate(v.uploadDate)].filter(Boolean).join(" · ");
+ const site = v.siteTitle ? ` (${v.siteTitle})` : "";
+ const hits = v.snippets.length;
+ return `[${n}] "${v.title}" — ${meta}${site} · ${hits} excerpt${hits === 1 ? "" : "s"}`;
+}
+
+// Tiered grounding for a large result set: an index of ALL videos (so the model
+// knows the full set and can cite or drill into any) + full excerpts for the top
+// `topN`. Numbering is shared — [n] in the index is the same video as [n] in the
+// excerpts (the top videos are the first `topN`).
+export function buildTieredGrounding(
+ question: string,
+ videos: RetrievedVideo[],
+ topN: number = EXCERPT_VIDEOS_CAP,
+): string {
+ const index = videos.map((v, i) => videoIndexLine(v, i + 1)).join("\n");
+ const top = videos.slice(0, topN);
+ return (
+ `${question}\n\n---\n` +
+ `All ${videos.length} matching videos (cite any by its number; use the ` +
+ `fetch_context tool — or ask me to "Load context" — to read more of one):\n` +
+ `${index}\n\n` +
+ `Full excerpts for the top ${top.length} (cite by number):\n${buildContext(top)}`
+ );
+}
+
// Assemble the grounded user content for a turn: the question plus this turn's
// retrieved excerpts (numbered for citation). No excerpts → just the question,
-// so a no-search follow-up leans on excerpts already in earlier turns.
+// so a no-search follow-up leans on excerpts already in earlier turns. A large
+// set switches to the tiered index+excerpts layout.
export function buildGroundedContent(
question: string,
videos: RetrievedVideo[],
): string {
if (videos.length === 0) return question;
+ if (videos.length > EXCERPT_VIDEOS_CAP) {
+ return buildTieredGrounding(question, videos);
+ }
return (
`${question}\n\n---\n` +
`Transcript excerpts you may cite (by number):\n${buildContext(videos)}`
diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts
@@ -169,8 +169,11 @@ export function formatResultsForModel(videos: RetrievedVideo[]): string {
// Tell the model about the pinned results it's grounded in (expand mode), each
// with the `ref` it passes to fetch_context and the moments that matched — so it
// can reason about them and read more around any moment.
+const SEED_DIGEST_CAP = 40;
+
export function buildSeedDigest(videos: RetrievedVideo[]): string {
- const lines = videos.map((v) => {
+ const shown = videos.slice(0, SEED_DIGEST_CAP);
+ const lines = shown.map((v) => {
const times = v.snippets
.slice(0, 6)
.map((s) => s.clock)
@@ -178,11 +181,16 @@ export function buildSeedDigest(videos: RetrievedVideo[]): string {
const site = v.siteTitle ? ` (${v.siteTitle})` : "";
return `- ref "${v.key}": "${v.title}" — ${v.channel}${site}${times ? `; matched at ${times}` : ""}`;
});
+ const overflow =
+ videos.length > shown.length
+ ? `\n…and ${videos.length - shown.length} more pinned results.`
+ : "";
return (
"PINNED RESULTS — the user handed you these search results as your starting " +
"grounding. You may search the archive for more, and you may read more of " +
"any of these around a moment:\n" +
- lines.join("\n")
+ lines.join("\n") +
+ overflow
);
}