Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a8ea49e65219f865bdec00d98011b612e4a31a00
parent 832b236b751407694eded8a48f85c7de4c8f3d1b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 19 Jul 2026 16:20:13 -0400

Ask chat: whole-corpus "sweep" — batch a search into one running report

Add a corpus sweep to /ask: read the ENTIRE matched search set in
batches, folding each batch's findings (claims, contradictions, cited
sources) into the persistent Report and discarding the raw excerpts as
it goes, so tokens stay bounded over hundreds of matches. Native-only
(inherits Report mode); a single Stop cancels the whole run and keeps
the partial report; fresh-vs-extend is explicit at the point of action.

- getFullGrounding(): the full matched set, caps lifted (SearchSessionContext)
- chunk / accumulationSystemPrompt / buildAccumulationContent (askConversation)
- runReportChunk: lean report-only gather, budget 0, fetch_context drill (searchAgent)
- runAccumulationReport driver + sweep UI (useAskChat, PinnedResultsPanel, ReportPanel, AskChat)
- unit + e2e coverage (askConversation.test, ask-chat.spec)

Also carries the branch's citations-open-transcript + focused-alias
answer-context work (PlayerProvider.usePlayerOptional, MessageBubble,
askRetrieval) — see the two export/CHANGELOG [Unreleased] entries.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
Mcommon/components/PlayerProvider.tsx | 7+++++++
Mcommon/components/SearchSessionContext.tsx | 37+++++++++++++++++++++++++++++++++++++
Mexport/CHANGELOG.md | 2++
Mexport/app/ask/AskChat.tsx | 27++++++++++++++++++++++++++-
Mexport/app/ask/MessageBubble.tsx | 103++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------
Mexport/app/ask/PinnedResultsPanel.tsx | 187+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/ask/ReportPanel.tsx | 16++++++++++++----
Mexport/app/ask/useAskChat.ts | 184++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mexport/app/lib/askConversation.test.ts | 88+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mexport/app/lib/askConversation.ts | 88++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Mexport/app/lib/askRetrieval.test.ts | 22+++++++++++++++++++---
Mexport/app/lib/askRetrieval.ts | 25++++++++++++++++++++-----
Mexport/app/lib/searchAgent.ts | 181+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mexport/e2e/ask-chat.spec.ts | 229+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
14 files changed, 1143 insertions(+), 53 deletions(-)

diff --git a/common/components/PlayerProvider.tsx b/common/components/PlayerProvider.tsx @@ -151,6 +151,13 @@ export function usePlayer(): PlayerState { return v; } +// Like usePlayer, but returns null instead of throwing when no PlayerProvider is +// mounted. The hub /ask (AskHub) has no player, so consumers there (e.g. citation +// opening in the chat) fall back to a non-modal behavior when this is null. +export function usePlayerOptional(): PlayerState | null { + return useContext(Ctx); +} + // Subscribe to the high-frequency time/playing state separately from the // rest of the player state. Components that only need to know "what time // is it now" should use this hook so re-renders triggered by 4 Hz progress diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx @@ -153,6 +153,12 @@ export const DEFAULT_MAX_HITS = 500; export const DEFAULT_FETCH_CONCURRENCY = 6; export const DEFAULT_FLUSH_INTERVAL_MS = 120; +// Whole-corpus sweep: hit excerpts carried per video when exposing the FULL +// matched set (getFullGrounding). Small so a large set stays token-bounded, but +// enough to reason over each video within its batch (vs. liveGrounding's tiered +// single-snippet tail). +export const SWEEP_HITS_PER_VIDEO = 4; + // Canonical hash of a freshly-constructed empty root — used to disable // "Reset layers" when the draft is already at the empty default. Module-level // so we don't reallocate a node every render; `canonicalHash` strips IDs so @@ -1425,6 +1431,35 @@ function useSearchSessionState() { return handoff.videos.length ? handoff : null; }, [hasActiveQuery, resultGroups, leavesById, summaries, capped]); + // The FULL matched set as chat grounding — every discovered video with a few + // hit excerpts each, caps lifted so nothing is dropped or thinned (unlike + // liveGrounding's tiered 100-video/top-12 payload). Lazy (a function, not a + // memo) because the full set can be large and is only needed when the user + // launches a whole-corpus "sweep", not on every keystroke. Bounded only by the + // engine's hit cap (~400–500 videos); `truncated` flags when that cap bit. + const getFullGrounding = useCallback((): SearchHandoff | null => { + if (!hasActiveQuery || resultGroups.length === 0) return null; + const queries = Array.from(leavesById.values()) + .map((l) => l.query.trim()) + .filter(Boolean); + const bySlug = new Map(summaries.map((s) => [s.slug, s])); + const handoff = buildSearchHandoff( + resultGroups, + (slug) => { + const s = bySlug.get(slug); + return s ? { id: s.id, webpageUrl: s.webpageUrl } : undefined; + }, + queries, + { + maxVideos: resultGroups.length, + fullExcerptVideos: resultGroups.length, + maxHitsPerVideo: SWEEP_HITS_PER_VIDEO, + searchCapped: capped, + }, + ); + return handoff.videos.length ? handoff : null; + }, [hasActiveQuery, resultGroups, leavesById, summaries, capped]); + const [resultsCopied, setResultsCopied] = useState(false); const resultsCopiedResetRef = useRef<number | null>(null); const copyResultsContext = useCallback(async () => { @@ -1570,6 +1605,8 @@ function useSearchSessionState() { // ── Results → AI ── // The current search as chat grounding (read live by /ask; no hand-off). liveGrounding, + // The FULL matched set (lazy) — for the /ask whole-corpus sweep. + getFullGrounding, copyResultsContext, resultsCopied, }; diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -1,6 +1,8 @@ # Changelog ## [Unreleased] +- **"Ask AI" chat — sweep a whole search into one report, in batches.** A single question can only reason over ~15 videos, and a big multi-channel search (e.g. *k-cup* → hundreds of matches) leaves the long tail unread. The pinned-results panel now has a **whole-corpus report** action: it reads the **entire** matched set in batches, folding each batch's findings — claims, contradictions with earlier claims, and their cited sources — into a persistent Report and **discarding each batch's raw excerpts** as it goes, so tokens stay bounded no matter how large the result set. Type what to focus on in the composer (e.g. *"major contradictions"*, or leave it blank for *key claims & contradictions*) and click **Build report from all N results**; a determinate progress strip counts the batches while the Report panel's document visibly grows below. When a report already exists the action splits into **Start new report** (replaces it) and **Add to report** (folds these in) so a replacement is never implicit. The model may still drill into a thin line via *fetch_context* mid-sweep, and a single **Stop** cancels the whole run while keeping the partial report. Tool-capable providers only (like Report mode); on the Scripted transport the control is disabled with a hint. See `common/components/SearchSessionContext.tsx` (`getFullGrounding`), `export/app/lib/{askConversation,searchAgent}.ts` (`chunk`/`accumulationSystemPrompt`/`buildAccumulationContent`, `runReportChunk`), `export/app/ask/{useAskChat,PinnedResultsPanel,ReportPanel,AskChat}.tsx`, and `export/e2e/ask-chat.spec.ts`. +- **"Ask AI" chat — citations now open the transcript, and the answer reads your aliases in context.** Clicking an inline `[1]`, `[2]`… citation (or any timestamp in the source list beneath an answer) now **opens the transcript/video modal seeked to that line** — the same modal a search-result hit opens — instead of just scrolling to the source. It opens as a sibling overlay on the same `/ask` route (via `replaceState`, no navigation), so dismissing it (✕ / Escape) drops you back into the chat exactly where you were. On the federated hub chat (no player) citations keep the old scroll-to-source behaviour. Separately, the answer prompt now gets a **focused alias block naming only the terms a search actually used** (e.g. a search whose alias is *Cake Cups* → *k-cups*), so the model reads mis-transcribed spellings in the excerpts as the same term — the gather phase still sees the full glossary so it can discover terms it hasn't searched yet. See `common/components/PlayerProvider.tsx` (`usePlayerOptional`), `export/app/ask/{MessageBubble,AskChat}.tsx`, `export/app/lib/{askRetrieval,searchAgent,askConversation}.ts`, and `export/e2e/ask-chat.spec.ts`. - **"Ask AI" chat — long answers no longer silently truncate, plus a debug export.** Long answers (especially reports on the free Gemini tier) used to get cut off mid-sentence with no indication, because the answer was hard-capped at 2048 output tokens and the app ignored the provider's "why did it stop" signal. Now: the answer budget defaults to **8192** and there's a **Max answer length** setting to push it higher; when a model *does* hit its output limit the answer shows a clear **"⚠ Cut off at the model's output limit"** notice (with a nudge to raise the limit or use Report mode); and a Gemini response that comes back empty because it was safety-blocked now says so instead of rendering a blank bubble. For troubleshooting, provider settings gain **Download / Copy debug JSON** — a redacted snapshot of the conversation (messages, phases, finish reasons, report, settings) plus the most recent raw API calls (request, response, finish reason, token usage); **your API key is never included**. See `export/app/lib/{askProvider,searchAgent,askDebug}.ts`, `export/app/lib/nativeTools/shared.ts`, and `export/app/ask/{useAskChat,ProviderSettings,MessageBubble}.tsx`. - **"Ask AI" chat — clickable citations and smaller niceties.** The `[1]`, `[2]`… citation markers in an answer are now **clickable**: click one to jump straight to that source in the list below (a smooth in-page scroll with a brief highlight — no new tab, no navigation). Also: a suggestion chip now **focuses the composer** when you pick it (so you can tweak and press Enter), and the answer **Copy** button now says "Copy failed" when the browser blocks clipboard access (e.g. on an insecure origin) instead of silently doing nothing. See `common/components/Markdown.tsx` (an optional `linkComponent`), `export/app/ask/{MessageBubble,Composer,AskChat}.tsx`. - **"Ask AI" chat — an opt-in Report mode for long research sessions.** Turn on **Report mode** (in the new **Report** panel; tool-capable providers only) and the assistant maintains a persistent Markdown **report** document — a running canvas it updates via an `update_report` tool as you keep chatting, upserting well-titled sections that persist across turns. Crucially, while it's on the conversation is **compacted into the report** instead of replaying every prior question and answer: each turn sends a single *"report so far"* summary plus the usual deduplicated excerpt pool, so a long back-and-forth stays bounded in tokens instead of growing every turn. The report renders live in the panel (with an *updating…* shimmer while a turn writes to it) and is saved with the conversation, so it survives reloads; *New chat* clears it. On the Scripted transport there's no tool to drive it, so the toggle is disabled with a hint. See `export/app/lib/nativeTools/{shared,anthropic,openai,gemini}.ts` (the `update_report` tool), `export/app/lib/askConversation.ts` (`applyReportPatch`), `export/app/lib/searchAgent.ts` (compaction + executor), `export/app/ask/{useAskChat.ts,ReportPanel.tsx,AskChat.tsx}`, and `export/e2e/ask-chat.spec.ts`. diff --git a/export/app/ask/AskChat.tsx b/export/app/ask/AskChat.tsx @@ -2,7 +2,8 @@ import { useEffect, useMemo, useRef, useState } from "react"; import { ArrowDownIcon, PlusIcon } from "lucide-react"; -import { useAskChat } from "./useAskChat"; +import { usePlayerOptional } from "yt-dlp-transcript-common/components/PlayerProvider"; +import { DEFAULT_SWEEP_DIRECTIVE, useAskChat } from "./useAskChat"; import { ProviderSettings } from "./ProviderSettings"; import { ContextPanel } from "./ContextPanel"; import { ReportPanel } from "./ReportPanel"; @@ -24,6 +25,17 @@ export default function AskChat() { markdownOn, } = s; + // Opening a citation seeks the shared transcript modal to the cited line. It + // writes ?v=&t=&vm= via replaceState on this same /ask route (no navigation, + // no history push) and doesn't touch `messages`, so the bottom-scroll effect + // never fires and the chat scroll position survives — dismissing the modal + // returns the reader to exactly where they were. Null in hub /ask (no player). + const player = usePlayerOptional(); + const onOpenCitation = player + ? (slug: string, seconds?: number) => + player.openTranscript(slug, seconds, { mode: "transcript" }) + : undefined; + const scrollRef = useRef<HTMLDivElement | null>(null); const composerRef = useRef<HTMLTextAreaElement | null>(null); const stuckRef = useRef(true); @@ -141,6 +153,17 @@ export default function AskChat() { onSetStrict={s.setStrictGrounding} onClear={s.detach} onExpandVideo={s.expandPinnedVideo} + canSweep={s.canSweep} + sweeping={s.sweeping} + sweepProgress={s.sweepProgress} + totalGroundingVideos={s.totalGroundingVideos} + sweepCapped={s.sweepCapped} + hasReport={s.report.trim() !== ""} + hasKey={!!apiKey.trim()} + directive={input} + defaultDirective={DEFAULT_SWEEP_DIRECTIVE} + onSweep={(mode) => s.sweep({ directive: input, mode })} + onStopSweep={s.stop} /> )} @@ -179,6 +202,7 @@ export default function AskChat() { available={s.reportModeAvailable} updating={s.reportUpdating} busy={busy} + sweeping={s.sweeping} /> <div className="relative"> @@ -233,6 +257,7 @@ export default function AskChat() { onEdit={ m.role === "user" && !busy ? () => s.editUserMessage(i) : undefined } + onOpenCitation={onOpenCitation} /> )) )} diff --git a/export/app/ask/MessageBubble.tsx b/export/app/ask/MessageBubble.tsx @@ -4,6 +4,7 @@ import { useId, useState, type AnchorHTMLAttributes } from "react"; import { CheckIcon, CopyIcon, PencilIcon, RotateCwIcon } from "lucide-react"; import { Markdown } from "yt-dlp-transcript-common/components/Markdown"; import type { UiMessage } from "../lib/askConversation"; +import type { RetrievedVideo } from "../lib/askRetrieval"; import { PipelineStatus } from "./PipelineStatus"; // Turn `[n]` citation markers in an answer into links to the matching source in @@ -24,38 +25,64 @@ function linkifyCitations(md: string, maxN: number, cid: string): string { .join(""); } -// Link renderer for answer Markdown: an in-page `#cite-…` anchor scrolls to and -// briefly highlights its source (no new tab); any other link opens externally. -function CitationLink({ href, children, ...rest }: AnchorHTMLAttributes<HTMLAnchorElement>) { - if (typeof href === "string" && href.startsWith("#cite-")) { +// Scroll to and briefly highlight the source `<li>` for citation `n` — the +// fallback when there's no player (hub /ask) or the source has no timestamp. +function scrollToSource(cid: string, n: number) { + const el = document.getElementById(`cite-${cid}-${n}`); + if (!el) return; + el.scrollIntoView({ behavior: "smooth", block: "center" }); + el.classList.add("bg-brand-soft"); + setTimeout(() => el.classList.remove("bg-brand-soft"), 900); +} + +// Build the link renderer for answer Markdown, closing over this message's +// sources + citation-scope id + the (optional) modal opener. A `#cite-<cid>-<n>` +// anchor opens the transcript modal at the cited video's first matched line when +// a player is available; otherwise it scrolls to the source below. Any other +// link opens externally. +function makeCitationLink( + cid: string, + sources: RetrievedVideo[], + onOpenCitation?: (slug: string, seconds?: number) => void, +) { + return function CitationLink({ + href, + children, + ...rest + }: AnchorHTMLAttributes<HTMLAnchorElement>) { + if (typeof href === "string" && href.startsWith("#cite-")) { + return ( + <a + href={href} + onClick={(e) => { + e.preventDefault(); + const n = parseInt(href.match(/-(\d+)$/)?.[1] ?? "", 10); + const src = Number.isFinite(n) ? sources[n - 1] : undefined; + const seconds = src?.snippets[0]?.seconds; + if (onOpenCitation && src && typeof seconds === "number") { + onOpenCitation(src.key, seconds); + } else { + scrollToSource(cid, n); + } + }} + className="font-mono text-brand no-underline hover:underline" + > + {children} + </a> + ); + } return ( <a href={href} - onClick={(e) => { - e.preventDefault(); - const el = document.getElementById(href.slice(1)); - if (!el) return; - el.scrollIntoView({ behavior: "smooth", block: "center" }); - el.classList.add("bg-brand-soft"); - setTimeout(() => el.classList.remove("bg-brand-soft"), 900); - }} - className="font-mono text-brand no-underline hover:underline" + target="_blank" + rel="noopener noreferrer" + className="text-brand underline decoration-brand/40 hover:decoration-brand" + {...rest} > {children} </a> ); - } - return ( - <a - href={href} - target="_blank" - rel="noopener noreferrer" - className="text-brand underline decoration-brand/40 hover:decoration-brand" - {...rest} - > - {children} - </a> - ); + }; } function CopyButton({ text }: { text: string }) { @@ -105,11 +132,15 @@ export function MessageBubble({ markdownOn, onRetry, onEdit, + onOpenCitation, }: { message: UiMessage; markdownOn: boolean; onRetry?: () => void; onEdit?: () => void; + // Open the transcript modal seeked to a cited line. Undefined in hub /ask + // (no PlayerProvider), where citations fall back to scroll-to-source. + onOpenCitation?: (slug: string, seconds?: number) => void; }) { const isUser = message.role === "user"; // Only true streaming shows the caret bubble; the pre-token "answering" state is @@ -134,7 +165,9 @@ export function MessageBubble({ message.phase === "done" && message.content ? "Regenerate" : "Retry"; // Stable per-message id so citation anchors don't collide across messages. const cid = useId().replace(/:/g, ""); - const sourceCount = message.sources?.length ?? 0; + const sources = message.sources ?? []; + const sourceCount = sources.length; + const CitationLink = makeCitationLink(cid, sources, onOpenCitation); return ( <div className="flex flex-col gap-2 animate-in fade-in slide-in-from-bottom-2 motion-reduce:animate-none"> @@ -256,7 +289,23 @@ export function MessageBubble({ <span className="text-muted-foreground/70"> — {s.channel} {s.siteTitle ? ` · ${s.siteTitle}` : ""} ·{" "} - {s.snippets.map((sn) => sn.clock).join(", ")} + {s.snippets.map((sn, sni) => ( + <span key={sn.seconds}> + {sni > 0 ? ", " : ""} + {onOpenCitation ? ( + <button + type="button" + onClick={() => onOpenCitation(s.key, sn.seconds)} + title="Open the transcript at this moment" + className="font-mono text-brand transition-colors hover:underline" + > + {sn.clock} + </button> + ) : ( + sn.clock + )} + </span> + ))} </span> </li> ))} diff --git a/export/app/ask/PinnedResultsPanel.tsx b/export/app/ask/PinnedResultsPanel.tsx @@ -6,6 +6,8 @@ import { Loader2Icon, PinIcon, PlusIcon, + SquareIcon, + TelescopeIcon, XIcon, } from "lucide-react"; import type { SearchHandoff } from "yt-dlp-transcript-common/lib/aiHandoff"; @@ -23,6 +25,18 @@ export function PinnedResultsPanel({ onSetStrict, onClear, onExpandVideo, + // ── Whole-corpus sweep ── + canSweep, + sweeping, + sweepProgress, + totalGroundingVideos, + sweepCapped, + hasReport, + hasKey, + directive, + defaultDirective, + onSweep, + onStopSweep, }: { pinned: SearchHandoff; strictGrounding: boolean; @@ -32,9 +46,34 @@ export function PinnedResultsPanel({ onSetStrict: (on: boolean) => void; onClear: () => void; onExpandVideo: (key: string, aroundSeconds?: number) => void; + // The sweep can run at all (native-tool-capable provider). + canSweep: boolean; + // A sweep is in flight. + sweeping: boolean; + // Per-batch progress while sweeping. + sweepProgress: { done: number; total: number; sections: number }; + // Full matched-set size (before caps) + whether the search itself was capped. + totalGroundingVideos: number; + sweepCapped: boolean; + // A report already exists (→ fresh-vs-extend split instead of one button). + hasReport: boolean; + // An API key is set (else the sweep button is disabled). + hasKey: boolean; + // The composer text used as the report directive, and the empty-box fallback. + directive: string; + defaultDirective: string; + onSweep: (mode: "fresh" | "extend") => void; + onStopSweep: () => void; }) { const [showList, setShowList] = useState(false); const n = pinned.videos.length; + // The sweep reads the WHOLE matched set (which can exceed the ~100 pinned here). + const sweepN = totalGroundingVideos || n; + const focus = directive.trim(); + const pct = + sweepProgress.total > 0 + ? Math.round((sweepProgress.done / sweepProgress.total) * 100) + : 0; return ( <div className="flex flex-col gap-3 rounded-lg border border-brand/40 bg-brand-soft/40 px-4 py-3"> @@ -155,6 +194,154 @@ export function PinnedResultsPanel({ what the assistant reads. </p> </div> + + {/* ── Whole-corpus report (sweep) — a heavier, corpus-scale action than a + normal turn: it reads EVERY match in batches, folding findings into one + persistent report. Its own labeled zone at the panel's foot. ── */} + <div className="flex flex-col gap-2 border-t border-border pt-3"> + {sweeping ? ( + // Signature moment: a determinate batch-progress strip. As the driver + // setReport()s each batch, the Report panel's document grows below while + // this bar climbs — chunked reduction made legible. + <> + <div className="flex items-center gap-2"> + <TelescopeIcon className="size-4 shrink-0 text-brand" /> + <span className="text-sm font-medium text-foreground"> + Sweeping {sweepN} result{sweepN === 1 ? "" : "s"}… + </span> + <button + type="button" + onClick={onStopSweep} + aria-label="Stop the sweep" + className="ml-auto inline-flex items-center gap-1 rounded-md border border-border px-2 py-1 text-xs text-muted-foreground transition-colors hover:text-foreground" + > + <SquareIcon className="size-3" /> Stop + </button> + </div> + <div + className="h-1.5 w-full overflow-hidden rounded-full bg-border" + role="progressbar" + aria-valuenow={sweepProgress.done} + aria-valuemin={0} + aria-valuemax={sweepProgress.total} + > + <div + className="h-full rounded-full bg-brand transition-[width] duration-500 ease-out motion-reduce:transition-none" + style={{ width: `${pct}%` }} + /> + </div> + <span className="flex items-center gap-1.5 text-xs text-muted-foreground"> + <Loader2Icon className="size-3 animate-spin motion-reduce:animate-none" /> + Reading batch {Math.min(sweepProgress.done + 1, sweepProgress.total)}{" "} + of {sweepProgress.total} + {sweepProgress.sections > 0 + ? ` · ${sweepProgress.sections} section${sweepProgress.sections === 1 ? "" : "s"} written` + : ""} + </span> + </> + ) : ( + <> + <div className="flex items-start gap-2"> + <TelescopeIcon className="mt-0.5 size-4 shrink-0 text-brand" /> + <div className="flex min-w-0 flex-col gap-0.5"> + <span className="text-[11px] font-semibold uppercase tracking-wide text-muted-foreground"> + Whole-corpus report + </span> + {!hasReport && ( + <span className="text-xs text-muted-foreground"> + Reads all {sweepN} match{sweepN === 1 ? "" : "es"} in batches, + folding findings into one report. + </span> + )} + </div> + </div> + + {/* Directive = the composer text, shown inline so the scope is visible. */} + <p className="text-xs text-muted-foreground"> + Report on:{" "} + {focus ? ( + <span className="text-foreground">{focus}</span> + ) : ( + <> + <span className="text-foreground">{defaultDirective}</span> + <span className="text-muted-foreground/70"> + {" "} + — type in the box below to focus it. + </span> + </> + )} + </p> + + {!canSweep ? ( + <> + <button + type="button" + disabled + className="inline-flex w-fit items-center gap-1.5 rounded-md bg-primary px-3 py-1.5 text-xs font-medium text-primary-foreground opacity-50" + > + <TelescopeIcon className="size-3.5" /> Build report from all{" "} + {sweepN} result{sweepN === 1 ? "" : "s"} + </button> + <p className="text-xs text-warning"> + Needs a tool-capable provider — switch the search mode off + Scripted. + </p> + </> + ) : !hasReport ? ( + // Nothing to overwrite → a single primary button. + <button + type="button" + onClick={() => onSweep("fresh")} + disabled={busy || !hasKey} + className="inline-flex w-fit items-center gap-1.5 rounded-md bg-primary px-3 py-1.5 text-xs font-medium text-primary-foreground transition-colors hover:bg-brand-strong disabled:opacity-50" + > + <TelescopeIcon className="size-3.5" /> + {sweepCapped + ? `Sweep the first ${sweepN} results` + : `Build report from all ${sweepN} result${sweepN === 1 ? "" : "s"}`} + </button> + ) : ( + // A report exists → split into two weighted actions, each captioned + // so replacement is never implicit. + <div className="flex flex-wrap gap-x-4 gap-y-2"> + <div className="flex flex-col gap-0.5"> + <button + type="button" + onClick={() => onSweep("fresh")} + disabled={busy || !hasKey} + className="inline-flex w-fit items-center gap-1.5 rounded-md bg-primary px-3 py-1.5 text-xs font-medium text-primary-foreground transition-colors hover:bg-brand-strong disabled:opacity-50" + > + <TelescopeIcon className="size-3.5" /> Start new report + </button> + <span className="text-xs text-muted-foreground/70"> + Replaces the current report. + </span> + </div> + <div className="flex flex-col gap-0.5"> + <button + type="button" + onClick={() => onSweep("extend")} + disabled={busy || !hasKey} + className="inline-flex w-fit items-center gap-1.5 rounded-md border border-border px-3 py-1.5 text-xs font-medium text-foreground transition-colors hover:border-brand disabled:opacity-50" + > + <PlusIcon className="size-3.5" /> Add to report + </button> + <span className="text-xs text-muted-foreground/70"> + Keeps it and folds these in. + </span> + </div> + </div> + )} + + {canSweep && sweepCapped && ( + <p className="text-xs text-muted-foreground/70"> + The search matched more than we can sweep — narrow it for full + coverage. + </p> + )} + </> + )} + </div> </div> ); } diff --git a/export/app/ask/ReportPanel.tsx b/export/app/ask/ReportPanel.tsx @@ -17,6 +17,7 @@ export function ReportPanel({ available, updating, busy, + sweeping = false, }: { report: string; reportMode: boolean; @@ -24,20 +25,27 @@ export function ReportPanel({ available: boolean; updating: boolean; busy: boolean; + // A whole-corpus sweep is folding batches into the report — force the panel + // open so its document is visibly growing, and keep the "updating" shimmer on. + sweeping?: boolean; }) { const [open, setOpen] = useState(false); const hasReport = report.trim() !== ""; + // While a sweep runs, keep the panel expanded (so its document is visibly + // growing) regardless of the user's toggle. + const expanded = open || sweeping; + const busyWriting = updating || sweeping; return ( <details className="rounded-lg border border-border bg-card/40" - open={open} + open={expanded} onToggle={(e) => setOpen((e.currentTarget as HTMLDetailsElement).open)} > <summary className="flex cursor-pointer items-center gap-2 px-4 py-2.5 text-sm font-medium text-foreground"> <FileTextIcon className="size-4 shrink-0 text-brand" /> Report - {updating && ( + {busyWriting && ( <span className="inline-flex items-center gap-1 text-xs font-normal text-brand"> <Loader2Icon className="size-3 animate-spin motion-reduce:animate-none" /> updating… @@ -64,7 +72,7 @@ export function ReportPanel({ </label> </summary> - {open && ( + {expanded && ( <div className="flex flex-col gap-3 border-t border-border px-4 py-3"> {!available && ( <p className="text-xs text-warning"> @@ -77,7 +85,7 @@ export function ReportPanel({ aria-live="polite" className={ "rounded-md border border-border bg-background px-3 py-2 text-sm text-foreground transition-opacity " + - (updating ? "opacity-60" : "opacity-100") + (busyWriting ? "opacity-60" : "opacity-100") } > <Markdown className="leading-relaxed">{report}</Markdown> diff --git a/export/app/ask/useAskChat.ts b/export/app/ask/useAskChat.ts @@ -14,6 +14,7 @@ import { PROVIDERS, type DebugCall, type Provider } from "../lib/askProvider"; import { DEFAULT_ANSWER_TOKENS, runAskTurn, + runReportChunk, supportsNativeTools, type AgentEvent, type AgentMode, @@ -23,6 +24,7 @@ import { buildDebugExport } from "../lib/askDebug"; import { buildApiMessages, buildGroundedContent, + chunk, parseContext, serializeContext, type SearchStep, @@ -35,12 +37,34 @@ const K_SEARCHMODE = "ytdlp-tb:ai:searchmode"; const K_MARKDOWN = "ytdlp-tb:ai:md"; const K_CONVO = "ytdlp-tb:ai:conversation"; const K_MAXANSWER = "ytdlp-tb:ai:maxanswer"; +// Optional test/power-user override for the sweep batch size (see sweepChunkSize). +const K_SWEEPCHUNK = "ytdlp-tb:ai:sweepchunk"; const keyFor = (p: Provider) => `ytdlp-tb:ai:key:${p}`; const modelFor = (p: Provider) => `ytdlp-tb:ai:model:${p}`; // Bounds for the user-set "Max answer length" (output tokens). const MAX_ANSWER_MIN = 256; const MAX_ANSWER_MAX = 32768; + +// Whole-corpus sweep: how many result videos each batch folds into the report. +// Small enough to stay within a native gather's token budget per pass, large +// enough that a big result set is a handful of batches, not hundreds. +const SWEEP_CHUNK_SIZE = 10; +// Directive fallback when the composer is empty at sweep time. +export const DEFAULT_SWEEP_DIRECTIVE = "key claims & contradictions"; + +// The effective batch size — SWEEP_CHUNK_SIZE, or a localStorage override (used +// by e2e to force multiple batches over the small fixture set; also a legitimate +// power-user tuning). Clamped to [1, 100]. +function sweepChunkSize(): number { + try { + const raw = Number(localStorage.getItem(K_SWEEPCHUNK)); + if (Number.isFinite(raw) && raw >= 1) return Math.min(100, Math.floor(raw)); + } catch { + /* ignore */ + } + return SWEEP_CHUNK_SIZE; +} // How many recent provider calls to retain for the debug export (ring buffer). const DEBUG_TRACE_CAP = 8; @@ -111,7 +135,9 @@ export function useAskChat() { const corpusError = summariesState.error?.message ?? null; // The live "active search" this workspace shares — the chat auto-grounds in it. - const { liveGrounding } = useSearchSession(); + // getFullGrounding lazily builds the WHOLE matched set (caps lifted) for the + // corpus sweep. + const { liveGrounding, getFullGrounding } = useSearchSession(); const [messages, setMessages] = useState<UiMessage[]>([]); // A human-pruned context that seeds the conversation: prepended to the replayed @@ -135,6 +161,15 @@ export function useAskChat() { // True while a turn is writing to the report (drives the panel shimmer); // cleared when the turn ends. const [reportUpdating, setReportUpdating] = useState(false); + // Whole-corpus sweep: true while a sweep runs, plus its per-batch progress + // ({done, total} batches + running section-write count) driving the progress + // strip in the pinned panel and the ReportPanel shimmer. + const [sweeping, setSweeping] = useState(false); + const [sweepProgress, setSweepProgress] = useState({ + done: 0, + total: 0, + sections: 0, + }); // Max output tokens for the answer phase (user-tunable; default 8192). Raising // it lets long reports finish instead of truncating at the model's output cap. const [maxAnswerTokens, setMaxAnswerTokensState] = useState(DEFAULT_ANSWER_TOKENS); @@ -666,6 +701,143 @@ export function useAskChat() { }); }, [input, busy, apiKey, summariesReady, provider, model, remember, runTurn]); + // Whole-corpus "sweep": read the ENTIRE matched set in batches, folding each + // batch's findings into the running report and discarding its raw excerpts, so + // a large multi-channel result set can be reasoned over without blowing the + // context. `mode` "fresh" starts a new report (clears the current one); "extend" + // folds into the existing report. Native-only (report mode); one AbortController + // for the whole sweep, so a single Stop cancels it and keeps the partial report. + const runAccumulationReport = useCallback( + async ({ + directive, + mode, + }: { + directive: string; + mode: "fresh" | "extend"; + }) => { + if ( + busy || + sendingRef.current || + !apiKey.trim() || + !summariesReady || + !reportModeAvailable + ) { + return; + } + const full = getFullGrounding(); + if (!full || full.videos.length === 0) return; + + sendingRef.current = true; + persistKey(provider, apiKey, model, remember); + + const videos = full.videos as RetrievedVideo[]; + const batches = chunk(videos, sweepChunkSize()); + const total = batches.length; + const cleanDirective = directive.trim() || DEFAULT_SWEEP_DIRECTIVE; + const mdl = model.trim() || PROVIDERS[provider].defaultModel; + + setBusy(true); + setSweeping(true); + setSweepProgress({ done: 0, total, sections: 0 }); + // Turn the report mode on so the ReportPanel is visible and subsequent + // chat turns keep maintaining the report the sweep just built. + setReportModeState(true); + const ac = new AbortController(); + abortRef.current = ac; + + // A sweep "assistant" message tracks the sweep in the transcript; it + // resolves to a summary (or a stopped/error line) when the sweep ends. + const assistantIndex = messagesRef.current.length; + setMessages((prev) => [ + ...prev, + { + role: "assistant", + content: `Sweeping ${videos.length} result${videos.length === 1 ? "" : "s"}…`, + phase: "gathering", + searchSteps: [], + }, + ]); + + let report = mode === "fresh" ? "" : reportRef.current; + if (mode === "fresh") setReport(""); + let sections = 0; + let done = 0; + + try { + for (let i = 0; i < batches.length; i++) { + if (ac.signal.aborted) { + throw new DOMException("Aborted", "AbortError"); + } + setReportUpdating(true); + const r = await runReportChunk({ + provider, + apiKey: apiKey.trim(), + model: mdl, + directive: cleanDirective, + videos: batches[i], + index: i + 1, + count: total, + report, + aliases, + signal: ac.signal, + onEvent: (e) => { + if (e.type === "report_update") { + sections += 1; + setSweepProgress((p) => ({ ...p, sections })); + } + }, + onDebug: pushDebug, + }); + report = r.report; + setReport(report); + done = i + 1; + setSweepProgress({ done, total, sections }); + } + patchAt(assistantIndex, (m) => ({ + ...m, + phase: "done", + content: + `Built a report from ${videos.length} result${videos.length === 1 ? "" : "s"} ` + + `across ${total} batch${total === 1 ? "" : "es"}. See the Report panel.`, + })); + } catch (e) { + const err = e as Error; + if (err.name === "AbortError") { + patchAt(assistantIndex, (m) => ({ + ...m, + phase: "stopped", + content: `Stopped after ${done} of ${total} batches — the partial report is saved.`, + })); + } else { + patchAt(assistantIndex, (m) => ({ + ...m, + phase: "error", + error: true, + content: `Couldn't finish the sweep (${err.message}). The report has what completed so far.`, + })); + } + } finally { + setBusy(false); + setSweeping(false); + setReportUpdating(false); + sendingRef.current = false; + abortRef.current = null; + } + }, + [ + busy, + apiKey, + summariesReady, + reportModeAvailable, + getFullGrounding, + provider, + model, + remember, + aliases, + pushDebug, + ], + ); + // Re-run a failed assistant turn. Reuses the grounding captured before the // answer stream failed (no re-search); falls back to a full turn otherwise. const retry = useCallback( @@ -845,6 +1017,16 @@ export function useAskChat() { setReportMode, reportModeAvailable, reportUpdating, + // whole-corpus sweep + sweep: runAccumulationReport, + sweeping, + sweepProgress, + // Native-tool-capable → the sweep can run (report mode is available). + canSweep: reportModeAvailable, + // The full matched-set size (before any cap) + whether the search itself was + // capped, so the panel can say "Build from all N" vs "Sweep the first N". + totalGroundingVideos: liveGrounding?.totalVideos ?? 0, + sweepCapped: liveGrounding?.truncated ?? false, // answer length + debug export maxAnswerTokens, setMaxAnswerTokens, diff --git a/export/app/lib/askConversation.test.ts b/export/app/lib/askConversation.test.ts @@ -2,11 +2,15 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { + accumulationSystemPrompt, applyReportPatch, + buildAccumulationContent, buildApiMessages, buildGroundedContent, + chunk, collectPriorPool, renderAliasGlossary, + renderUsedAliasContext, gatherSystemPrompt, answerSystemPrompt, serializeContext, @@ -210,9 +214,89 @@ test("applyReportPatch matches the heading case-insensitively", () => { assert.doesNotMatch(after, /Old\./); }); -test("answerSystemPrompt asks for Markdown + citations and carries the glossary", () => { +test("renderUsedAliasContext lists ONLY the used aliases with their variant spellings", () => { + const c = renderUsedAliasContext(ALIASES); + // Framed for reading excerpts, not for searching. + assert.match(c, /may contain AI-transcription/); + // The used alias, its variant spellings joined, and its note. + assert.match(c, /Graham Platner: platner, plattner, platter — often mis-transcribed/); + // Nothing used → empty string; disabled aliases are dropped; deduped by id. + assert.equal(renderUsedAliasContext([]), ""); + assert.equal(renderUsedAliasContext([{ ...ALIASES[0], enabled: false }]), ""); + const dupCount = ( + renderUsedAliasContext([ALIASES[0], ALIASES[0]]).match(/Graham Platner:/g) ?? [] + ).length; + assert.equal(dupCount, 1); +}); + +test("chunk splits into fixed-size batches; last is the remainder; empty → []", () => { + assert.deepEqual(chunk([1, 2, 3, 4, 5], 2), [[1, 2], [3, 4], [5]]); + // Exact multiple → no trailing empty batch. + assert.deepEqual(chunk([1, 2, 3, 4], 2), [[1, 2], [3, 4]]); + // A single full batch when size ≥ length. + assert.deepEqual(chunk([1, 2, 3], 10), [[1, 2, 3]]); + // Empty input → no batches. + assert.deepEqual(chunk([], 3), []); + // Degenerate size ≤ 0 → one all-in batch (never an infinite loop). + assert.deepEqual(chunk([1, 2], 0), [[1, 2]]); +}); + +test("buildAccumulationContent emits the directive, a batch label, and numbered excerpts", () => { + const videos = [ + video({ key: "a", title: "First", snippets: [{ clock: "0:05", seconds: 5, text: "alpha here" }] }), + video({ key: "b", title: "Second", snippets: [{ clock: "1:00", seconds: 60, text: "beta there" }] }), + ]; + const out = buildAccumulationContent("major contradictions", videos, 3, 12); + // The directive leads. + assert.match(out, /^major contradictions/); + // The batch label names position within the sweep. + assert.match(out, /Batch 3 of 12/); + // Excerpts are numbered from [1] (restarting per batch) with their clocks. + assert.match(out, /\[1\] "First" — Rekieta/); + assert.match(out, /\[0:05\] alpha here/); + assert.match(out, /\[2\] "Second" — Rekieta/); +}); + +test("accumulationSystemPrompt carries the update_report directive + focused alias block", () => { + const p = accumulationSystemPrompt(ALIASES); + // Reduce-into-report framing, native tools named, and searching forbidden. + assert.match(p, /update_report tool/); + assert.match(p, /Do NOT search/); + assert.match(p, /fetch_context tool/); + assert.match(p, /finish tool/); + // The FOCUSED used-alias block (not the full "search the full term" glossary). + assert.match(p, /TERMS USED IN THIS SEARCH/); + assert.match(p, /Graham Platner: platner, plattner, platter/); + assert.doesNotMatch(p, /search the full term/); + // No aliases → just the base, no dangling block. + assert.doesNotMatch(accumulationSystemPrompt([]), /TERMS USED IN THIS SEARCH/); +}); + +test("applyReportPatch composes across sweep chunks: fold section A then B → both survive", () => { + // Batch 1 folds in "Claims"; batch 2 folds in "Contradictions" — the reduce + // threads the growing report, so both sections coexist afterwards. + let report = ""; + report = applyReportPatch(report, "Claims", "Alpha claims X. [1]"); + report = applyReportPatch(report, "Contradictions", "Beta disputes X. [2]"); + assert.match(report, /## Claims\n\nAlpha claims X\. \[1\]/); + assert.match(report, /## Contradictions\n\nBeta disputes X\. \[2\]/); + assert.ok(report.indexOf("## Claims") < report.indexOf("## Contradictions")); + // A later batch merging MORE into an existing section replaces just that body. + report = applyReportPatch(report, "Claims", "Alpha claims X and Y. [1][3]"); + assert.match(report, /## Claims\n\nAlpha claims X and Y\. \[1\]\[3\]/); + assert.doesNotMatch(report, /Alpha claims X\. \[1\]$/m); + // The neighbour is untouched by the update. + assert.match(report, /## Contradictions\n\nBeta disputes X\. \[2\]/); +}); + +test("answerSystemPrompt asks for Markdown + citations and carries the FOCUSED used-alias block", () => { const p = answerSystemPrompt(ALIASES); assert.match(p, /Markdown/); assert.match(p, /\[1\]/); - assert.match(p, /Graham Platner/); + // The focused block (not the full glossary) — no "search the full term" directive. + assert.match(p, /TERMS USED IN THIS SEARCH/); + assert.match(p, /Graham Platner: platner, plattner, platter/); + assert.doesNotMatch(p, /search the full term/); + // No used aliases → just the base prompt, no dangling block. + assert.doesNotMatch(answerSystemPrompt([]), /TERMS USED IN THIS SEARCH/); }); diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts @@ -286,6 +286,34 @@ function withGlossary(base: string, aliases: SearchAlias[]): string { return glossary ? `${base}\n\n${glossary}` : base; } +// A FOCUSED alias block for the answer phase: only the aliases a search actually +// used this turn, framed for reading the excerpts (not for deciding what to +// search — that's the gather phase's full glossary). It tells the model that the +// listed variant spellings in the excerpts are all the same term, so it reads a +// mis-transcribed name/word correctly (e.g. a "k-cup" search whose alias is +// "Cake Cups"). Empty string when nothing usable fired. Deduped by id. +export function renderUsedAliasContext(used: SearchAlias[]): string { + const usable: SearchAlias[] = []; + const seen = new Set<string>(); + for (const a of used) { + if (a.enabled === false || seen.has(a.id)) continue; + seen.add(a.id); + usable.push(a); + } + if (usable.length === 0) return ""; + const lines = usable.map((a) => { + const spellings = a.triggers.join(", "); + const note = a.note ? ` — ${a.note}` : ""; + return `- ${a.label}: ${spellings}${note}`; + }); + return ( + "TERMS USED IN THIS SEARCH — the excerpts below may contain AI-transcription " + + "mis-spellings of these curated terms. Treat every listed spelling as the " + + "same term when reading and citing the excerpts:\n" + + lines.join("\n") + ); +} + // System prompt for the gather (search-decision) phase. `mode` tailors the // closing instruction: native tool-calling vs. the scripted text protocol. export function gatherSystemPrompt( @@ -328,8 +356,12 @@ export function gatherSystemPrompt( return withGlossary(`${base}\n\n${tail}`, aliases); } -// System prompt for the final answer phase. -export function answerSystemPrompt(aliases: SearchAlias[]): string { +// System prompt for the final answer phase. Unlike the gather phase (which gets +// the full glossary so it can discover terms to search), this gets a FOCUSED +// block naming only the aliases the turn's searches actually used — so the model +// reads the specific mis-transcribed terms in these excerpts correctly without +// the noise of every enabled alias. +export function answerSystemPrompt(usedAliases: SearchAlias[]): string { const base = "You are answering a question about a video-transcript archive. Base your " + "answer on the transcript excerpts provided in this conversation. Excerpts " + @@ -339,5 +371,55 @@ export function answerSystemPrompt(aliases: SearchAlias[]): string { "Each excerpt shows a video title, channel, and timestamped lines. If the " + "excerpts do not contain enough to answer, say so plainly rather than " + "guessing. Format your answer in GitHub-flavored Markdown."; - return withGlossary(base, aliases); + const context = renderUsedAliasContext(usedAliases); + return context ? `${base}\n\n${context}` : base; +} + +// ─── whole-corpus sweep (map result-chunks → reduce into the running report) ─── + +// Split a list into fixed-size chunks; the last chunk may be shorter. A size ≤ 0 +// yields a single chunk with everything (or none for an empty list). Pure + +// testable — the sweep driver folds each chunk into the report in turn. +export function chunk<T>(items: T[], size: number): T[][] { + if (size <= 0) return items.length ? [items.slice()] : []; + const out: T[][] = []; + for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size)); + return out; +} + +// System prompt for one sweep chunk: the model reads a batch of transcript +// excerpts and folds findings into a running report via update_report, then +// discards the batch's raw text. Native-only (needs the update_report tool). Gets +// the FOCUSED used-alias block (like the answer phase) so mis-transcribed terms in +// the batch read correctly — NOT the full glossary (there's no searching here). +export function accumulationSystemPrompt(usedAliases: SearchAlias[]): string { + const base = + "You are building a running report by reading transcript excerpts in " + + "batches. Cross-reference this batch against the report so far. Use the " + + "update_report tool to add or merge findings — claims, contradictions with " + + "earlier claims, and their sources (title + timestamp, cited by [n]) — into " + + "well-titled sections, keeping the report the single source of truth. Do NOT " + + "search; these excerpts are your evidence, but you may call the fetch_context " + + "tool to read more transcript around a specific line when one is too thin to " + + "judge. Call the finish tool when you are done with this batch."; + const context = renderUsedAliasContext(usedAliases); + return context ? `${base}\n\n${context}` : base; +} + +// The per-chunk user message for a sweep: the report directive, a "Batch i of n" +// label, and this batch's numbered excerpts (same citation format as a grounded +// turn, reusing buildContext). Numbering restarts per batch — the raw excerpts +// are discarded once the batch folds into the report, so cross-batch [n] identity +// isn't needed. +export function buildAccumulationContent( + directive: string, + videos: RetrievedVideo[], + index: number, + count: number, +): string { + return ( + `${directive}\n\n` + + `Batch ${index} of ${count} — transcript excerpts you may cite (by number):\n` + + buildContext(videos) + ); } diff --git a/export/app/lib/askRetrieval.test.ts b/export/app/lib/askRetrieval.test.ts @@ -160,7 +160,7 @@ const PLATNER_ALIAS: SearchAlias = { }; test("buildSearchRoot: a single-word alias fires and drops the plain leaf", () => { - const root = buildSearchRoot("lolly banana", ["lolly", "banana"], [LOLI_ALIAS]); + const { root } = buildSearchRoot("lolly banana", ["lolly", "banana"], [LOLI_ALIAS]); const leaves = root.children.filter(isLeaf); const aliasLeaf = leaves.find((l) => l.id === "t#a0")!; // "lolly" matches the alias trigger → a regex leaf uses the suggestion. @@ -174,7 +174,7 @@ test("buildSearchRoot: a single-word alias fires and drops the plain leaf", () = test("buildSearchRoot: a MULTI-WORD alias fires on the full phrase (the bug)", () => { // The whole-query phrase contains "graham platner" → the regex leaf appears. - const root = buildSearchRoot( + const { root } = buildSearchRoot( "graham platner", ["graham", "platner"], [PLATNER_ALIAS], @@ -189,10 +189,26 @@ test("buildSearchRoot: a MULTI-WORD alias fires on the full phrase (the bug)", ( test("buildSearchRoot: a fragment does NOT fire a multi-word alias", () => { // Searching just "graham" can't satisfy the two-token trigger → no regex leaf. - const root = buildSearchRoot("graham", ["graham"], [PLATNER_ALIAS]); + const { root, firedT, firedM } = buildSearchRoot("graham", ["graham"], [PLATNER_ALIAS]); const leaves = root.children.filter(isLeaf); assert.ok(!leaves.some((l) => l.useRegex)); assert.ok(leaves.some((l) => l.query === "graham")); + // Nothing fired → no aliases surfaced for the answer prompt. + assert.equal(firedT.length, 0); + assert.equal(firedM.length, 0); +}); + +test("buildSearchRoot: reports the aliases that fired (for focused answer context)", () => { + const { firedT, firedM } = buildSearchRoot("lolly banana", ["lolly", "banana"], [LOLI_ALIAS]); + // The matching alias fires in both scopes (it has no scope restriction). + assert.deepEqual( + firedT.map((a) => a.id), + ["loli"], + ); + assert.deepEqual( + firedM.map((a) => a.id), + ["loli"], + ); }); test("buildContext numbers videos and indents snippets", () => { diff --git a/export/app/lib/askRetrieval.ts b/export/app/lib/askRetrieval.ts @@ -200,7 +200,7 @@ export function buildSearchRoot( query: string, keywords: string[], aliases: SearchAlias[] = [], -): ReturnType<typeof newGroup> { +): { root: ReturnType<typeof newGroup>; firedT: SearchAlias[]; firedM: SearchAlias[] } { const children: QueryNode[] = []; const firedT = aliases.length ? matchAliases(query, "transcripts", aliases) : []; const firedM = aliases.length ? matchAliases(query, "metadata", aliases) : []; @@ -240,7 +240,17 @@ export function buildSearchRoot( children.push(newLeaf({ id: `m#${i}`, query: kw, scope: "metadata", contributeHits: true })); }); - return newGroup({ op: "OR", children }); + return { root: newGroup({ op: "OR", children }), firedT, firedM }; +} + +// The distinct aliases that fired for a query across both scopes, deduped by id +// (an alias enabled in transcripts + metadata fires in both). This is what +// `retrieve` reports as `firedAliases` so the answer prompt can focus on only +// the aliases the search actually used. +function unionFired(firedT: SearchAlias[], firedM: SearchAlias[]): SearchAlias[] { + const byId = new Map<string, SearchAlias>(); + for (const a of [...firedT, ...firedM]) if (!byId.has(a.id)) byId.set(a.id, a); + return [...byId.values()]; } // Run the question through the shared search engine and return ranked videos. @@ -248,7 +258,7 @@ export function buildSearchRoot( // engine reports done; rejects with an AbortError if the signal fires first. export function retrieve( opts: RetrieveOptions, -): Promise<{ videos: RetrievedVideo[]; truncated: boolean }> { +): Promise<{ videos: RetrievedVideo[]; truncated: boolean; firedAliases: SearchAlias[] }> { const { question, summaries, signal } = opts; const aliases = opts.aliases ?? []; const keywords = extractKeywords(question); @@ -259,7 +269,7 @@ export function retrieve( return; } if (summaries.length === 0 || keywords.length === 0) { - resolve({ videos: [], truncated: false }); + resolve({ videos: [], truncated: false, firedAliases: [] }); return; } @@ -267,7 +277,11 @@ export function retrieve( // and its keyword leaves. Each term gets a transcript-cue leaf (timestamped // snippets) plus a fetch-free metadata leaf (title/channel) for cheap title // recall. contributeHits stays true so both surface citations. - const root = buildSearchRoot(question, keywords, aliases); + const { root, firedT, firedM } = buildSearchRoot(question, keywords, aliases); + // The aliases this search actually applied — surfaced so the answer prompt + // can tell the model exactly which curated terms may be mis-transcribed in + // the excerpts it's about to read. + const firedAliases = unionFired(firedT, firedM); let settled = false; const finish = (fn: () => void) => { @@ -290,6 +304,7 @@ export function retrieve( resolve({ videos: rankResults(p, summaries, opts), truncated: p.capped, + firedAliases, }), ); }, diff --git a/export/app/lib/searchAgent.ts b/export/app/lib/searchAgent.ts @@ -18,8 +18,10 @@ import { } from "./askProvider"; import { retrieve, type RetrievedVideo } from "./askRetrieval"; import { + accumulationSystemPrompt, answerSystemPrompt, applyReportPatch, + buildAccumulationContent, buildApiMessages, buildGroundedContent, collectPriorPool, @@ -34,7 +36,10 @@ import { ToolsUnavailableError, type NativeGatherContext, } from "./nativeTools/shared"; -import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; +import { + matchAliases, + type SearchAlias, +} from "yt-dlp-transcript-common/lib/searchAliases"; import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; import { fetchTranscript } from "yt-dlp-transcript-common/components/transcriptCache"; import { @@ -315,6 +320,21 @@ export async function runAskTurn( ? compactHistory() : opts.historyOverride ?? buildApiMessages(prior); + // Aliases this turn actually used. Seeded from the question itself (so a + // strict / no-search turn still gets focused context) and grown with every + // alias a search fires (accumulated in runSearch below). Deduped by id, then + // handed to the answer prompt so the model reads THIS turn's specific + // mis-transcribed terms right — without the full glossary's noise. + const usedAliases = new Map<string, SearchAlias>(); + const addUsedAliases = (list: SearchAlias[]) => { + for (const a of list) if (!usedAliases.has(a.id)) usedAliases.set(a.id, a); + }; + if (aliases.length) { + addUsedAliases(matchAliases(question, "transcripts", aliases)); + addUsedAliases(matchAliases(question, "metadata", aliases)); + } + const usedAliasList = () => [...usedAliases.values()]; + // Retry path: grounding already gathered on a prior attempt — skip gather and // go straight to (re)streaming the answer over the same excerpts. if (opts.precomputedGrounding) { @@ -324,7 +344,7 @@ export async function runAskTurn( provider, apiKey, model, - system: answerSystemPrompt(aliases), + system: answerSystemPrompt(usedAliasList()), messages: [...history, { role: "user", content: groundedContent }], maxTokens: answerTokens, signal, @@ -381,6 +401,7 @@ export async function runAskTurn( videos.set(v.key, v); freshKeys.add(v.key); } + addUsedAliases(r.firedAliases); if (r.truncated) truncated = true; queries.push(query); onEvent({ type: "search_done", query, count: r.videos.length }); @@ -513,7 +534,7 @@ export async function runAskTurn( provider, apiKey, model, - system: answerSystemPrompt(aliases), + system: answerSystemPrompt(usedAliasList()), messages: [...answerHistory, { role: "user", content: groundedContent }], maxTokens: answerTokens, signal, @@ -531,3 +552,157 @@ export async function runAskTurn( blockReason: answerBlock, }; } + +// ─── whole-corpus sweep: fold ONE chunk of results into the running report ─── + +export type RunReportChunkOptions = { + provider: Provider; + apiKey: string; + model: string; + // The report directive (e.g. "major contradictions") — the sweep's focus. + directive: string; + // This chunk's videos, with their hit excerpts (the batch's evidence). + videos: RetrievedVideo[]; + // 1-based batch position, for the "Batch i of n" label. + index: number; + count: number; + // The running report carried in; the returned report threads to the next chunk. + report: string; + aliases: SearchAlias[]; + signal?: AbortSignal; + onEvent: (e: AgentEvent) => void; + onDebug?: (rec: DebugCall) => void; +}; + +export type ReportChunkResult = { + // The report after folding this chunk in. + report: string; + // How many update_report writes the model made this chunk. + writes: number; +}; + +// Reduce one chunk of search results into the running report. A lean cousin of +// runAskTurn's gather phase, report-mode-only: NO answer stream (the report IS the +// output — the *Gather transport returns when the model stops calling tools) and +// NO searching (budget 0 — the chunk's excerpts are the evidence; the prompt says +// so). The model folds findings via update_report and may drill into a thin line +// via fetch_context. Native providers only — the caller gates on report-mode +// availability. The chunk's raw excerpts are dropped when this returns; only the +// updated report survives. +export async function runReportChunk( + opts: RunReportChunkOptions, +): Promise<ReportChunkResult> { + const { provider, apiKey, model, directive, index, count, aliases, signal, onEvent } = + opts; + let report = opts.report ?? ""; + let writes = 0; + + // This chunk's videos, keyed for fetch_context ref resolution (and in-place + // snippet enrichment when the model reads more transcript around a line). + const videos = new Map<string, RetrievedVideo>(); + for (const v of opts.videos) videos.set(v.key, v); + + // Focused aliases: those the DIRECTIVE references (no search fires here, so the + // batch's mis-transcribed terms are steered purely from the sweep's focus). + const usedAliases = new Map<string, SearchAlias>(); + const addUsedAliases = (list: SearchAlias[]) => { + for (const a of list) if (!usedAliases.has(a.id)) usedAliases.set(a.id, a); + }; + if (aliases.length) { + addUsedAliases(matchAliases(directive, "transcripts", aliases)); + addUsedAliases(matchAliases(directive, "metadata", aliases)); + } + + // The report is the carried state: a single compacted "report so far" message + // (empty report → no history), mirroring runAskTurn's report-mode compaction. + const compactHistory = (): ChatMessage[] => + report.trim() === "" + ? [] + : [ + { + role: "assistant", + content: + "Report so far (keep it current with update_report):\n\n" + report, + }, + ]; + + const runUpdateReport = async ( + section: string, + content: string, + ): Promise<string> => { + report = applyReportPatch(report, section, content); + writes += 1; + onEvent({ type: "report_update", section }); + return `Updated the "${section}" section.`; + }; + + // Drill into a thin line: window this chunk-video's transcript around a moment + // and merge it into that video's citable excerpts (same windowing as runAskTurn). + const runFetchContext = async ( + rawRef: string, + aroundSeconds?: number, + ): Promise<string> => { + const ref = rawRef.trim(); + const v = videos.get(ref); + if (!v) return `No batch video with ref "${ref}".`; + const center = + typeof aroundSeconds === "number" + ? aroundSeconds + : v.snippets[0]?.seconds ?? 0; + onEvent({ type: "fetch_start", ref: v.key, label: v.title }); + let cues; + try { + const detail = await fetchTranscript(v.key); + cues = detail.cues ?? []; + } catch { + onEvent({ type: "fetch_done", ref: v.key, count: 0 }); + return `Couldn't load the transcript for "${v.title}".`; + } + const snips = cuesToSnippets(windowCues(cues, center, FETCH_WINDOW)); + v.snippets = mergeSnippets(v.snippets, snips, SNIPPETS_PER_VIDEO_CAP); + onEvent({ type: "fetch_done", ref: v.key, count: snips.length }); + if (snips.length === 0) { + return `No transcript lines found near ${hms(center)} in "${v.title}".`; + } + const body = snips.map((s) => `[${s.clock}] ${s.text}`).join("\n"); + return `Transcript excerpt from "${v.title}" around ${hms(center)}:\n${body}`; + }; + + const system = + `${accumulationSystemPrompt([...usedAliases.values()])}\n\n` + + buildSeedDigest(opts.videos); + const question = buildAccumulationContent(directive, opts.videos, index, count); + + const ctx: NativeGatherContext = { + apiKey, + model, + system, + history: compactHistory(), + question, + // No searching — the chunk is the evidence. A stray search call is refused + // by the transport (budget reached) and the prompt reinforces "do not search". + budget: 0, + runSearch: async () => + "Searching is disabled during a corpus sweep — use the batch excerpts provided.", + runFetchContext, + runUpdateReport, + onDebug: opts.onDebug, + signal, + }; + + switch (provider) { + case "anthropic": + await anthropicGather(ctx); + break; + case "openai": + await openaiGather(ctx); + break; + case "gemini": + await geminiGather(ctx); + break; + default: + throw new Error("A corpus sweep needs a native-tool-capable provider."); + } + + return { report, writes }; +} diff --git a/export/e2e/ask-chat.spec.ts b/export/e2e/ask-chat.spec.ts @@ -53,6 +53,15 @@ const TWO_RESULT_TREE: SGroup = { ], }; +// A three-video grounding: transcripts:"alpha" alone matches all three fixtures. +// Used by the corpus-sweep tests, which force a small batch size so three videos +// span multiple batches. +const ALL_RESULT_TREE: SGroup = { + k: "g", + o: "AND", + c: [{ k: "l", q: "alpha", s: "transcripts" }], +}; + // Run a real search on `/` (via qt=), wait for it to settle, then cross to the // chat through the shared workspace nav so it auto-grounds in those results. async function searchThenChat(page: Page, tree: SGroup, cardCount: number) { @@ -205,21 +214,47 @@ test.describe("ask chat", () => { await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible(); }); - test("inline [n] citations link to their source (in-page, no navigation)", async ({ + test("clicking an inline [n] citation opens the transcript modal at the cited line", async ({ page, }) => { await setup(page); await ask(page, "tell me about the alpha discussion"); await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible(); - // The [1] in the answer is a clickable in-page anchor to its source. + // The [1] in the answer is a clickable anchor; its href stays a #cite- anchor + // (a11y + hub fallback) even though the click opens the modal. const cite = page.getByRole("link", { name: "[1]", exact: true }); await expect(cite).toBeVisible(); await expect(cite).toHaveAttribute("href", /^#cite-/); - // Clicking scrolls to the source — it must NOT navigate or add a hash. + + // Clicking it opens the transcript modal seeked to the cited video — the URL + // gains ?v=<slug>&t=<seconds> (via replaceState on the same /ask route). + // vm defaults to "transcript", which is intentionally not written to the URL. await cite.click(); + await expect(page).toHaveURL(/[?&]v=/); + await expect(page).toHaveURL(/[?&]t=\d/); + const close = page.getByRole("button", { name: "Close player" }); + await expect(close).toBeVisible(); + + // Dismissing (✕) returns to /ask with the chat intact (no ?v=). + await close.click(); await expect(page).toHaveURL(/\/ask\/$/); - expect(page.url()).not.toContain("#"); + await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible(); + }); + + test("a source-list timestamp opens the transcript modal at its moment", async ({ + page, + }) => { + await setup(page); + await ask(page, "tell me about the alpha discussion"); + await expect(page.locator("li", { hasText: "point one" }).first()).toBeVisible(); + + // Each source's timestamp is a button that opens the modal at that second. + const ts = page.getByTitle("Open the transcript at this moment").first(); + await expect(ts).toBeVisible(); + await ts.click(); + await expect(page).toHaveURL(/[?&]v=/); + await expect(page.getByRole("button", { name: "Close player" })).toBeVisible(); }); test("a truncated answer (stop_reason max_tokens) shows the cut-off notice", async ({ @@ -758,6 +793,192 @@ test.describe("ask chat", () => { expect(lastAnswer).not.toContain("first question about the alpha topic"); }); + test("corpus sweep: accumulates report sections across batches; extend keeps, fresh clears", async ({ + page, + }) => { + await installRoutes(page); + // Force a small batch size so the 3 fixtures span 2 batches ([2,1]). + await page.addInitScript(() => + localStorage.setItem("ytdlp-tb:ai:sweepchunk", "2"), + ); + // A monotonic write counter → each folded section is uniquely named, so we + // can tell an EXTEND (prior sections survive) from a FRESH (prior cleared). + let writeSeq = 0; + const batches = new Set<string>(); + await page.route("https://api.anthropic.com/**", async (route) => { + if (route.request().method() === "OPTIONS") { + await route.fulfill({ status: 204, headers: CORS }); + return; + } + const body = route.request().postDataJSON() as { + system?: string; + tools?: { name?: string }[]; + messages?: { role: string; content: unknown }[]; + }; + const system = body.system ?? ""; + const msgs = body.messages ?? []; + const lastUser = [...msgs].reverse().find((m) => m.role === "user"); + const isToolResult = Array.isArray(lastUser?.content); + const text = typeof lastUser?.content === "string" ? lastUser.content : ""; + // Sweep gather turn (native tools + the accumulation system prompt): write + // one uniquely-numbered section per batch, then finish on the tool_result. + if (Array.isArray(body.tools) && /running report/.test(system)) { + if (isToolResult) { + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "application/json" }, + body: toolUse("finish", {}), + }); + return; + } + batches.add(/Batch (\d+) of/.exec(text)?.[1] ?? "?"); + writeSeq += 1; + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "application/json" }, + body: toolUse("update_report", { + section: `Finding ${writeSeq}`, + content: `Recorded finding ${writeSeq}.`, + }), + }); + return; + } + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("DONE"), + }); + }); + + await searchThenChat(page, ALL_RESULT_TREE, 3); + await keyIn(page, "Native tools"); + await expect(page.getByText(/Grounded in 3 results/)).toBeVisible(); + + // No report yet → a single primary "Build report from all 3 results". + await page + .getByRole("button", { name: /Build report from all 3 results/ }) + .click(); + + // Resolves with a summary naming the batch count, and BOTH batches folded in. + await expect( + page.getByText(/Built a report from 3 results across 2 batches/), + ).toBeVisible(); + expect([...batches].sort()).toEqual(["1", "2"]); + // The Report panel auto-opened and accumulated both batches' sections. + await expect(page.getByText("Recorded finding 1.")).toBeVisible(); + await expect(page.getByText("Recorded finding 2.")).toBeVisible(); + + // A report now exists → the action splits. EXTEND folds new sections in while + // keeping the old ones. + await page.getByRole("button", { name: "Add to report" }).click(); + await expect(page.getByText("Recorded finding 3.")).toBeVisible(); + await expect(page.getByText("Recorded finding 4.")).toBeVisible(); + // Prior sections survived the extend. + await expect(page.getByText("Recorded finding 1.")).toBeVisible(); + + // FRESH replaces: the prior sections are cleared before the new ones land. + await page.getByRole("button", { name: "Start new report" }).click(); + await expect(page.getByText("Recorded finding 5.")).toBeVisible(); + await expect(page.getByText("Recorded finding 6.")).toBeVisible(); + // The pre-fresh sections are gone. + await expect(page.getByText("Recorded finding 1.")).toHaveCount(0); + await expect(page.getByText("Recorded finding 4.")).toHaveCount(0); + }); + + test("corpus sweep: Stop aborts mid-sweep and keeps the partial report", async ({ + page, + }) => { + await installRoutes(page); + await page.addInitScript(() => + localStorage.setItem("ytdlp-tb:ai:sweepchunk", "2"), + ); + // Delay each sweep response so batch 2 is still in flight when we click Stop. + await page.route("https://api.anthropic.com/**", async (route) => { + if (route.request().method() === "OPTIONS") { + await route.fulfill({ status: 204, headers: CORS }); + return; + } + const body = route.request().postDataJSON() as { + system?: string; + tools?: { name?: string }[]; + messages?: { role: string; content: unknown }[]; + }; + const system = body.system ?? ""; + const msgs = body.messages ?? []; + const lastUser = [...msgs].reverse().find((m) => m.role === "user"); + const isToolResult = Array.isArray(lastUser?.content); + const text = typeof lastUser?.content === "string" ? lastUser.content : ""; + if (Array.isArray(body.tools) && /running report/.test(system)) { + await new Promise((r) => setTimeout(r, 500)); + if (isToolResult) { + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "application/json" }, + body: toolUse("finish", {}), + }); + return; + } + const b = /Batch (\d+) of/.exec(text)?.[1] ?? "1"; + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "application/json" }, + body: toolUse("update_report", { + section: `Batch ${b} finding`, + content: `Finding recorded in batch ${b}.`, + }), + }); + return; + } + await route.fulfill({ + status: 200, + headers: { ...CORS, "content-type": "text/event-stream" }, + body: sse("DONE"), + }); + }); + + await searchThenChat(page, ALL_RESULT_TREE, 3); + await keyIn(page, "Native tools"); + await expect(page.getByText(/Grounded in 3 results/)).toBeVisible(); + + await page + .getByRole("button", { name: /Build report from all 3 results/ }) + .click(); + + // Batch 1 folds in — its progress strip renders while batch 2 is delayed. + await expect(page.getByText("Finding recorded in batch 1.")).toBeVisible({ + timeout: 15_000, + }); + await expect(page.getByText(/Reading batch/)).toBeVisible(); + + // Stop the sweep from the pinned panel while batch 2 is still running. + await page.getByRole("button", { name: "Stop the sweep" }).click(); + + // The sweep message resolves to a "stopped" line, and the partial report + // (batch 1 only) survives — batch 2 never landed. + await expect( + page.getByText(/Stopped after 1 of 2 batches — the partial report is saved/), + ).toBeVisible(); + await expect(page.getByText("Finding recorded in batch 1.")).toBeVisible(); + await expect(page.getByText("Finding recorded in batch 2.")).toHaveCount(0); + }); + + test("corpus sweep is disabled on the scripted transport", async ({ page }) => { + await installRoutes(page); + await searchThenChat(page, ALL_RESULT_TREE, 3); + await keyIn(page, "Scripted"); + await expect(page.getByText(/Grounded in 3 results/)).toBeVisible(); + // The sweep block appears but its action is disabled, with the report-mode + // voice pointing at the Scripted transport. + const build = page.getByRole("button", { + name: /Build report from all 3 results/, + }); + await expect(build).toBeVisible(); + await expect(build).toBeDisabled(); + await expect( + page.getByText(/Needs a tool-capable provider — switch the search mode off/), + ).toBeVisible(); + }); + test("report mode is disabled on the scripted transport", async ({ page }) => { await installRoutes(page); await page.goto("/ask/");