// Conversation assembly + prompts for the /ask chat. // // This is the layer that fixes the "follow-up loses its grounding" bug: prior // user turns are replayed with the EXACT text that was sent (question + the // excerpts retrieved that turn), not the bare question — so excerpts from // earlier turns stay in context and a follow-up like "format it in a timeline" // can reuse them. The search agent (searchAgent.ts) drives retrieval; this // module owns the message shapes and the system prompts (including an alias // glossary so the model accounts for AI-transcription misspellings). // // Pure (no I/O, no React) so it can be unit-tested. import type { ChatMessage } from "./askProvider"; import { buildContext, type RetrievedVideo } from "./askRetrieval"; import { mergeSnippets } from "yt-dlp-transcript-common/lib/transcriptWindow"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; // One retrieval step the agent ran this turn (shown live in the pipeline UI). // `kind` is "search" (default) or "fetch" (reading more of a pinned video's // transcript); for a fetch step `query` holds the video title. `count` is // undefined while in flight, then the match/line count. export type SearchStep = { query: string; count?: number; kind?: "search" | "fetch" }; // Top-level stage of a pending assistant turn, for the status indicator. export type AssistantPhase = | "gathering" // agent is deciding / running searches | "answering" // gather done, awaiting the first answer token | "streaming" // answer tokens are arriving | "done" | "error" | "stopped"; export type UiMessage = { role: "user" | "assistant"; content: string; // Assistant turns: the videos whose excerpts grounded this answer. This is the // persisted grounding — the full grounded text is rebuilt from it on demand // (retry, follow-up pool) rather than stored, keeping the saved blob small. sources?: RetrievedVideo[]; searchSteps?: SearchStep[]; phase?: AssistantPhase; truncated?: boolean; error?: boolean; // Assistant turns: which model produced this answer (shown as a per-message // attribution, since the provider/model can change mid-conversation). model?: string; // Assistant turns: the answer's normalized finish reason ("length" = cut off at // the output limit; "safety" = blocked) so the UI can flag it. `blockReason` is // the provider's raw block/safety string when present. finishReason?: string; blockReason?: string; }; // Replay completed prior turns for the model as BARE text — user turns are the // question only, assistant turns are the answer. Excerpts are deliberately NOT // re-embedded here: re-sending every prior turn's excerpts in every turn's // history grew the context quadratically (the "trapped in context" bug). Instead // the excerpts a follow-up may reuse are carried forward once, via the cumulative // pool (`collectPriorPool`) that the caller folds into THIS turn's grounding. // Error/empty turns are dropped. The current turn is appended by the caller. export function buildApiMessages(prior: UiMessage[]): ChatMessage[] { return prior .filter((m) => !m.error && m.content.trim() !== "") .map((m) => ({ role: m.role, content: m.content })); } // The cumulative excerpt pool for a conversation: every video cited in an earlier // assistant turn, deduped by `key` in first-seen order, with snippets merged // across turns (a video expanded later keeps its fuller excerpts). Seeded into a // new turn's grounding so a reformat/summarise follow-up still sees earlier // excerpts — but each excerpt is sent ONCE per turn, not re-embedded per prior // turn. Pure + testable. export function collectPriorPool( prior: UiMessage[], perVideoCap = 30, ): RetrievedVideo[] { const order: string[] = []; const byKey = new Map(); for (const m of prior) { if (m.role !== "assistant" || m.error || !m.sources) continue; for (const v of m.sources) { const existing = byKey.get(v.key); if (!existing) { order.push(v.key); byKey.set(v.key, { ...v, snippets: v.snippets.slice() }); } else { existing.snippets = mergeSnippets( existing.snippets, v.snippets, perVideoCap, ); } } } return order.map((k) => byKey.get(k)!); } // ─── report source registry (what a report's global `[n]` indexes into) ─── // The report's persisted source registry: every video cited across all report // writes (report-mode turns + sweep batches), ordered, deduped by `key`. A // report's `[n]` is this list's n-th entry, and `[n @ mm:ss]` seeks to the cited // moment. Stored TRIMMED — snippet text is dropped (only clock/seconds + the video // meta are needed to resolve or label a citation) so a large sweep's registry // stays storage-light. const REPORT_SNIPPETS_CAP = 30; // Drop snippet text (keep clock/seconds) so the persisted registry is light. The // RetrievedVideo shape is preserved (text becomes ""), so it still renders in the // source list and resolves a citation's second. function trimReportSource(v: RetrievedVideo): RetrievedVideo { return { ...v, snippets: v.snippets.map((s) => ({ clock: s.clock, seconds: s.seconds, text: "" })), }; } // Fold a batch/turn's videos into the report registry: append videos not seen // before (by `key`, first-seen order preserved) and merge snippet times into ones // already present (a video cited again in a later batch keeps all its moments). // Mirrors the cumulative-pool dedup in collectPriorPool. Pure + testable. export function mergeReportSources( existing: RetrievedVideo[], batch: RetrievedVideo[], ): RetrievedVideo[] { const order: string[] = existing.map((v) => v.key); const byKey = new Map(existing.map((v) => [v.key, v])); for (const raw of batch) { const v = trimReportSource(raw); const cur = byKey.get(v.key); if (!cur) { order.push(v.key); byKey.set(v.key, v); } else { byKey.set(v.key, { ...cur, snippets: mergeSnippets(cur.snippets, v.snippets, REPORT_SNIPPETS_CAP), }); } } return order.map((k) => byKey.get(k)!); } // A 1-based global-index lookup over a registry, keyed by video `key` — the // numbering a report write uses so its `[n]` matches the registry's source list. export function reportNumbering( registry: RetrievedVideo[], ): (v: RetrievedVideo, index: number) => number { const indexByKey = new Map(registry.map((v, i) => [v.key, i + 1])); return (v, index) => indexByKey.get(v.key) ?? index + 1; } // ─── report mode (persistent markdown document, maintained via update_report) ─── // Upsert a `##
` block into the running report. If a section with that // heading already exists (matched case-insensitively at the `##` level), its body // is replaced with `content`; otherwise a new `##
` block is appended to // the end. Any preamble and the other sections are preserved, and runs of 3+ // newlines are collapsed to a single blank line so repeated edits stay tidy. // Pure + testable — the executor in searchAgent applies it as the model calls the // tool, threading the result through the turn. export function applyReportPatch( report: string, section: string, content: string, ): string { const heading = /^##\s+(.+?)\s*$/; const target = section.trim().toLowerCase(); const body = content.trim(); const lines = report.split("\n"); // Index every `##`-level section heading (exactly two hashes: `### …` and // deeper don't match, so sub-headings inside a section body are left intact). const heads: { line: number; title: string }[] = []; lines.forEach((l, i) => { const m = heading.exec(l); if (m) heads.push({ line: i, title: m[1] }); }); const hit = heads.findIndex((h) => h.title.trim().toLowerCase() === target); let out: string; if (hit === -1) { // Append a new section, keeping any existing preamble/sections ahead of it. const block = `## ${section.trim()}\n\n${body}`; out = report.trim() === "" ? block : `${report.replace(/\s+$/, "")}\n\n${block}`; } else { // Replace the matched section's body, preserving its original heading line. const start = heads[hit].line; const end = hit + 1 < heads.length ? heads[hit + 1].line : lines.length; const before = lines.slice(0, start); const after = lines.slice(end); const block = `${lines[start]}\n\n${body}`; out = [...before, ...block.split("\n"), ...after].join("\n"); } return out.replace(/\n{3,}/g, "\n\n").trim(); } // ─── editable context (view / prune / forward to a new session) ─── const ROLE_MARKER = /^(USER|ASSISTANT):\s*$/i; // Serialize a message list into an editable transcript: a `USER:` / `ASSISTANT:` // header line per turn followed by its content (excerpts inline). This is what // the human sees and prunes in the context panel. export function serializeContext(messages: ChatMessage[]): string { return messages .map((m) => `${m.role.toUpperCase()}:\n${m.content}`) .join("\n\n"); } // Parse an edited context transcript back into messages, splitting on the // `USER:` / `ASSISTANT:` header lines. Text before any marker (or with no markers // at all) becomes a single user message, so a hand-pasted blob still works. export function parseContext(text: string): ChatMessage[] { const lines = text.split(/\r?\n/); const out: ChatMessage[] = []; let role: "user" | "assistant" = "user"; let buf: string[] = []; let sawMarker = false; const flush = () => { const content = buf.join("\n").trim(); if (content) out.push({ role, content }); buf = []; }; for (const line of lines) { const m = line.match(ROLE_MARKER); if (m) { flush(); role = m[1].toLowerCase() === "assistant" ? "assistant" : "user"; sawMarker = true; } else { buf.push(line); } } flush(); if (!sawMarker) { const content = text.trim(); return content ? [{ role: "user", content }] : []; } return out; } // Above this many grounding videos we switch to a tiered layout: an index of // EVERY video plus full excerpts for only the top ones. Keeps a large result set // bounded in tokens while still telling the model every match exists (it can // fetch_context — or the user can "Load context" — to pull excerpts for any). export const EXCERPT_VIDEOS_CAP = 12; function formatUploadDate(d: string): string { const m = /^(\d{4})(\d{2})(\d{2})/.exec(d); return m ? `${m[1]}-${m[2]}-${m[3]}` : d; } // A one-line index entry for a video: number, title, channel, date, hit count. function videoIndexLine(v: RetrievedVideo, n: number): string { const meta = [v.channel, formatUploadDate(v.uploadDate)].filter(Boolean).join(" · "); const site = v.siteTitle ? ` (${v.siteTitle})` : ""; const hits = v.snippets.length; return `[${n}] "${v.title}" — ${meta}${site} · ${hits} excerpt${hits === 1 ? "" : "s"}`; } // Tiered grounding for a large result set: an index of ALL videos (so the model // knows the full set and can cite or drill into any) + full excerpts for the top // `topN`. Numbering is shared — [n] in the index is the same video as [n] in the // excerpts (the top videos are the first `topN`). export function buildTieredGrounding( question: string, videos: RetrievedVideo[], topN: number = EXCERPT_VIDEOS_CAP, ): string { const index = videos.map((v, i) => videoIndexLine(v, i + 1)).join("\n"); const top = videos.slice(0, topN); return ( `${question}\n\n---\n` + `All ${videos.length} matching videos (cite any by its number; use the ` + `fetch_context tool — or ask me to "Load context" — to read more of one):\n` + `${index}\n\n` + `Full excerpts for the top ${top.length} (cite by number):\n${buildContext(top)}` ); } // Assemble the grounded user content for a turn: the question plus this turn's // retrieved excerpts (numbered for citation). No excerpts → just the question, // so a no-search follow-up leans on excerpts already in earlier turns. A large // set switches to the tiered index+excerpts layout. export function buildGroundedContent( question: string, videos: RetrievedVideo[], ): string { if (videos.length === 0) return question; if (videos.length > EXCERPT_VIDEOS_CAP) { return buildTieredGrounding(question, videos); } return ( `${question}\n\n---\n` + `Transcript excerpts you may cite (by number):\n${buildContext(videos)}` ); } // A directive glossary of known terms that AI transcription mangles. It tells // the model, per alias, to SEARCH THE FULL TERM (not a fragment) so the archive // can apply the curated match, and to treat the variant spellings as the same // entity when reading excerpts. Empty when there are no usable aliases. // // This matters because the transcript search only applies an alias's regex when // the search query contains the alias's full trigger phrase — searching a // fragment (e.g. "Graham" instead of "graham platner") silently misses the // curated misspelling coverage. The instruction below is what steers the model // to search the whole term. Each line also carries the alias `note` so the model // understands why the term is special. export function renderAliasGlossary(aliases: SearchAlias[], max = 40): string { const usable = aliases.filter((a) => a.enabled !== false); if (usable.length === 0) return ""; const shown = usable.slice(0, max); const lines = shown.map((a) => { const term = a.triggers[0] ?? a.label; const alts = a.triggers.length > 1 ? ` (also written: ${a.triggers.slice(1).join(", ")})` : ""; const note = a.note ? ` — ${a.note}` : ""; return `- ${a.label}: when your search concerns this, search the full term "${term}"${alts}, NOT a fragment; the archive then automatically broadens it to catch mis-transcriptions${note}`; }); const overflow = usable.length > max ? `\n…and ${usable.length - max} more.` : ""; return ( "KNOWN TERMS — this archive has curated aliases for names/terms that AI " + "transcription commonly mis-spells. When a search concerns one of these, " + "search the FULL term shown below (never an abbreviated fragment) so the " + "curated match applies; and when reading excerpts, treat the listed variant " + "spellings as the same entity:\n" + lines.join("\n") + overflow ); } function withGlossary(base: string, aliases: SearchAlias[]): string { const glossary = renderAliasGlossary(aliases); return glossary ? `${base}\n\n${glossary}` : base; } // A FOCUSED alias block for the answer phase: only the aliases a search actually // used this turn, framed for reading the excerpts (not for deciding what to // search — that's the gather phase's full glossary). It tells the model that the // listed variant spellings in the excerpts are all the same term, so it reads a // mis-transcribed name/word correctly (e.g. a "k-cup" search whose alias is // "Cake Cups"). Empty string when nothing usable fired. Deduped by id. export function renderUsedAliasContext(used: SearchAlias[]): string { const usable: SearchAlias[] = []; const seen = new Set(); for (const a of used) { if (a.enabled === false || seen.has(a.id)) continue; seen.add(a.id); usable.push(a); } if (usable.length === 0) return ""; const lines = usable.map((a) => { const spellings = a.triggers.join(", "); const note = a.note ? ` — ${a.note}` : ""; return `- ${a.label}: ${spellings}${note}`; }); return ( "TERMS USED IN THIS SEARCH — the excerpts below may contain AI-transcription " + "mis-spellings of these curated terms. Treat every listed spelling as the " + "same term when reading and citing the excerpts:\n" + lines.join("\n") ); } // System prompt for the gather (search-decision) phase. `mode` tailors the // closing instruction: native tool-calling vs. the scripted text protocol. export function gatherSystemPrompt( aliases: SearchAlias[], mode: "native" | "scripted", budget: number, canFetch = false, canReport = false, ): string { const base = "You are helping answer a question about a video-transcript archive. " + "Before answering you may search the transcripts to gather relevant " + "excerpts. Use focused keyword or name queries. Read each result set and, " + "if it helps, search again with a refined query — for example correcting a " + "spelling, trying an alternate name, or narrowing to a specific event. " + "Only search when you need transcript evidence: if the latest message just " + "asks to reformat, summarise, translate, or expand on the previous answer, " + `do not search. You may run at most ${budget} searches.`; const fetchLine = canFetch ? " When a snippet is too short to answer confidently, call the " + "fetch_context tool with the video's ref (and optionally a timestamp in " + "seconds) to read the surrounding transcript before answering." : ""; const reportLine = canReport ? " You are maintaining a running report document that persists across turns. " + "After gathering evidence, call the update_report tool to record findings in " + "well-titled sections (pass the section heading and its full markdown " + "content), citing each source as [n @ mm:ss] — the excerpt's bracketed " + "number plus the cited line's timestamp; keep the report the source of " + "truth. Then answer briefly." : ""; const tail = mode === "native" ? "Call the search_transcripts tool to search." + fetchLine + reportLine + " Call the finish tool as soon as you have enough excerpts, or " + "immediately if no search is needed." : "Reply with EXACTLY one line and nothing else: either " + "`SEARCH: ` to run a search, or `DONE` when you have enough " + "excerpts (or need no search)."; return withGlossary(`${base}\n\n${tail}`, aliases); } // System prompt for the final answer phase. Unlike the gather phase (which gets // the full glossary so it can discover terms to search), this gets a FOCUSED // block naming only the aliases the turn's searches actually used — so the model // reads the specific mis-transcribed terms in these excerpts correctly without // the noise of every enabled alias. export function answerSystemPrompt(usedAliases: SearchAlias[]): string { const base = "You are answering a question about an archive of video transcripts AND " + "social posts (X/Twitter, Bluesky, forum threads) from the same commentators — one corpus, " + "two kinds of source. Base your " + "answer on the excerpts provided in this conversation. Excerpts " + "may appear in earlier turns, and a follow-up request (reformatting, " + "summarising, expanding) should reuse the relevant excerpts already " + "provided. Cite each excerpt you use as [n @ mm:ss]: its bracketed number " + "plus the specific line's timestamp (e.g. [1 @ 4:12]); a bare [n] is fine " + "when no single line applies. A video excerpt shows a title, channel, and " + "timestamped lines. A POST excerpt is marked `post by ` and has NO " + "timestamps — always cite a post as a bare [n], never with @ mm:ss. If the " + "excerpts do not contain enough to answer, say so plainly rather than " + "guessing. Format your answer in GitHub-flavored Markdown."; const context = renderUsedAliasContext(usedAliases); return context ? `${base}\n\n${context}` : base; } // ─── whole-corpus sweep (map result-chunks → reduce into the running report) ─── // Split a list into fixed-size chunks; the last chunk may be shorter. A size ≤ 0 // yields a single chunk with everything (or none for an empty list). Pure + // testable — the sweep driver folds each chunk into the report in turn. export function chunk(items: T[], size: number): T[][] { if (size <= 0) return items.length ? [items.slice()] : []; const out: T[][] = []; for (let i = 0; i < items.length; i += size) out.push(items.slice(i, i + size)); return out; } // System prompt for one sweep chunk: the model reads a batch of transcript // excerpts and folds findings into a running report via update_report, then // discards the batch's raw text. Native-only (needs the update_report tool). Gets // the FOCUSED used-alias block (like the answer phase) so mis-transcribed terms in // the batch read correctly — NOT the full glossary (there's no searching here). export function accumulationSystemPrompt(usedAliases: SearchAlias[]): string { const base = "You are building a running report by reading transcript excerpts in " + "batches. Cross-reference this batch against the report so far. Use the " + "update_report tool to add or merge findings — claims, contradictions with " + "earlier claims, and their sources — into well-titled sections, keeping the " + "report the single source of truth. CITE each source as [n @ mm:ss]: the " + "excerpt's bracketed number plus the specific line's timestamp (e.g. " + "[3 @ 12:34]); the numbers are GLOBAL and stable across batches, so reuse a " + "source's number if it reappears. Do NOT search; these excerpts are your " + "evidence, but you may call the fetch_context tool to read more transcript " + "around a specific line when one is too thin to judge. Call the finish tool " + "when you are done with this batch."; const context = renderUsedAliasContext(usedAliases); return context ? `${base}\n\n${context}` : base; } // The per-chunk user message for a sweep: the report directive, a "Batch i of n" // label, and this batch's excerpts (same format as a grounded turn, reusing // buildContext). `numberOf` labels each excerpt by its GLOBAL registry index so a // report's `[n]` is stable across every batch — the source registry the report's // citations resolve against persists even though the raw excerpts are discarded // once the batch folds in. Omitted → 1-based per-batch numbering (legacy). export function buildAccumulationContent( directive: string, videos: RetrievedVideo[], index: number, count: number, numberOf?: (video: RetrievedVideo, i: number) => number, ): string { return ( `${directive}\n\n` + `Batch ${index} of ${count} — transcript excerpts (cite each as [n @ mm:ss]: ` + `the bracketed number + the line's timestamp):\n` + buildContext(videos, numberOf) ); }