import type { Cue } from "./vtt"; import { hms } from "./aiHandoff"; // Turning a hit into its surrounding transcript. Both the /ask chat's user-driven // "expand this hit" and the AI's fetch_context tool share these helpers: fetch a // full transcript (client-side, IDB-cached), take a bounded window of cues around // a timestamp, and merge them into a video's snippets (the citable excerpts). All // pure so they unit-test without I/O. // Same shape as HandoffSnippet / RetrievedVideo's snippet, so a windowed slice // drops straight into the grounding. export type WindowSnippet = { clock: string; seconds: number; text: string }; // Match askRetrieval's per-snippet slice so windowed lines stay the same size as // search hits. const SNIPPET_MAX_CHARS = 240; // Cues whose start falls within [center-before, center+after] seconds. When more // than `maxCues` qualify, keep the `maxCues` closest to the center (a token guard // for very dense stretches) but return them back in chronological order. export function windowCues( cues: Cue[], centerSeconds: number, opts: { before?: number; after?: number; maxCues?: number } = {}, ): Cue[] { const before = opts.before ?? 45; const after = opts.after ?? 45; const maxCues = opts.maxCues ?? 60; const lo = centerSeconds - before; const hi = centerSeconds + after; const inWindow = cues.filter((c) => c.start >= lo && c.start <= hi); if (inWindow.length <= maxCues) return inWindow; return inWindow .slice() .sort( (a, b) => Math.abs(a.start - centerSeconds) - Math.abs(b.start - centerSeconds), ) .slice(0, maxCues) .sort((a, b) => a.start - b.start); } // Split a whole transcript into SEQUENTIAL, overlapping, context-sized slices — // the chunker the digest generator drives its engine with, one call per slice. // // This is a different job from windowCues() above and cannot be expressed with // it: that one is CENTER-based (a window around a search hit, measured in // seconds) and has no notion of covering a transcript exactly once. Here the // contract is coverage: every cue appears in at least one slice, slices are in // order, and consecutive slices share `overlapCues` cues. // // The overlap exists because a topic that straddles a seam is otherwise invisible // to both calls — each sees half a discussion and titles it wrong. With the // overlap, at least one call sees it whole; the parser's seam de-dup then drops // the resulting near-identical chapters. // // `maxCues` is measured in cues rather than tokens deliberately: cues are what we // slice, and the token estimate that sizes them belongs with the prompt (see // DIGEST_MAX_CUES_PER_CHUNK), not here. export function chunkCuesForContext( cues: Cue[], opts: { maxCues?: number; overlapCues?: number } = {}, ): Cue[][] { const maxCues = Math.max(1, Math.floor(opts.maxCues ?? 1200)); // Overlap must leave forward progress, or the loop never advances. const overlapCues = Math.max( 0, Math.min(Math.floor(opts.overlapCues ?? 0), maxCues - 1), ); if (cues.length === 0) return []; if (cues.length <= maxCues) return [cues.slice()]; const step = maxCues - overlapCues; const out: Cue[][] = []; for (let start = 0; start < cues.length; start += step) { out.push(cues.slice(start, start + maxCues)); // Stop once this slice reached the end, so we never emit a trailing slice // that is pure overlap (it would be re-processed for nothing). if (start + maxCues >= cues.length) break; } return out; } // How many chunks chunkCuesForContext() WOULD produce for a transcript of // `cueCount` cues — the same count, without needing the cues themselves. // // This exists because a chunk is the digest sweep's real unit of work (one chunk // = one model call), and the pricing pass has only `statsByPath.cueCount` to work // from — reading 73k transcripts to count slices would cost more than the thing // it is pricing. Chunk density varies FOURFOLD across the corpus (6.8 // chunks/audio-hour under 15 min against 1.7 over 8 h), which is why // seconds-per-audio-hour is not a stable unit and three past cost estimates // looked contradictory when they agreed to within 2% per chunk. // // Deliberately in this file, immediately below the chunker it mirrors: the two // must not drift, and a test pins this against the real slicing. export function countCueChunks( cueCount: number, opts: { maxCues?: number; overlapCues?: number } = {}, ): number { // Clamp exactly as the chunker does, so an odd config can't make the two // disagree about what "one chunk" means. const maxCues = Math.max(1, Math.floor(opts.maxCues ?? 1200)); const overlapCues = Math.max( 0, Math.min(Math.floor(opts.overlapCues ?? 0), maxCues - 1), ); const n = Math.max(0, Math.floor(cueCount)); if (n === 0) return 0; if (n <= maxCues) return 1; return 1 + Math.ceil((n - maxCues) / (maxCues - overlapCues)); } // Render cues as snippet objects (clock + seconds + collapsed, capped text). // Empty cues are dropped. export function cuesToSnippets(cues: Cue[]): WindowSnippet[] { const out: WindowSnippet[] = []; for (const c of cues) { const text = c.text.trim().replace(/\s+/g, " ").slice(0, SNIPPET_MAX_CHARS); if (!text) continue; out.push({ clock: hms(c.start), seconds: Math.max(0, Math.floor(c.start)), text, }); } return out; } // Merge freshly-windowed snippets into a video's existing ones: keep every // existing snippet (the original hits are the most relevant), append incoming // ones (deduped by second) until `cap`, and return sorted by timestamp. Never // drops existing hits, so enrichment only ever adds context. export function mergeSnippets( existing: T[], incoming: T[], cap: number, ): T[] { const seen = new Set(existing.map((s) => s.seconds)); const merged = existing.slice(); for (const s of incoming) { if (merged.length >= cap) break; if (seen.has(s.seconds)) continue; seen.add(s.seconds); merged.push(s); } return merged.sort((a, b) => a.seconds - b.seconds); }