// The correctness layer between a local 7B model and the artifact on disk. // // Every guard here is traceable to a MEASURED failure, not to defensive // instinct. On a real transcript, before the prompt/schema were hardened, // qwen2.5:7b produced: a malformed stamp (":00:27"), 11 of 11 starts outside the // input's range (one 55 minutes past the end), and titles that drifted into // Chinese. The hardened schema eliminated all three at the decoder — but a parser // that trusts the decoder is a parser that breaks the day an engine ignores the // schema, so each rule is enforced here too. // // Nothing is ever silently dropped. Every rejection lands in `warnings[]` with // the offending value, because a multi-week sweep is only tunable if its failures // are inspectable, and the Phase 11a review queue is built entirely on that array. // // Pure (no I/O) — unit-tested in digestParse.test.ts. import { chapterId, tagId, type DigestChapter, type DigestTag, type DigestTimestampMode, type DigestWarning, } from "./digest"; import { HMS_RE, hmsToSeconds, promptOffsetSeconds, toHms, } from "./digestPrompt"; import type { Cue } from "./vtt"; // One engine call's output, with the range that call was RESPONSIBLE for. The // range is per-chunk on purpose: a whole-video check would have accepted the // measured 01:10:29 for a 15-minute input, because the video was 176 minutes long. export type DigestChunkOutput = { index: number; startSeconds: number; endSeconds: number; // Whatever the engine returned. Unvalidated by construction. data: unknown; // Which numbering the engine was PROMPTED in. Must match what the caller // rendered the chunk with; digestVideo passes the same value to both. Defaults // to "absolute", so every existing caller is unaffected. // // Under "chunk-local" the model counted from 00:00:00 and the parser adds // startSeconds back BEFORE any guard runs. That ordering is deliberate: the // range clamp then still checks the chunk's real range (unchanged in // strength), monotonicity is still compared in real video time, the cue snap // still lands on a real cue, and every warning still names a real video time // rather than an offset the reader would have to undo by hand. timestampMode?: DigestTimestampMode; }; // How far outside its own range a chunk's start may fall before it is rejected. // Small and non-zero: a cue that begins a hair before the slice's first cue start // is a rounding artifact, not a hallucination. const RANGE_TOLERANCE_SECONDS = 2; // Two chapters closer than this, with equivalent titles, are the same chapter // seen through the chunk overlap. Sized to the overlap (40 cues ≈ 2-4 minutes of // speech) so a genuine topic change at a 3-minute gap survives. const SEAM_DEDUP_SECONDS = 90; // Scripts that mean the model stopped writing English. The prompt and system // message both demand English titles, so ANY character from one of these is // drift, not a loanword — and drift is the signal that a chunk confused the // model, which makes the rest of its output for that chunk suspect too. const NON_LATIN_RE = /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}\p{Script=Cyrillic}\p{Script=Arabic}\p{Script=Hebrew}\p{Script=Devanagari}\p{Script=Thai}\p{Script=Greek}]/u; export type ParsedChapters = { chapters: DigestChapter[]; warnings: DigestWarning[]; }; export type ParsedTags = { tags: DigestTag[]; warnings: DigestWarning[]; }; // GUARD 5 — snap a model-emitted second to the nearest real cue boundary. // // A chapter must start where someone actually starts speaking, or the viewer // jumps into the middle of a sentence. Hand-rolled binary search over cue starts, // the same shape as findActiveIndex in TranscriptModal.tsx. (resolveCitationSeconds // in the viewer scans SNIPPETS — the ~5 matched lines per video — not cues, so it // cannot be reused here.) export function snapToCueStart(cues: Cue[], seconds: number): number { if (cues.length === 0) return Math.max(0, Math.floor(seconds)); let lo = 0; let hi = cues.length - 1; let found = -1; while (lo <= hi) { const mid = (lo + hi) >> 1; if (cues[mid].start <= seconds) { found = mid; lo = mid + 1; } else { hi = mid - 1; } } // Before the first cue: the first cue is the only sensible boundary. if (found < 0) return Math.max(0, Math.floor(cues[0].start)); const before = cues[found].start; const after = found + 1 < cues.length ? cues[found + 1].start : null; if (after !== null && after - seconds < seconds - before) { return Math.max(0, Math.floor(after)); } return Math.max(0, Math.floor(before)); } // Titles are compared for seam de-dup after this normalization, so "Court filing // deadlines" and "court filing deadlines." collapse. function normalizeTitle(title: string): string { return title .toLowerCase() .replace(/[^\p{L}\p{N}]+/gu, " ") .trim(); } function titlesEquivalent(a: string, b: string): boolean { const na = normalizeTitle(a); const nb = normalizeTitle(b); if (!na || !nb) return false; if (na === nb) return true; // The overlap frequently yields one call's fuller phrasing of the other's. return na.includes(nb) || nb.includes(na); } type RawChapter = { start: unknown; title: unknown }; function readRawChapters(data: unknown): RawChapter[] | null { if (!data || typeof data !== "object") return null; const arr = (data as { chapters?: unknown }).chapters; if (!Array.isArray(arr)) return null; return arr as RawChapter[]; } // Parse ONE chunk's chapters, applying guards 1-4 in the order that keeps the // warnings legible: shape, then range, then language, then monotonicity. function parseChapterChunk( chunk: DigestChunkOutput, cues: Cue[], ): ParsedChapters { const warnings: DigestWarning[] = []; const warn = ( code: DigestWarning["code"], value?: string, detail?: string, ): void => { warnings.push({ code, section: "chapters", chunk: chunk.index, ...(value !== undefined ? { value } : {}), ...(detail !== undefined ? { detail } : {}), }); }; const raw = readRawChapters(chunk.data); if (raw === null) { warn("parse-failed", JSON.stringify(chunk.data ?? null).slice(0, 200)); return { chapters: [], warnings }; } if (raw.length === 0) { warn("empty-output"); return { chapters: [], warnings }; } // Added back to every emitted stamp before the guards run. Zero unless the // chunk was prompted in chunk-local numbering. const offset = promptOffsetSeconds(chunk); const lo = chunk.startSeconds - RANGE_TOLERANCE_SECONDS; const hi = chunk.endSeconds + RANGE_TOLERANCE_SECONDS; const kept: DigestChapter[] = []; // Monotonicity is checked against the model's EMITTED order, not sorted order: // a chunk that jumps backwards has lost track of where it is, and that entry is // the suspect one. let lastStart = -1; for (const entry of raw) { const rawStart = typeof entry?.start === "string" ? entry.start : ""; const rawTitle = typeof entry?.title === "string" ? entry.title.trim() : ""; // GUARD 1 — timestamp shape. The schema pins this, so a hit here means an // engine ignored the schema; recording it is how we'd find that out. if (!HMS_RE.test(rawStart)) { warn("malformed-timestamp", rawStart || String(entry?.start)); continue; } const emitted = hmsToSeconds(rawStart); if (emitted === null) { warn("malformed-timestamp", rawStart); continue; } // From here on everything is in REAL VIDEO TIME. `clock` follows suit, so a // stored chapter never carries a stamp in one numbering next to a `start` in // the other. In absolute mode both are identical to what the model emitted. const seconds = emitted + offset; const clock = toHms(seconds); // Only worth saying when the two differ, i.e. chunk-local. const emittedNote = clock === rawStart ? "" : ` (model emitted ${rawStart})`; if (!rawTitle) { warn("empty-title", clock); continue; } // GUARD 2 — language drift. if (NON_LATIN_RE.test(rawTitle)) { warn("language-drift", rawTitle); continue; } // GUARD 3 — per-chunk range clamp, always against the chunk's REAL range. // Under chunk-local this is the guard the offset is designed to let the // model pass; it is not weakened to do so. if (seconds < lo || seconds > hi) { warn( "out-of-range", clock, `outside ${toHms(chunk.startSeconds)}–${toHms(chunk.endSeconds)}${emittedNote}`, ); continue; } // GUARD 4a — monotonic starts within the chunk. if (seconds <= lastStart) { warn("non-monotonic", clock, `after ${toHms(lastStart)}${emittedNote}`); continue; } lastStart = seconds; // GUARD 5 — snap onto a real cue boundary. const snapped = snapToCueStart(cues, seconds); kept.push({ id: chapterId(snapped), start: snapped, clock, title: rawTitle, decidedBy: "ai", }); } return { chapters: kept, warnings }; } // Parse every chunk and merge, applying guard 4b (seam de-dup) across chunk // boundaries. Chunks are processed in `index` order so "first wins" is stable. export function parseChapters( chunks: DigestChunkOutput[], cues: Cue[], ): ParsedChapters { const warnings: DigestWarning[] = []; const all: DigestChapter[] = []; for (const chunk of [...chunks].sort((a, b) => a.index - b.index)) { const parsed = parseChapterChunk(chunk, cues); warnings.push(...parsed.warnings); all.push(...parsed.chapters); } all.sort((a, b) => a.start - b.start || a.title.localeCompare(b.title)); const kept: DigestChapter[] = []; for (const chapter of all) { const prev = kept[kept.length - 1]; if (prev) { // Snapping can land two chunks' views of one moment on the same cue. if (prev.start === chapter.start) { warnings.push({ code: "seam-duplicate", section: "chapters", value: chapter.title, detail: `same start as "${prev.title}"`, }); continue; } if ( chapter.start - prev.start <= SEAM_DEDUP_SECONDS && titlesEquivalent(prev.title, chapter.title) ) { warnings.push({ code: "seam-duplicate", section: "chapters", value: chapter.title, detail: `equivalent to "${prev.title}" ${chapter.start - prev.start}s earlier`, }); continue; } } kept.push(chapter); } return { chapters: kept, warnings }; } // --------------------------------------------------------------------------- // Tags // --------------------------------------------------------------------------- export function parseTags(chunks: DigestChunkOutput[]): ParsedTags { const warnings: DigestWarning[] = []; const byId = new Map(); for (const chunk of [...chunks].sort((a, b) => a.index - b.index)) { const data = chunk.data; const arr = data && typeof data === "object" ? (data as { tags?: unknown }).tags : undefined; if (!Array.isArray(arr)) { warnings.push({ code: "parse-failed", section: "tags", chunk: chunk.index, value: JSON.stringify(data ?? null).slice(0, 200), }); continue; } if (arr.length === 0) { warnings.push({ code: "empty-output", section: "tags", chunk: chunk.index }); continue; } for (const raw of arr) { const tag = typeof raw === "string" ? raw.trim().toLowerCase() : ""; if (!tag) { warnings.push({ code: "empty-title", section: "tags", chunk: chunk.index, value: String(raw), }); continue; } if (NON_LATIN_RE.test(tag)) { warnings.push({ code: "language-drift", section: "tags", chunk: chunk.index, value: tag, }); continue; } const id = tagId(tag); // Chunks overlap, so the same tag arrives repeatedly — that is expected, // not a failure, so it is deduped WITHOUT a warning (unlike a chapter seam // duplicate, which indicates a real segmentation ambiguity). if (!byId.has(id)) byId.set(id, { id, tag, decidedBy: "ai" }); } } return { tags: Array.from(byId.values()).sort((a, b) => a.tag.localeCompare(b.tag)), warnings, }; }