// The prompt + JSON schema the digest engines are driven with, and the ONE // constant (PROMPT_VERSION) that invalidates generated sections when either // changes. Pure — no I/O — so it can be unit-tested and imported anywhere. // // WHY THE SCHEMA IS THIS STRICT. Measured on a real transcript with qwen2.5:7b: // // naive prompt, loose schema → malformed stamps (":00:27"), 11 of 11 starts // out of range, output drifted to Chinese // hardened prompt + this → 0 malformed, 0 out of range, 0 drift // // Pinning `start` to a full HH:MM:SS regex, stating the chunk's own time range in // the prompt, and demanding English titles eliminated EVERY correctness failure. // A 7B model will not voluntarily honor a text contract; schema-constrained // decoding is what makes the local lane usable. Do not loosen the pattern. // // Bump PROMPT_VERSION for any change to the prompt text, the schema, or the // chunking constants below — a section whose recorded promptVersion differs is // regenerated, and a section whose version matches is skipped. That is what // keeps a re-run minutes long instead of weeks. import { DEFAULT_DIGEST_MAX_CUES_PER_CHUNK, DEFAULT_DIGEST_NUM_CTX, type DigestTimestampMode, } from "./digest"; export const PROMPT_VERSION = 2; // Version 1 -> 2 is a DEFAULTS change, and the bump is what makes it honest. // // The measured-best configuration (chunk-local timestamps, an 8192 context and // the 600-cue chunk sized to it) is now the default. digestPromptVariant() // derives its string RELATIVE TO THE DEFAULT CONSTANTS, so a value equal to the // default contributes nothing and the variant comes out `undefined`. That is // correct and stays as it is — but it means flipping the defaults alone would // have been silently destructive: the 39 pilot digests were generated under // absolute/1200 and recorded `promptVariant: undefined`, so under the new // defaults they would compare EQUAL and be skipped as fresh forever. Output from // the configuration the bake-off measured as worst would have been frozen into // the corpus, indistinguishable from output of the best one. // // isSectionFresh compares promptVersion, so bumping it invalidates every section // explicitly. That is the right tool for a default change; promptVariant is the // tool for comparing several shapes concurrently WITHOUT invalidating the // corpus, which is what a bake-off round needs. Cost here is 39 regenerations. // // (Version 1 also predates the addition of timestampMode. In "absolute" mode the // rendered prompt was byte-identical to what version 1 always produced, which is // why adding the knob did not itself require a bump.) // Chunking. ~6k usable tokens of transcript per call inside the default 8k // context leaves room for the prompt and the response. Expressed in CUES because // that is what the chunker slices; ~10 tokens/cue is the corpus average, so 600 // cues ≈ 6k tokens. The overlap exists so a topic straddling a seam is visible // whole to at least one call; the parser de-dups the resulting near-identical // chapters. export const DIGEST_MAX_CUES_PER_CHUNK = DEFAULT_DIGEST_MAX_CUES_PER_CHUNK; // Cues per chunk, SIZED TO THE CONFIGURED CONTEXT. // // The 600-cue default is sized for the default 8k window (~10 tokens/cue -> ~6k // tokens of transcript, leaving room for the prompt and the response). Lowering // `numCtx` without lowering this feeds the engine more transcript than its // window holds, and ollama TRUNCATES SILENTLY — the model then summarizes // whatever fragment survived and the result looks like a bad model rather than a // misconfiguration. That failure is measured (see FACTS.md, smoke test 1) and it // is why this derivation exists rather than a bare constant. // // Halving the context is close to free on throughput — measured 24.2 vs 25.1 // projected sweep days — because twice as many calls each carry half the prompt. export function maxCuesForContext(numCtx?: number): number { const ctx = numCtx && numCtx > 0 ? numCtx : DEFAULT_DIGEST_NUM_CTX; return Math.max( 100, Math.round((DIGEST_MAX_CUES_PER_CHUNK * ctx) / DEFAULT_DIGEST_NUM_CTX), ); } export const DIGEST_OVERLAP_CUES = 40; // Segmentation density target. 3 chapters for 22 minutes (the measured // hardened-prompt result) is too thin to be useful, so the prompt states an // explicit rate and the schema carries a matching minItems floor. This is the // open tuning question Stage B exists to settle — it is a knob, not a fact. export const DIGEST_MINUTES_PER_CHAPTER = 4; // Never demand more than this from one chunk, however long it is: an unreachable // minItems floor makes a constrained decoder pad with junk. export const DIGEST_MAX_CHAPTERS_PER_CHUNK = 24; // The regex that eliminated every malformed timestamp. Two digits per field, so // ":00:27" and "1:2:3" are both rejected by the decoder itself. export const HMS_PATTERN = "^[0-9][0-9]:[0-9][0-9]:[0-9][0-9]$"; // The parser's own copy of the same rule (guard 1): the schema pins it, but the // parser must still reject a bad stamp in case an engine ignores the schema. export const HMS_RE = /^[0-9][0-9]:[0-9][0-9]:[0-9][0-9]$/; export function hmsToSeconds(clock: string): number | null { if (!HMS_RE.test(clock)) return null; const [h, m, s] = clock.split(":").map(Number); if (m > 59 || s > 59) return null; return h * 3600 + m * 60 + s; } // Zero-padded HH:MM:SS. Deliberately NOT aiHandoff's hms(), which drops the hour // field for short videos ("2:36") — the schema pattern requires all three fields. export function toHms(totalSeconds: number): string { const n = Math.max(0, Math.floor(totalSeconds)); const h = Math.floor(n / 3600); const m = Math.floor((n % 3600) / 60); const s = n % 60; return [h, m, s].map((v) => String(v).padStart(2, "0")).join(":"); } export function minChaptersForSpan(spanSeconds: number): number { const byRate = Math.floor(spanSeconds / 60 / DIGEST_MINUTES_PER_CHAPTER); return Math.max(1, Math.min(DIGEST_MAX_CHAPTERS_PER_CHUNK, byRate)); } export function maxChaptersForSpan(spanSeconds: number): number { return Math.max( minChaptersForSpan(spanSeconds) + 2, Math.min( DIGEST_MAX_CHAPTERS_PER_CHUNK, Math.ceil(spanSeconds / 60 / Math.max(1, DIGEST_MINUTES_PER_CHAPTER - 2)), ), ); } // --------------------------------------------------------------------------- // Chapters // --------------------------------------------------------------------------- export type ChapterPromptInput = { title: string; channel: string; // The chunk's OWN range, in seconds, in REAL video time. Stating it in the // prompt is one of the three changes that fixed out-of-range output. startSeconds: number; endSeconds: number; // The transcript slice, already rendered as `[HH:MM:SS] text` lines by // transcriptToMarkdown with stampForCue. transcript: string; // Optional per-channel context note (Phase 1.5). Plumbed from the start so // adding notes later doesn't invalidate the corpus — see contextHash. contextNote?: string; // Optional cast list for THIS chunk, when the caller rendered speaker labels // into `transcript`. Experimental: set by the bake-off's speakers-on arm only, // never by the shipped lane. // // SEPARATE FROM contextNote ON PURPOSE, though both carry names. contextNote // is per-CHANNEL, hand-authored, and hashed into contextHash for freshness; a // roster is per-CHUNK and derived from that video's attribution.json. Folding // the roster into contextNote would mislabel it to the model ("context for // this channel"), and would make a channel note and a roster mutually // exclusive. // // ABSENT MUST RENDER BYTE-IDENTICALLY, which is what keeps this addition free // of a PROMPT_VERSION bump — the same argument that let timestampMode in. See // digestPrompt.test.ts. speakerRoster?: string; // Which numbering the caller rendered `transcript` with. Defaults to // "absolute". Under "chunk-local" the caller has re-based every marker to // 00:00:00, so the range this prompt states must be re-based to match — a // prompt that says "01:31:43 to 02:16:09" over markers that start at 00:00:00 // would contradict itself and is worse than either mode alone. timestampMode?: DigestTimestampMode; }; // The offset the caller subtracted from every marker, and that the parser must // add back. Zero in absolute mode by construction. export function promptOffsetSeconds(input: { startSeconds: number; timestampMode?: DigestTimestampMode; }): number { return input.timestampMode === "chunk-local" ? Math.max(0, Math.floor(input.startSeconds)) : 0; } export function chapterSchema(spanSeconds: number): Record { return { type: "object", properties: { chapters: { type: "array", minItems: minChaptersForSpan(spanSeconds), maxItems: maxChaptersForSpan(spanSeconds), items: { type: "object", properties: { // The pin. Every correctness failure measured before this existed. start: { type: "string", pattern: HMS_PATTERN }, title: { type: "string", minLength: 3, maxLength: 90 }, }, required: ["start", "title"], }, }, }, required: ["chapters"], }; } export const CHAPTER_SYSTEM_PROMPT = [ "You segment transcripts into chapters.", "You reply with JSON only, matching the provided schema exactly.", "Every title you write is in ENGLISH, regardless of the transcript's language.", "You never invent a timestamp: every start you emit is copied from a", "[HH:MM:SS] marker that appears in the transcript you were given.", ].join(" "); export function buildChapterPrompt(input: ChapterPromptInput): string { const offset = promptOffsetSeconds(input); const from = toHms(input.startSeconds - offset); const to = toHms(input.endSeconds - offset); const span = Math.max(0, input.endSeconds - input.startSeconds); const minItems = minChaptersForSpan(span); const lines: string[] = []; lines.push( `Below is one section of the transcript of "${input.title}" (${input.channel}).`, ); lines.push(""); lines.push( `This section covers ${from} to ${to} — ${Math.round(span / 60)} minutes of material.`, ); lines.push( `EVERY start you emit MUST be between ${from} and ${to} inclusive. A start outside`, `that range is wrong even if the topic is real. Copy starts from the [HH:MM:SS]`, "markers in the transcript; do not compute or estimate them.", ); lines.push(""); lines.push( `Aim for roughly one chapter per ${DIGEST_MINUTES_PER_CHAPTER} minutes of material —`, `at least ${minItems} for this section. A chapter marks where the subject genuinely`, "changes; do not split one continuous discussion into several chapters, and do not", "merge unrelated subjects into one.", ); lines.push(""); lines.push( "Each title is a specific, concrete English noun phrase naming what is discussed", '(e.g. "Court filing deadlines" — not "Discussion" or "Part two"). Do not use the', "speaker's own words as a quote, and do not editorialize.", ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel (use it for names and recurring topics):"); lines.push(input.contextNote.trim()); } if (input.speakerRoster?.trim()) { lines.push(""); lines.push(input.speakerRoster.trim()); // Without this the model titles sections after whoever is talking // ("Erica Kirk responds"), which is a worse table of contents than the // generic titles it replaces. A title names a SUBJECT. lines.push( "Use the speakers to tell apart who is talking and to spell names correctly.", "A title still names the SUBJECT under discussion, never the speaker.", ); } lines.push(""); lines.push("Transcript section:"); lines.push(""); lines.push(input.transcript); return lines.join("\n"); } // --------------------------------------------------------------------------- // Tags // --------------------------------------------------------------------------- export const TAG_MIN_ITEMS = 3; export const TAG_MAX_ITEMS = 12; export function tagSchema(): Record { return { type: "object", properties: { tags: { type: "array", minItems: TAG_MIN_ITEMS, maxItems: TAG_MAX_ITEMS, items: { type: "string", minLength: 2, maxLength: 40 }, }, }, required: ["tags"], }; } export const TAG_SYSTEM_PROMPT = [ "You extract topic tags from transcripts.", "You reply with JSON only, matching the provided schema exactly.", "Every tag is in ENGLISH, lowercase, and is a topic — not a sentence,", "not a summary, and not a person's opinion of the topic.", ].join(" "); export function buildTagPrompt(input: ChapterPromptInput): string { // Re-based the same way as the chapter prompt: the tag prompt sees the SAME // rendered transcript, so a range stated in the other numbering would // contradict the markers in front of it. const offset = promptOffsetSeconds(input); const lines: string[] = []; lines.push( `Below is one section of the transcript of "${input.title}" (${input.channel}),`, `covering ${toHms(input.startSeconds - offset)} to ${toHms(input.endSeconds - offset)}.`, ); lines.push(""); lines.push( `List between ${TAG_MIN_ITEMS} and ${TAG_MAX_ITEMS} lowercase English topic tags for`, "what this section is ABOUT. Prefer the specific over the generic: name the", "subject, event, or field, not the format of the video.", ); if (input.contextNote?.trim()) { lines.push(""); lines.push("Context for this channel:"); lines.push(input.contextNote.trim()); } lines.push(""); lines.push("Transcript section:"); lines.push(""); lines.push(input.transcript); return lines.join("\n"); }