import type { Cue } from "./vtt"; import { formatDate, formatDuration } from "./format"; // The subset of a transcripts-page record (TranscriptDetail, see transcripts.ts) // needed to render a self-contained markdown document. Kept as its own loose // type so callers on the client (which reconstruct records from shard JSON) and // on the server (MCP, build tools) can all feed it without importing the full // TranscriptDetail chain. export type TranscriptMarkdownInput = { id: string; title: string; channel?: string; channelSlug?: string; uploadDate?: string; // "YYYYMMDD" duration?: number; // seconds webpageUrl?: string; description?: string; tags?: string[]; cues?: Cue[] | undefined; }; export type TranscriptMarkdownOptions = { // Prefix each cue line with a [h:mm:ss] timestamp. Default true — timestamps // let an LLM cite a moment and let a reader jump to it. Turn off for the // cleanest possible prose block. timestamps?: boolean; // Include the video description section. Default true. includeDescription?: boolean; // Include a "Tags" line. Default false — tags are noisy for most Q&A. includeTags?: boolean; // Cap the number of cue lines emitted (for fitting a context window). When // truncated, a marker line is appended. Default: no cap. maxCues?: number; // Optional builder turning a cue's start seconds into a deep link. When it // returns a URL, the timestamp is rendered as a Markdown link // (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when // `timestamps` is false or when it returns null. See `momentUrl`. linkForCue?: (seconds: number) => string | null; // Optional formatter for the bracketed label's CONTENTS (e.g. "2:36|156" → // rendered as `[2:36|156] text`). Receives the pre-formatted clock and the // cue's raw start seconds. Takes precedence over `linkForCue`. Ignored when // `timestamps` is false. stampForCue?: (clock: string, seconds: number) => string; // Extra metadata lines rendered as `- ` after the tags line (e.g. // "moment_base: "). extraMeta?: string[]; // Optional per-line prefix inserted between the timestamp and the cue text // (`[0:12] Mr. Obvious: and that is ...`). Return null for "no prefix on this // line" — returning null for EVERY cue reproduces the un-prefixed line byte // for byte, which is what lets a speaker-aware caller share this renderer // instead of forking a second one. Pinned by a test in // transcriptToMarkdown.test.ts. // // DELIBERATELY NOT FOLDED INTO stampForCue. That formats the BRACKET's // contents, and the digest prompt instructs the model to copy the [HH:MM:SS] // markers verbatim while HMS_PATTERN rejects anything else — a speaker name // smuggled inside the bracket would make every start the model echoed // unparseable. The prefix has to live outside the bracket. prefixForCue?: (cue: Cue, index: number) => string | null; }; // [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so // handle the zero case explicitly here (a transcript's first cue is often 0s). function stamp(totalSeconds: number): string { const s = Math.max(0, Math.floor(totalSeconds)); return s === 0 ? "0:00" : formatDuration(s); } // Render a transcript record as a clean, self-contained markdown document: // a metadata header, an optional description, and the transcript body. This is // the single source of truth for "transcript → text for an AI" across the MCP // server, the in-browser copy buttons, and the chat retrieval context. export function transcriptToMarkdown( input: TranscriptMarkdownInput, options: TranscriptMarkdownOptions = {}, ): string { const { timestamps = true, includeDescription = true, includeTags = false, maxCues, linkForCue, stampForCue, prefixForCue, extraMeta, } = options; const lines: string[] = []; lines.push(`# ${input.title || input.id}`); lines.push(""); const meta: string[] = []; if (input.channel) meta.push(`- Channel: ${input.channel}`); if (input.uploadDate) meta.push(`- Uploaded: ${formatDate(input.uploadDate)}`); if (typeof input.duration === "number" && input.duration > 0) { meta.push(`- Duration: ${formatDuration(input.duration)}`); } meta.push(`- Video ID: ${input.id}`); if (input.webpageUrl) meta.push(`- Source: ${input.webpageUrl}`); if (includeTags && input.tags && input.tags.length > 0) { meta.push(`- Tags: ${input.tags.join(", ")}`); } for (const line of extraMeta ?? []) meta.push(`- ${line}`); lines.push(...meta); if (includeDescription && input.description && input.description.trim()) { lines.push(""); lines.push("## Description"); lines.push(""); lines.push(input.description.trim()); } lines.push(""); lines.push("## Transcript"); lines.push(""); const cues = input.cues; if (!cues || cues.length === 0) { lines.push("_(no transcript available)_"); return lines.join("\n") + "\n"; } const limit = typeof maxCues === "number" && maxCues >= 0 ? Math.min(maxCues, cues.length) : cues.length; for (let i = 0; i < limit; i++) { const cue = cues[i]; const rawText = cue.text.trim(); if (!rawText) continue; // Empty string and null both mean "nothing here", so a caller can return // either without changing the output. const prefix = prefixForCue ? prefixForCue(cue, i) : null; const text = prefix ? `${prefix}${rawText}` : rawText; if (!timestamps) { lines.push(text); continue; } if (stampForCue) { lines.push(`[${stampForCue(stamp(cue.start), cue.start)}] ${text}`); continue; } const url = linkForCue ? linkForCue(cue.start) : null; const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start); lines.push(`[${label}] ${text}`); } if (limit < cues.length) { lines.push(""); lines.push(`_(transcript truncated: showing ${limit} of ${cues.length} cues)_`); } return lines.join("\n") + "\n"; }