import { readFile } from "node:fs/promises"; import path from "node:path"; import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; import { momentKeyOf } from "yt-dlp-transcript-common/lib/citations/moments"; import { evidenceSpan, resolveEvidenceSource } from "yt-dlp-transcript-common/lib/evidenceClip-server"; import { reportMediaDir, reportMediaIndexFile } from "yt-dlp-transcript-common/publish/reportMedia"; import { readCues } from "@/lib/projects/report.mjs"; import { CHANNELS_DIR } from "@/lib/paths"; import { channelPosts, corpusMediaUrl, postShotFile, recordDir, recordMeta } from "./article"; import { sitesPaths } from "./sites"; // What a citation's EVIDENCE panel shows: the cited seconds with the transcript // around them, and something to play -- found, never fetched. // // Playback, best first: // 1. prepared the site's own evidence clip (.export-index/sites// // report-media/, what the build publishes), when prepare has run; // 2. window a clip window the editor fetched into data//clips/; // 3. saved a saved video or audio file in data// (through its media // tier link -- read, never written); // 4. otherwise nothing to play, and the line that fetches it through the // editor: the MCP `fetch_clip` tool. umtool's own fetch client // (api/report/fetch) is a manifest's, so it cannot ask for a // window no project names. Never yt-dlp. export const CONTEXT_CUES = 6; export type EvidenceCue = { start: number; end: number; text: string; cited: boolean }; export type EvidencePlay = { kind: "prepared" | "window" | "saved" | "audio"; url: string; /** Seconds into the file where the cited span starts. */ offset: number; audio: boolean; label: string; }; export type Evidence = { cite: string; kind: string; quote: string; speaker?: string; date?: string; label?: string; originalUrl?: string; record?: { channel: string; id: string; title?: string; channelTitle?: string }; start?: number; end?: number; cues: EvidenceCue[]; cuesNote?: string; play: EvidencePlay | null; fetchLine?: string; post?: { author?: string; text?: string; url?: string; shot?: string }; }; type Cite = NonNullable[string]; /** The ±CONTEXT_CUES cues around [start, end], the overlapping ones marked. */ export async function cueContext(channel: string, id: string, start: number, end: number) { const file = path.join(/* turbopackIgnore: true */ recordDir(channel, id), "transcript.cues.json"); const doc = (await readCues(file)) as { cues?: { start: number; end: number; text?: string }[] } | null; const cues = doc?.cues ?? []; if (!cues.length) return { cues: [] as EvidenceCue[], note: doc ? "no cues" : "no transcript.cues.json" }; const EPS = 0.05; let first = cues.findIndex((c) => c.end > start + EPS); if (first < 0) first = cues.length - 1; let last = first; while (last + 1 < cues.length && cues[last + 1].start < end - EPS) last += 1; const from = Math.max(0, first - CONTEXT_CUES); const to = Math.min(cues.length - 1, last + CONTEXT_CUES); return { cues: cues.slice(from, to + 1).map((c, i) => ({ start: c.start, end: c.end, text: String(c.text ?? "").replace(/\s+/g, " ").trim(), cited: from + i >= first && from + i <= last, })), }; } async function preparedClip(siteId: string, c: Cite): Promise { if (c.kind !== "video" && c.kind !== "audio") return null; const key = momentKeyOf(c); if (!key) return null; try { const index = JSON.parse(await readFile(/* turbopackIgnore: true */ reportMediaIndexFile(sitesPaths(), siteId), "utf8")); const entry = index?.moments?.[key]; if (!entry || (entry.kind !== "video" && entry.kind !== "audio") || typeof entry.file !== "string") return null; const abs = path.join(/* turbopackIgnore: true */ reportMediaDir(sitesPaths(), siteId), entry.file); let from = evidenceSpan(c).from; try { const side = JSON.parse(await readFile(/* turbopackIgnore: true */ abs.replace(/\.(mp4|m4a)$/, ".json"), "utf8")); if (typeof side?.span?.from === "number") from = side.span.from; } catch { // no sidecar: the citation's own pad is the best guess } return { kind: "prepared", url: `/api/sites/media?site=${encodeURIComponent(siteId)}&moment=${encodeURIComponent(key)}`, offset: Math.max(0, c.start - from), audio: entry.kind === "audio", label: "prepared evidence clip", }; } catch { return null; } } async function corpusClip(c: Cite): Promise { if (c.kind !== "video" && c.kind !== "audio") return null; const span = { from: c.start, to: c.end }; for (const audio of c.kind === "audio" ? [true] : [false, true]) { const hit = await resolveEvidenceSource({ channelsDir: CHANNELS_DIR, slug: c.channel, id: c.id, span, audio }).catch(() => null); if (!hit) continue; const kind = hit.kind === "corpus-window" ? "window" : hit.kind === "saved-video" ? "saved" : "audio"; return { kind, url: corpusMediaUrl(hit.path), offset: Math.max(0, c.start - hit.windowStart), audio: hit.kind === "audio", label: kind === "window" ? `clip window ${hit.name}` : kind === "saved" ? `saved ${hit.name}` : `audio ${hit.name}`, }; } return null; } /** The MCP line that fetches this span through the editor. */ export function fetchClipLine(c: { channel: string; id: string; start: number; end: number }, reason: string): string { const s = (n: number) => Number(n.toFixed(2)); return `fetch_clip ${JSON.stringify({ channel: c.channel, video: c.id, start: s(c.start), end: s(c.end), reason })}`; } export async function citationEvidence(siteId: string, report: Report, citeId: string): Promise { const c = report.citations?.[citeId]; if (!c) return null; const base: Evidence = { cite: citeId, kind: c.kind, quote: c.quote, ...(c.speaker ? { speaker: c.speaker } : {}), ...(c.date ? { date: c.date } : {}), ...(c.label ? { label: c.label } : {}), cues: [], play: null, }; if (c.kind === "video" || c.kind === "audio") { const meta = await recordMeta(c.channel, c.id); const ctx = await cueContext(c.channel, c.id, c.start, c.end); const play = (await preparedClip(siteId, c)) ?? (await corpusClip(c)); return { ...base, start: c.start, end: c.end, record: { channel: c.channel, id: c.id, ...(meta?.title ? { title: meta.title } : {}), ...(meta?.channelTitle ? { channelTitle: meta.channelTitle } : {}) }, ...(meta?.webpageUrl ? { originalUrl: meta.webpageUrl } : {}), cues: ctx.cues, ...(ctx.note ? { cuesNote: ctx.note } : {}), play, ...(play ? {} : { fetchLine: fetchClipLine(c, `${siteId}/${report.id} ${citeId}`) }), }; } if (c.kind === "post") { const post = (await channelPosts(c.channel)).get(c.id); const shot = await postShotFile(c.channel, c.id); return { ...base, record: { channel: c.channel, id: c.id }, ...(post?.url ? { originalUrl: post.url } : {}), post: { ...(post ? { author: post.authorName ?? post.author, text: post.text, url: post.url } : {}), ...(shot ? { shot: corpusMediaUrl(shot) } : {}), }, }; } if (c.kind === "page") return { ...base, originalUrl: c.url }; if (c.kind === "source") { const s = report.sources?.[c.source]; return { ...base, ...(s?.url ? { originalUrl: s.url } : {}), ...(s?.title ? { label: c.label ?? s.title } : {}) }; } return base; }