// ONE VIDEO'S CUES OFF DISK, WITHOUT AN INDEX (release 19 A7). // // A video imported or transcribed since the last index build — or one whose // transcript.cues.json was never written (a youtube-handling channel's // captions are only normalized by an explicit pass) — is invisible to every // reader of the published shards. This answers it from the video directory, // and writes nothing: // // cues.json transcript.cues.json, when it is fresh (isCuesJsonFresh — the // question every reader of it asks) // built what normalize WOULD write (buildNormalizedTranscript): the // metadata summary and the cues of the transcript the index // would pick, the caption-track rule for VTTs // vtt no metadata.info.json: the English VTT the rule picks, cues // only // // The ops transcript route serves it; the MCP's get_transcript falls back to it // through the editor when the archive has no such video. import path from "node:path"; import { readdir, stat } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import type { Cue } from "../lib/vtt"; import { assertChannelTextReadable } from "../lib/channelMedia"; import { readEnglishVttCues, readVideoFiles, CUES_JSON_FILENAME } from "../lib/videoStatus"; import { readChannelConfig } from "./channels"; import { buildNormalizedTranscript, isCuesJsonFresh, readNormalizedTranscript, type NormalizedTranscript, } from "./normalizeTranscript"; export type VideoCues = { slug: string; id: string; source: "cues.json" | "built" | "vtt"; // What transcript.cues.json says about itself: absent, stale, or fresh. cuesJson: "fresh" | "stale" | "missing"; title?: string; channel?: string; uploadDate?: string; duration?: number; webpageUrl?: string; platform?: string; description?: string; // The file the cues came from, when they came from a raw transcript. file?: string; cues: Cue[]; }; export type VideoCuesError = { error: string; status: 400 | 404 | 409 | 503 }; const ONE_SEGMENT = (s: string) => s !== "." && s !== ".." && !/[/\\\0]/.test(s) && s.length > 0; // The channel's held ids: its data// directories. An absent data/ is a // channel with nothing held; any other read error is thrown, never read as // "nothing held" (the text guard runs first). export async function heldVideoIds(dataDir: string): Promise> { try { const entries = await readdir(dataDir, { withFileTypes: true }); return new Set(entries.filter((d) => d.isDirectory() && !d.name.startsWith(".")).map((d) => d.name)); } catch (err) { if ((err as NodeJS.ErrnoException).code === "ENOENT") return new Set(); throw err; } } // The channels holding data//, for a read that was not told the channel. export async function channelsHoldingVideo(paths: Paths, id: string): Promise { if (!ONE_SEGMENT(id)) return []; const out: string[] = []; const entries = await readdir(paths.channelsDir, { withFileTypes: true }).catch(() => []); for (const e of entries) { if (!e.isDirectory() || e.name.startsWith(".")) continue; const st = await stat(path.join(paths.channelsDir, e.name, "data", id)).catch(() => null); if (st?.isDirectory()) out.push(e.name); } return out.sort(); } function fromNormalized( slug: string, id: string, t: NormalizedTranscript, source: VideoCues["source"], cuesJson: VideoCues["cuesJson"], ): VideoCues { const r = t as NormalizedTranscript & Record; const str = (v: unknown) => (typeof v === "string" && v ? v : undefined); return { slug, id, source, cuesJson, ...(str(r.title) ? { title: str(r.title) } : {}), ...(str(r.channel) ? { channel: str(r.channel) } : {}), ...(str(r.uploadDate) ? { uploadDate: str(r.uploadDate) } : {}), ...(typeof r.duration === "number" ? { duration: r.duration } : {}), ...(str(r.webpageUrl) ? { webpageUrl: str(r.webpageUrl) } : {}), ...(str(r.platform) ? { platform: str(r.platform) } : {}), ...(str(r.description) ? { description: str(r.description) } : {}), ...(t.vttFile ? { file: t.vttFile } : t.source === "whisper" ? { file: "transcript.json" } : {}), cues: t.cues ?? [], }; } export async function readVideoCues( paths: Paths, opts: { slug?: string; id: string }, ): Promise { const id = opts.id; if (!ONE_SEGMENT(id)) return { error: `"${id}" is not a video id (one path segment)`, status: 400 }; let slug = opts.slug; if (!slug) { const holders = await channelsHoldingVideo(paths, id); if (holders.length === 0) return { error: `no channel holds a video "${id}"`, status: 404 }; if (holders.length > 1) { return { error: `${holders.length} channels hold a video "${id}" (${holders.join(", ")}) — name the channel`, status: 409 }; } slug = holders[0]; } const config = await readChannelConfig(paths, slug); if (!config) return { error: `Channel "${slug}" not found`, status: 404 }; try { await assertChannelTextReadable(paths, slug, config); } catch (err) { return { error: (err as Error).message, status: 503 }; } const videoDir = path.join(paths.channelsDir, slug, "data", id); const st = await stat(videoDir).catch(() => null); if (!st?.isDirectory()) return { error: `Channel "${slug}" holds no video "${id}"`, status: 404 }; const files = await readVideoFiles(videoDir); const freshness = await isCuesJsonFresh(videoDir); const cuesJson: VideoCues["cuesJson"] = files.entries.includes(CUES_JSON_FILENAME) ? freshness.fresh ? "fresh" : "stale" : "missing"; if (cuesJson === "fresh") { const t = await readNormalizedTranscript(freshness.cuesPath); if (t) return fromNormalized(slug, id, t, "cues.json", cuesJson); } const built = await buildNormalizedTranscript({ videoDir, channelSlug: slug, configName: config.name }, files); if (built.status === "built") return fromNormalized(slug, id, built.transcript, "built", cuesJson); if (built.reason === "no-metadata") { const read = await readEnglishVttCues(videoDir, files.entries); if (read) return { slug, id, source: "vtt", cuesJson, file: read.filename, cues: read.cues }; } return { error: `${slug}/${id} has no transcript to read (no transcript.json and no English VTT)`, status: 404 }; }