import path from "node:path"; import { readdir, readFile, stat } from "node:fs/promises"; import { isPartAudioFile, isRealAudioFile } from "./mediaFiles"; import { parseVtt, type Cue } from "./vtt"; export type VideoFiles = { hasMeta: boolean; hasYtVtt: boolean; // The resolved primary English VTT filename by name (transcript.en-orig.vtt, // else transcript.en.vtt, else the best regional/auto English track — see // resolvePrimaryVtt; the cues may come from a later one, readEnglishVttCues). // Null when no English VTT exists. hasYtVtt === (ytVttFile !== null). ytVttFile: string | null; // True when a transcript..vtt exists under a non-canonical name (i.e. // anything other than transcript.en.vtt). Lets diagnostics surface videos // whose only/primary transcript rides on a non-standard VTT name. hasNonCanonicalVtt: boolean; hasWhisper: boolean; hasCuesJson: boolean; // A speaker-diarization sidecar is present. Existence only — the cleanup // guard in cleanAudioFromTranscribed validates the CONTENT before letting the // sweep delete audio, because a half-written file must not read as "done". // This cheaper flag is what the snapshot's cleanup buckets use. hasDiarization: boolean; isUntranscribable: boolean; audioFiles: string[]; // yt-dlp leaves audio..part on disk when a download is interrupted. // Surfaced separately so the UI can offer a Resume action; audio-check's // own internal snapshots (.part.good, .part.testing) are excluded. partAudioFiles: string[]; // The raw readdir() listing this whole record was derived from. // // Exposed so a caller that needs a question this type does not already answer // — "is there a source container?", "is there a saved-video pointer?" — can // ask it WITHOUT a second readdir. That matters in exactly one place and it is // the expensive one: the channel snapshot reads this per video across 77,000 // of them, so a per-video re-listing is the difference between an indicator // that is free and one that doubles the report's I/O. See // lib/operations.ts, which classifies missing vs missing-input from it. entries: string[]; }; export type IndexTranscript = | { kind: "whisper"; filename: "transcript.json" } // filename is the most preferred English track by name (resolvePrimaryVtt): // transcript.en-orig.vtt, transcript.en.vtt, or a regional/auto one such as // transcript.en-US.vtt. The cues are read with readEnglishVttCues. | { kind: "vtt"; filename: string }; export const VTT_FILENAME = "transcript.en.vtt"; export const WHISPER_FILENAME = "transcript.json"; export const CUES_JSON_FILENAME = "transcript.cues.json"; export const LIVE_CHAT_FILENAME = "transcript.live_chat.json"; export const LIVE_CHAT_CUES_FILENAME = "live_chat.cues.json"; export const META_FILENAME = "metadata.info.json"; // Speaker-diarization sidecar. Deliberately NOT named transcript..: // SUB_FILE_RE below would claim any such file as a subtitle track. See // lib/diarization.ts for the record shape. export const DIARIZATION_FILENAME = "diarization.json"; // Media-file naming and the predicates over it live in lib/mediaFiles.ts — // pure string rules, no filesystem — and are re-exported here so every existing // import of videoStatus keeps working. The split exists because the editor's // VideoPanel is a client component that needs the same extension lists. export { AUDIO_EXTS, VIDEO_CONTAINER_EXTS, MEDIA_EXTS, SOURCE_MEDIA_BASENAME, isRealAudioFile, audioFilesToRemove, isVideoContainer, isSourceMediaFile, findSourceMedia, isPartAudioFile, } from "./mediaFiles"; export type SubTrack = { // "live_chat" | language code like "es", "en-orig", etc. track: string; filename: string; ext: "vtt" | "json" | "json3" | "srv1" | "srv2" | "srv3"; }; // Treat any transcript.. file as a sub track when it isn't a transcript // (transcript.json, or ANY English VTT — the primary and its alternate tracks, // lib/captionTracks.ts) or a derived/auxiliary file (transcript.cues.json). Live chat lands as // transcript.live_chat.json; non-en languages as transcript..vtt. // Exported for lib/sidecar-server.ts, which refuses to declare a sidecar this // matches (a sidecar so named would be read as a subtitle track). export const SUB_FILE_RE = /^transcript\.([^.]+)\.([^.]+)$/; const SUB_EXT_VALUES = ["vtt", "json", "json3", "srv1", "srv2", "srv3"] as const; type SubExt = (typeof SUB_EXT_VALUES)[number]; function isSubExt(value: string): value is SubExt { return (SUB_EXT_VALUES as readonly string[]).includes(value); } // THE CAPTION-TRACK RULE — which English VTT a video's transcript is read from. // It lives here and nowhere else; every reader that turns captions into cues // (the index, normalize, report compose) goes through englishVttsByPreference / // readEnglishVttCues, and the MCP, the export and report-to-video read what // those wrote. // // A file counts as English iff the FIRST segment of its language code is `en`, // so translations like transcript.ab-en-US.vtt / transcript.es-en-US.vtt are // excluded. Preference: // // en-orig the captions of the ORIGINAL audio — the speaker's words. // en canonical. Usually the same text as en-orig, but for some videos // YouTube serves a rewritten/translated `en` that changes facts // (a date, a word), and for some livestream VODs one in a cue-block // shape with no word timing. // en-US, en-GB, … regional/manual. // en-en-* auto-translated en→en variants. // // AN OPERATOR'S PICK BEATS ALL OF IT. The editor's "Set as transcript" copies // the chosen track to transcript.en.vtt and writes TRANSCRIPT_PIN_FILENAME // beside it (lib/transcriptPin-server.ts); while that file is present, // transcript.en.vtt ranks first. Only its presence is read — the rule stays a // function of the listing. // // Name order is half the rule. The other half is CONTENT: the transcript is the // first track in this order that parses to at least one cue (readEnglishVttCues), // so a track that is present but empty never hides one that has text. // CAPTION_TRACK_RULE_VERSION names this rule; bump it when the order or the // fallback changes, and the next index build re-reads the records the change // can reach (buildIndex.ts, "Caption track"), and a cues.json normalized under // an older one reads as stale where it could differ (normalizeTranscript.ts). // umtool/report-to-video/cues.mjs carries a copy (plain node, no tsx); change // one, change the other — captionTrack.test.ts holds them equal. export const CAPTION_TRACK_RULE_VERSION = 1; export const ORIG_VTT_FILENAME = "transcript.en-orig.vtt"; // Not transcript..: SUB_FILE_RE would read it as a subtitle track. export const TRANSCRIPT_PIN_FILENAME = "transcript-pin.json"; const EN_VTT_RE = /^transcript\.(en(?:-[^.]+)?)\.vtt$/; function englishVttRank(track: string, pinned: boolean): number { if (track === "en" && pinned) return -1; if (track === "en-orig") return 0; if (track === "en") return 1; if (/^en-en(?:-|$)/.test(track)) return 3; // auto-translated en→en variants return 2; // regional/manual en-US, en-GB, … } // Whether a transcript sidecar filename is one of the ENGLISH VTT tracks // resolvePrimaryVtt considers (en, en-orig, en-US, en-en-*, …). Foreign-language // tracks — including translations like es-en-US — return false. Exported for the // superseded-auto-caption purge, which must leave non-English tracks alone. export function isEnglishVtt(name: string): boolean { return EN_VTT_RE.test(name); } // Every English VTT in a listing, most preferred first (ties by name, so the // order never depends on readdir's). export function englishVttsByPreference(entries: readonly string[]): string[] { const pinned = entries.includes(TRANSCRIPT_PIN_FILENAME); const ranked: { name: string; rank: number }[] = []; for (const e of entries) { const m = e.match(EN_VTT_RE); if (m) ranked.push({ name: e, rank: englishVttRank(m[1], pinned) }); } ranked.sort((a, b) => a.rank - b.rank || (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); return ranked.map((r) => r.name); } // The most preferred English VTT BY NAME — no file is read. This is the // track the index stats for change detection and the one the editor labels the // primary; the cues themselves come from readEnglishVttCues, which moves past // it when it parses to nothing. export function resolvePrimaryVtt(entries: readonly string[]): string | null { return englishVttsByPreference(entries)[0] ?? null; } // The files a caption transcript is derived from: every English VTT (the // content fallback can reach any of them) and the operator's pin. Their newest // mtime is what a cues.json, or an index record, is compared against. export function captionInputs(entries: readonly string[]): string[] { const out = englishVttsByPreference(entries); if (out.length > 0 && entries.includes(TRANSCRIPT_PIN_FILENAME)) { out.push(TRANSCRIPT_PIN_FILENAME); } return out; } // The cues of a video's caption transcript: the first English VTT, in // preference order, that parses to at least one cue. When every track parses // to nothing, the most preferred one is returned with its empty list (the // video has captions, they say nothing). Null when there is no English VTT, or // none can be read. `entries` is the caller's readdir of `videoDir`, when it // has one. export async function readEnglishVttCues( videoDir: string, entries?: readonly string[], ): Promise<{ filename: string; cues: Cue[] } | null> { const names = englishVttsByPreference( entries ?? (await readdir(videoDir).catch(() => [] as string[])), ); let first: { filename: string; cues: Cue[] } | null = null; for (const filename of names) { let cues: Cue[]; try { cues = parseVtt(await readFile(path.join(videoDir, filename), "utf8")); } catch { continue; } if (cues.length > 0) return { filename, cues }; first ??= { filename, cues }; } return first; } // Any transcript..vtt file (any language code). These are the candidate // transcript tracks a user can promote to the canonical transcript.en.vtt. const TRANSCRIPT_VTT_RE = /^transcript\.[^.]+\.vtt$/; export function isTranscriptVtt(name: string): boolean { return TRANSCRIPT_VTT_RE.test(name); } export function listTranscriptVtts(entries: string[]): string[] { return entries.filter(isTranscriptVtt).sort(); } // Whisper "empty transcription" outputs include systeminfo/model/params/result // keys around an empty `transcription: []`, so they can be up to ~620B in // practice. Real transcripts observed start at ~2.9KB. A 4KB cutoff lets us // stat() instead of reading multi-MB JSON for every transcribed video. const EMPTY_WHISPER_MAX_BYTES = 4096; export async function readVideoFiles( videoDir: string, opts: { checkUntranscribable?: boolean } = {}, ): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); const hasMeta = entries.includes(META_FILENAME); const ytVttFile = resolvePrimaryVtt(entries); const hasYtVtt = ytVttFile !== null; const hasNonCanonicalVtt = entries.some( (e) => isTranscriptVtt(e) && e !== VTT_FILENAME, ); const hasWhisper = entries.includes(WHISPER_FILENAME); const hasCuesJson = entries.includes(CUES_JSON_FILENAME); const hasDiarization = entries.includes(DIARIZATION_FILENAME); const audioFiles = entries.filter(isRealAudioFile); const partAudioFiles = entries.filter(isPartAudioFile); let isUntranscribable = false; if (hasWhisper && opts.checkUntranscribable) { const whisperPath = path.join(videoDir, WHISPER_FILENAME); try { const st = await stat(whisperPath); if (st.size <= EMPTY_WHISPER_MAX_BYTES) { const raw = await readFile(whisperPath, "utf8"); const parsed = JSON.parse(raw) as { transcription?: unknown[] }; isUntranscribable = Array.isArray(parsed.transcription) && parsed.transcription.length === 0; } } catch { isUntranscribable = false; } } return { hasMeta, hasYtVtt, ytVttFile, hasNonCanonicalVtt, hasWhisper, hasCuesJson, hasDiarization, isUntranscribable, audioFiles, partAudioFiles, entries, }; } export async function readSubTracks(videoDir: string): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); const tracks: SubTrack[] = []; for (const entry of entries) { if (entry === WHISPER_FILENAME) continue; // Every English VTT is a CAPTION track, not a sub track: the primary is // the transcript, and the others are its alternate tracks, kept only // where their words differ (lib/captionTracks.ts) — not shipped again, // identical or not, as subtitles. if (isEnglishVtt(entry)) continue; if (entry === CUES_JSON_FILENAME) continue; if (entry === LIVE_CHAT_CUES_FILENAME) continue; const m = entry.match(SUB_FILE_RE); if (!m) continue; const ext = m[2].toLowerCase(); if (!isSubExt(ext)) continue; tracks.push({ track: m[1], filename: entry, ext }); } return tracks; } export function pickIndexTranscript(files: VideoFiles): IndexTranscript | null { if (files.hasWhisper) return { kind: "whisper", filename: WHISPER_FILENAME }; if (files.ytVttFile) return { kind: "vtt", filename: files.ytVttFile }; return null; } export function isVideoTranscribed(files: VideoFiles): boolean { return files.hasWhisper || files.hasYtVtt; } // HAS THIS VIDEO EVER BEEN FETCHED? Any artifact, or even just the metadata a // prefetch left behind. The metadata scan and the snapshot must answer this the // same way or the advertised backlog and what Run actually fetches disagree — // which is how a scan ends up re-requesting videos the report says are done. export function isVideoFetched(files: VideoFiles): boolean { return isVideoDownloaded(files) || files.hasMeta; } export function isVideoDownloaded(files: VideoFiles): boolean { return ( files.hasWhisper || files.hasYtVtt || files.audioFiles.length > 0 ); } // A video's runtime in seconds from metadata.info.json, or null when there is no // metadata, it will not parse, or it carries no usable duration. // // DELIBERATELY NOT PART OF VideoFiles. That type carries no duration on purpose // — reusing the caller's readdir listing is what makes classification free // across 77,000 videos — and this reads and JSON-parses a file, which is several // orders of magnitude more expensive. Call it only after a cheap check has // already narrowed the population; the diarization cap calls it inside the // would-be-`missing` branch, where there are ~835 videos rather than 77,000. // // NULL IS "UNKNOWN", NEVER "ZERO". Every caller has to decide what to do without // an answer, and for the cap that decision is "do not defer" — the same // cheap-first rule verifyBeforeClean follows, where anything unresolved is left // alone rather than acted on. export async function readVideoDurationSec( videoDir: string, ): Promise { try { const raw = await readFile(path.join(videoDir, META_FILENAME), "utf8"); const meta = JSON.parse(raw) as { duration?: unknown }; const d = meta?.duration; if (typeof d !== "number" || !Number.isFinite(d) || d <= 0) return null; return d; } catch { return null; } } // One chapter as the UPLOADER authored it. Facts (`start`, `end`) and expression // (`title`) in one record, and callers are expected to treat them differently — // see readUploaderChapters. export type UploaderChapter = { start: number; end: number | null; title: string; }; // The uploader's own chapter marks from metadata.info.json, sorted by start, or // null when there is no metadata, it will not parse, or it carries no chapters. // // ALREADY ON DISK FOR THE WHOLE ARCHIVE, AND READ BY NOTHING ELSE. yt-dlp writes // `chapters` alongside `duration`, so this is a free signal for the ~19% of // videos that have it (measured: 14,923 video dirs carry a non-empty array, // 12,139 of those with >=4 chapters and a transcript). Nothing in the codebase // consumed it before boundary scoring did. // // THE TIMESTAMPS AND THE TITLES HAVE DIFFERENT STANDING. A chapter start is a // fact about where a subject changes; a chapter title is the uploader's own // expression. Boundary scoring uses the starts as ground truth, which reproduces // nothing. Admitting the titles into the corpus as our own chapter text would // republish the uploader's copy and is rejected on the grounds PLAN.md sets out. // Callers that touch `title` should be doing something like boilerplate // detection, not authorship. // // Same cost warning as readVideoDurationSec, more so: metadata.info.json runs to // ~100 KB and this parses all of it. Narrow the population with a cheap check // first — never map this over the whole archive in a render path. export async function readUploaderChapters( videoDir: string, ): Promise { try { const raw = await readFile(path.join(videoDir, META_FILENAME), "utf8"); const meta = JSON.parse(raw) as { chapters?: unknown }; if (!Array.isArray(meta?.chapters) || meta.chapters.length === 0) return null; const chapters: UploaderChapter[] = []; for (const entry of meta.chapters) { const c = entry as { start_time?: unknown; end_time?: unknown; title?: unknown }; // start_time is the only required field: a chapter with no start is not a // boundary and cannot be scored against. if (typeof c?.start_time !== "number" || !Number.isFinite(c.start_time)) continue; chapters.push({ start: c.start_time, end: typeof c.end_time === "number" && Number.isFinite(c.end_time) ? c.end_time : null, title: typeof c.title === "string" ? c.title : "", }); } if (chapters.length === 0) return null; return chapters.sort((a, b) => a.start - b.start); } catch { return null; } }