import path from "node:path"; import { createReadStream } from "node:fs"; import { readFile } from "node:fs/promises"; import { englishVttsByPreference, type VideoFiles } from "./videoStatus"; // Where a VTT transcript came from: YouTube's speech recognition ("asr", what // yt-dlp downloads under --write-auto-subs) or a human-authored/uploaded track // ("manual", --write-subs). "unknown" means we could not tell, and is treated as // manual everywhere it matters — we never replace a transcript we can't prove is // machine-generated. // // The authoritative answer lives in metadata.info.json (`subtitles` vs // `automatic_captions`), but those files average ~490 KB, so parsing one per // video per snapshot regen is not viable across a 77k-video corpus. Instead we // sniff the first few KB of the VTT itself: YouTube's ASR cues carry // `align:start position:N%` cue settings and inline `` word // timings, and manual tracks carry neither. Measured on real data across 5 // channels (12 manual + 12 ASR): ASR files marked 96–100% of cues, manual files // 0%. The metadata parse is kept as the tie-breaker for the rare "unknown". export type SubtitleProvenance = "asr" | "manual" | "unknown"; // One read of this many bytes per English-VTT-having video per snapshot regen. // Big enough to hold several cues even for long, densely-marked ASR output. export const PROVENANCE_SNIFF_BYTES = 4096; // Below this many cues in the sniffed window there isn't enough evidence to // call it either way (a 2-cue stub could be anything). const MIN_CUES = 2; // Fraction of cues that must carry ASR fingerprints for an "asr" verdict. const ASR_RATIO = 0.5; const CUE_ARROW_RE = /-->/g; // yt-dlp writes YouTube's ASR cue settings verbatim: "align:start position:0%". const ASR_ALIGN_RE = /align:start position:\d+%/g; // Inline per-word timing tags, e.g. "<00:00:03.919>" — ASR-only in practice. const WORD_TIMING_RE = /<\d{2}:\d{2}:\d{2}\.\d{3}>/g; function count(re: RegExp, text: string): number { // Each call needs its own lastIndex reset — the /g regexes are module-level. re.lastIndex = 0; let n = 0; while (re.exec(text) !== null) n++; return n; } // Pure classifier over the first PROVENANCE_SNIFF_BYTES of a VTT file. Kept // separate from the I/O so it is directly unit-testable. export function sniffVttProvenance(head: string): SubtitleProvenance { const cues = count(CUE_ARROW_RE, head); if (cues < MIN_CUES) return "unknown"; const aligned = count(ASR_ALIGN_RE, head); const worded = count(WORD_TIMING_RE, head); if (aligned / cues >= ASR_RATIO || worded / cues >= ASR_RATIO) return "asr"; // No ASR fingerprint at all on a file with real cues: a human-authored track. if (aligned === 0 && worded === 0) return "manual"; // Some markers but not enough to clear the bar — refuse to guess. return "unknown"; } // Read only the head of the VTT (createReadStream start/end), never the whole // file: a 3-hour ASR transcript is multiple MB. export async function readVttProvenance( videoDir: string, vttFile: string, ): Promise { let head: string; try { head = await readHead(path.join(videoDir, vttFile), PROVENANCE_SNIFF_BYTES); } catch { return "unknown"; } return sniffVttProvenance(head); } function readHead(file: string, bytes: number): Promise { return new Promise((resolve, reject) => { const stream = createReadStream(file, { encoding: "utf8", start: 0, end: bytes - 1, }); let out = ""; stream.on("data", (chunk) => { out += chunk; }); stream.on("error", reject); stream.on("end", () => resolve(out)); }); } // The language/track segment of a transcript sidecar filename: // "transcript.en-US.vtt" -> "en-US". Null for anything else. export function vttTrackName(filename: string): string | null { const m = filename.match(/^transcript\.([^.]+)\.vtt$/); return m ? m[1] : null; } type CaptionMetadata = { subtitles?: Record; automatic_captions?: Record; }; // The authoritative check, from an already-parsed metadata.info.json. Mirrors // the shape read by runYtdlp/downloadOneManaged: `subtitles` is what YouTube // calls manual/uploaded captions, `automatic_captions` is ASR. A track listed in // both counts as manual (the manual file is what yt-dlp would have written). export function provenanceFromMetadata( meta: unknown, track: string, ): SubtitleProvenance { if (!meta || typeof meta !== "object") return "unknown"; const m = meta as CaptionMetadata; const subs = m.subtitles; const auto = m.automatic_captions; const has = (rec: Record | undefined): boolean => !!rec && typeof rec === "object" && Object.prototype.hasOwnProperty.call(rec, track); if (has(subs)) return "manual"; if (has(auto)) return "asr"; return "unknown"; } // Sniff first; fall back to the (expensive) metadata parse only when the sniff // can't decide. That keeps the 490 KB read rare while still classifying the odd // stub file correctly. export async function resolveVttProvenance( videoDir: string, vttFile: string, ): Promise { const sniffed = await readVttProvenance(videoDir, vttFile); if (sniffed !== "unknown") return sniffed; const track = vttTrackName(vttFile); if (!track) return "unknown"; try { const raw = await readFile(path.join(videoDir, "metadata.info.json"), "utf8"); return provenanceFromMetadata(JSON.parse(raw), track); } catch { return "unknown"; } } // Where a video's English CAPTIONS came from, taken over every English VTT and // not just the one the caption-track rule reads: "asr" only when each of them // is, "manual" when any is, else "unknown". The rule prefers the original-audio // ASR track (en-orig) even beside a human `en`, so asking only the track it // reads would call a video with human captions ASR-only and schedule them for // replacement. Single-track dirs pay the one sniff they always did. export async function resolveCaptionsProvenance( videoDir: string, entries: readonly string[], ): Promise { const vtts = englishVttsByPreference(entries); if (vtts.length === 0) return null; let unknown = false; for (const name of vtts) { const p = await resolveVttProvenance(videoDir, name); if (p === "manual") return "manual"; if (p === "unknown") unknown = true; } return unknown ? "unknown" : "asr"; } // True when this video's ONLY transcript is YouTube ASR — the work-lane // candidate rule, shared by the snapshot buckets, the whisper gate // (transcribeOneFromQueue) and the auto-runner's download override so all four // agree on what "auto-captions only" means. `files` is the already-read dir // listing; only the 4 KB VTT sniff is done here. export async function isAutoSubsOnly( videoDir: string, files: VideoFiles, ): Promise { if (!files.ytVttFile || files.hasWhisper || files.isUntranscribable) { return false; } return (await resolveCaptionsProvenance(videoDir, files.entries)) === "asr"; }