// Detect badly truncated transcripts: when the audio download silently stopped // early (yt-dlp exits 0, download-outcome "ok"), whisper transcribes only the // few minutes that landed, so the transcript covers a tiny fraction of the real // runtime. We catch that by comparing the last cue's end time against the // recorded video duration. // // This module is pure (no fs) and is the single source of truth for the // threshold — the snapshot builder, the editor pages, and buildStats all call // it so the definition never drifts. import type { Cue } from "./vtt"; // A flagged video must be this long; short videos with a few cues are normal. export const INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC = 600; // 10 min // Flag when the transcript covers less than this fraction of the duration. export const INCOMPLETE_TRANSCRIPT_MAX_COVERAGE = 0.5; // < 50% covered export type TranscriptCoverage = { lastCueEnd: number; // max cue.end in seconds; 0 when there are no cues duration: number; // recorded video duration in seconds // lastCueEnd / duration. null when duration is unknown/zero (can't judge). // May exceed 1 when a cue end runs slightly past the recorded duration. coverage: number | null; }; export function transcriptCoverage( cues: ReadonlyArray> | undefined, duration: number, ): TranscriptCoverage { let lastCueEnd = 0; if (cues) { for (const c of cues) { const end = c.end ?? 0; if (end > lastCueEnd) lastCueEnd = end; } } const coverage = duration > 0 ? lastCueEnd / duration : null; return { lastCueEnd, duration, coverage }; } // The download-time analogue of isIncompleteTranscript: compare the DOWNLOADED // audio's actual duration to the video's metadata duration. A large shortfall // means the source served a truncated stream (e.g. a CDN-truncated HLS rung), // so the audio is short before whisper ever runs. Shares the same thresholds so // the two guards agree. Returns false (no judgement) for livestreams (unreliable // durations), short videos, or when either duration is unusable. export function isShortAudio( audioDurationSec: number | null, metaDurationSec: number | null | undefined, opts?: { isLivestream?: boolean }, ): boolean { if (opts?.isLivestream) return false; if ( typeof metaDurationSec !== "number" || !Number.isFinite(metaDurationSec) || metaDurationSec < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC ) { return false; } if ( audioDurationSec === null || !Number.isFinite(audioDurationSec) || audioDurationSec <= 0 ) { return false; } return audioDurationSec / metaDurationSec < INCOMPLETE_TRANSCRIPT_MAX_COVERAGE; } export function isIncompleteTranscript( cov: TranscriptCoverage, opts?: { isLivestream?: boolean }, ): boolean { // Livestream durations are unreliable (the recorded length often doesn't match // the captured stream), so don't judge coverage for them. if (opts?.isLivestream) return false; if (cov.duration < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC) return false; // No cues at all (lastCueEnd === 0) is an empty/untranscribable marker, not a // truncation — leave it to the untranscribable path rather than flag it here. if (cov.lastCueEnd <= 0) return false; if (cov.coverage === null) return false; return cov.coverage < INCOMPLETE_TRANSCRIPT_MAX_COVERAGE; }