// Scan media already on disk for corruption. Reports; deletes nothing. // // ALMOST NONE OF THIS IS NEW CODE, and that is the point. The probes have // existed since the audio-checked download landed and are already standalone // (bin, file, signal) -> result functions with no download coupling. They were // simply never run against files that are ALREADY here, which is precisely the // gap: a container that arrived corrupt, or was truncated by a disk that filled, // is invisible until something tries to read it — and by then the audio may be // the only copy. // // TIERED, BECAUSE TIER 2 IS A FULL DECODE. probeAudioStream transcodes the whole // file; run over every audio-hour on disk that is hours of CPU, so it has to be // budgeted like the digest sweep rather than like a stat() walk. // // Tier 1 (cheap, always): ask ffprobe for the container's duration. Costs // milliseconds per file and catches today's known case outright — a // truncated mp4 fails with "moov atom not found" immediately. // Tier 2 (full decode, opt-in): only for files that PASS tier 1 but whose // duration disagrees with metadata.info.json. This tier is not optional for // correctness: ffmpegStreamProbe.ts records that a truncated container exits // 0 and classifies as `partial`, so comparing durations is the only way to // catch truncation at all. // // The rule from verifyBeforeClean.ts is kept: anything that cannot be resolved // is reported `unknown` and never acted on. A missing ffprobe makes this scan // say "I could not tell", not "these files are corrupt". // // DELETION IS A SEPARATE, EXPLICIT CLICK. The house convention here is report // first, delete second — not a dry-run flag on a deleting command. import path from "node:path"; import { readdir, readFile, stat } from "node:fs/promises"; import { writeJsonAtomic } from "../lib/jsonFile-server"; import { execa } from "execa"; import type { Paths } from "../lib/paths"; import { MEDIA_EXTS, isRealAudioFile, isSourceMediaFile, } from "../lib/mediaFiles"; import { META_FILENAME } from "../lib/videoStatus"; import { probeAudioStream } from "../ytdlp/ffmpegStreamProbe"; import { DEFAULT_DURATION_TOLERANCE_RATIO, DEFAULT_DURATION_TOLERANCE_SEC, MEDIA_SCAN_FILENAME, MEDIA_SCAN_OVERRIDES_FILENAME, MEDIA_SCAN_OVERRIDES_VERSION, MEDIA_SCAN_REPORT_VERSION, sanitizeMediaScanOverrides, type MediaScanChannelTotals, type MediaScanFinding, type MediaScanOverrides, type MediaScanReport, type MediaScanVerdict, } from "../lib/mediaScan"; const MEDIA_EXT_SET = new Set(MEDIA_EXTS); // ffprobe exited non-zero. Whether that means "this file is broken" or "I could // not run properly" is the difference between a finding and a false accusation, // so it is decided from what ffprobe actually said rather than from the exit // code alone. Anything unmatched falls through to `unknown`. const UNREADABLE_PATTERNS = [ /moov atom not found/i, /Invalid data found when processing input/i, /End of file/i, /Invalid argument/i, /could not find codec parameters/i, /Format .* detected only with low score/i, ]; export type ScanCorruptMediaOptions = { paths: Paths; // Channels to scan. Absent means every channel. channels?: string[]; // Run the tier-2 full decode on files whose duration is suspect. OFF by // default: it is a real transcode per file. deepProbe?: boolean; toleranceSeconds?: number; toleranceRatio?: number; onLog?: (msg: string) => void; onProgress?: (done: number, total: number) => void; signal?: AbortSignal; }; function emptyTotals(scannedAt: string): MediaScanChannelTotals { return { videosScanned: 0, filesScanned: 0, ok: 0, unreadable: 0, truncated: 0, stray: 0, unknown: 0, deepProbed: 0, bytesAtRisk: 0, scannedAt, }; } function fileExt(name: string): string { const dot = name.lastIndexOf("."); return dot >= 0 ? name.slice(dot + 1).toLowerCase() : ""; } // Every file in a video dir that CLAIMS to be media by its extension — // deliberately wider than isRealAudioFile/isSourceMediaFile. Those predicates // answer "is this one of our finished outputs?", and the answer for the corrupt // file this scan was written for (source-media.temp.mp4, a yt-dlp postprocessor // scratch, 4 GiB, unreadable) is NO. Scanning only recognized files would miss // exactly the population most likely to be broken. function mediaCandidates(entries: string[]): string[] { return entries.filter((e) => MEDIA_EXT_SET.has(fileExt(e))).sort(); } type ContainerProbe = { seconds: number | null; verdict: "ok" | "unreadable" | "unknown"; detail?: string; }; // Tier 1. Deliberately NOT probeMediaDurationSec: that helper collapses "the // container is unreadable" and "I could not run ffprobe" into the same null, // which is the one distinction this scan exists to make. async function probeContainer( ffprobeBin: string, file: string, signal?: AbortSignal, ): Promise { const args = [ "-v", "error", "-show_entries", "format=duration", "-of", "default=noprint_wrappers=1:nokey=1", file, ]; let result; try { result = await execa(ffprobeBin, args, { cancelSignal: signal, reject: false, }); } catch (err) { // ffprobe itself could not be run (missing binary, cancelled). Not a verdict // about the file. return { seconds: null, verdict: "unknown", detail: `ffprobe could not run: ${(err as Error).message}`, }; } const stderr = String(result.stderr ?? "").trim(); const firstLine = stderr.split("\n").find((l) => l.trim()) ?? ""; if (result.exitCode !== 0) { const unreadable = UNREADABLE_PATTERNS.some((re) => re.test(stderr)); return { seconds: null, verdict: unreadable ? "unreadable" : "unknown", detail: firstLine || `ffprobe exited ${result.exitCode}`, }; } const sec = Number.parseFloat(String(result.stdout).trim()); if (!Number.isFinite(sec) || sec <= 0) { return { seconds: null, verdict: "unknown", detail: firstLine || "ffprobe reported no usable duration", }; } return { seconds: sec, verdict: "ok" }; } async function readMetadataDuration(videoDir: string): Promise { try { const raw = await readFile(path.join(videoDir, META_FILENAME), "utf8"); const d = (JSON.parse(raw) as { duration?: unknown }).duration; return typeof d === "number" && Number.isFinite(d) && d > 0 ? d : null; } catch { return null; } } export async function scanCorruptMedia( opts: ScanCorruptMediaOptions, ): Promise { const log = opts.onLog ?? ((m: string) => console.log(m)); const deepProbe = opts.deepProbe === true; const toleranceSeconds = opts.toleranceSeconds ?? DEFAULT_DURATION_TOLERANCE_SEC; const toleranceRatio = opts.toleranceRatio ?? DEFAULT_DURATION_TOLERANCE_RATIO; const startedAt = Date.now(); const scannedAt = new Date().toISOString(); const allChannels = ( await readdir(opts.paths.channelsDir).catch(() => [] as string[]) ).sort(); const channels = opts.channels && opts.channels.length > 0 ? allChannels.filter((c) => opts.channels!.includes(c)) : allChannels; log( `Media scan over ${channels.length} channel(s), ` + (deepProbe ? "with the full-decode probe for suspect durations." : "container read only (the full-decode probe is off)."), ); const findings: MediaScanFinding[] = []; const totals: Record = {}; let done = 0; for (const channelSlug of channels) { if (opts.signal?.aborted) { log("Cancelled."); break; } const dataDir = path.join(opts.paths.channelsDir, channelSlug, "data"); const ids = (await readdir(dataDir).catch(() => [] as string[])).sort(); const t = emptyTotals(scannedAt); for (const videoId of ids) { if (opts.signal?.aborted) break; const videoDir = path.join(dataDir, videoId); const entries = await readdir(videoDir).catch(() => [] as string[]); const candidates = mediaCandidates(entries); if (candidates.length === 0) continue; t.videosScanned++; let metadataSeconds: number | null | undefined; for (const file of candidates) { if (opts.signal?.aborted) break; const full = path.join(videoDir, file); const bytes = await stat(full) .then((s) => s.size) .catch(() => 0); t.filesScanned++; const recognized = isRealAudioFile(file) || isSourceMediaFile(file); const push = ( verdict: Exclude, tier: 1 | 2, containerSeconds: number | null, detail?: string, ) => { t[verdict]++; t.bytesAtRisk += bytes; findings.push({ slug: `${channelSlug}/${videoId}`, channelSlug, videoId, file, bytes, verdict, tier, containerSeconds, metadataSeconds: metadataSeconds ?? null, ...(detail ? { detail: detail.slice(0, 300) } : {}), }); }; // Tier 1. const probe = await probeContainer( opts.paths.ffprobeBin, full, opts.signal, ); if (probe.verdict === "unreadable") { push("unreadable", 1, null, probe.detail); continue; } if (probe.verdict === "unknown") { push("unknown", 1, null, probe.detail); continue; } // Readable. An unrecognized name is reported as a leftover rather than // as damage — it decodes fine, it just is not one of ours. if (!recognized) { push( "stray", 1, probe.seconds, "Readable media the app does not recognize as one of its outputs (scratch or temp leftover).", ); continue; } if (metadataSeconds === undefined) { metadataSeconds = await readMetadataDuration(videoDir); } const expected = metadataSeconds; const actual = probe.seconds ?? 0; const suspect = expected !== null && expected - actual > Math.max(toleranceSeconds, expected * toleranceRatio); if (!suspect) { t.ok++; continue; } // Tier 2, and ONLY here. Without the deep probe this stays a suspicion, // reported honestly as one rather than promoted to a verdict the cheap // tier cannot support. if (!deepProbe) { push( "unknown", 1, probe.seconds, `Container is ${Math.round(expected - actual)}s shorter than the metadata; run the full-decode probe to confirm.`, ); continue; } t.deepProbed++; const deep = await probeAudioStream({ ffmpegBin: opts.paths.ffmpegBin, file: full, signal: opts.signal ?? new AbortController().signal, }); if (deep.verdict === "malformed") { push("unreadable", 2, probe.seconds, deep.stderr.split("\n")[0]); } else { // `clean` AND `partial` both land here: ffmpeg exits 0 on a truncated // container, so the duration gap — not ffmpeg's verdict — is what says // the file is short. push( "truncated", 2, probe.seconds, `Decodes, but ${Math.round(expected - actual)}s shorter than the metadata (ffmpeg: ${deep.verdict}).`, ); } } done++; if (done % 500 === 0) opts.onProgress?.(done, done); } totals[channelSlug] = t; if (t.filesScanned > 0) { log( `${channelSlug}: ${t.filesScanned} file(s) in ${t.videosScanned} video dir(s) — ` + `${t.ok} ok, ${t.unreadable} unreadable, ${t.truncated} truncated, ` + `${t.stray} stray, ${t.unknown} unknown.`, ); } } const report = await mergeAndWriteReport(opts.paths, { version: MEDIA_SCAN_REPORT_VERSION, generatedAt: scannedAt, runConfig: { deepProbe, toleranceSeconds, toleranceRatio }, channels: totals, findings, }); const flagged = report.findings.length; log( `Done: ${flagged} finding(s) across the scanned channels → ${MEDIA_SCAN_FILENAME} ` + `in ${Math.round((Date.now() - startedAt) / 1000)}s. Nothing was deleted.`, ); return report; } // A per-channel re-scan must not wipe every other channel's result, and a // corpus-wide scan must not leave a stale channel behind. So: replace the // scanned channels' totals and findings, keep the rest. async function mergeAndWriteReport( paths: Paths, fresh: MediaScanReport, ): Promise { const prior = await readMediaScanReport(paths); const scanned = new Set(Object.keys(fresh.channels)); const channels = { ...(prior?.channels ?? {}) }; for (const [slug, totals] of Object.entries(fresh.channels)) { channels[slug] = totals; } const kept = (prior?.findings ?? []).filter( (f) => !scanned.has(f.channelSlug), ); const merged: MediaScanReport = { ...fresh, channels, findings: [...kept, ...fresh.findings].sort( (a, b) => b.bytes - a.bytes || a.slug.localeCompare(b.slug) || a.file.localeCompare(b.file), ), }; const outPath = path.join(paths.transcriptsDir, MEDIA_SCAN_FILENAME); // Compact, no trailing newline: the report's historical bytes. await writeJsonAtomic(outPath, merged, { indent: 0, newline: false, mkdir: true }); return merged; } export async function readMediaScanReport( paths: Paths, ): Promise { try { const raw = await readFile( path.join(paths.transcriptsDir, MEDIA_SCAN_FILENAME), "utf8", ); const parsed = JSON.parse(raw) as MediaScanReport; if (typeof parsed.version !== "number") return null; if (!Array.isArray(parsed.findings)) return null; return parsed; } catch { return null; } } export function mediaScanOverridesPath(paths: Paths): string { return path.join(paths.transcriptsDir, MEDIA_SCAN_OVERRIDES_FILENAME); } export async function readMediaScanOverrides( paths: Paths, ): Promise { try { const raw = await readFile(mediaScanOverridesPath(paths), "utf8"); return sanitizeMediaScanOverrides(JSON.parse(raw)); } catch { return sanitizeMediaScanOverrides(null); } } // Record one finding as reviewed (or clear it), preserving every other decision. // Read-modify-write with the same atomic tmp+rename the report uses. export async function updateMediaScanOverride( paths: Paths, key: string, patch: { reviewed: boolean; note?: string }, ): Promise { const current = await readMediaScanOverrides(paths); const reviewed = { ...current.reviewed }; if (patch.reviewed) { reviewed[key] = { reviewedAt: new Date().toISOString(), ...(patch.note ? { note: patch.note } : {}), }; } else { delete reviewed[key]; } const out: MediaScanOverrides = { version: MEDIA_SCAN_OVERRIDES_VERSION, reviewed, }; const file = mediaScanOverridesPath(paths); await writeJsonAtomic(file, out, { mkdir: true }); return out; }