import { describe, it } from "node:test"; import assert from "node:assert/strict"; import { AUDIO_READ_PREFERENCE, audioFilesToRemove, findSourceMedia, isPartAudioFile, isRealAudioFile, isSourceMediaFile, isVideoContainer, pickPreferredAudio, } from "./mediaFiles"; // These predicates decide what a DELETE sweep removes and what ffmpeg is handed // as input, and they had never had a test. The fixtures below are not invented: // every "junk" name is a real file surveyed out of the 78,963-video corpus, and // audio.tmp-2760235.mp3 was observed live in a video dir while a transcode was // running. // Real, finalized audio in the corpus today (900 / 12 / 1 by count). const REAL_AUDIO = ["audio.mp3", "audio.mp4", "audio.aac"]; // The four junk names that the old prefix-plus-denylist admitted as audio, plus // this app's own transcode scratch, which an extension allowlist alone would // re-admit (its extension is perfectly valid) and which the cleanup sweep would // then race. const NOT_AUDIO = [ "audio.en-orig.vtt", // a SUBTITLE. No denylist rule excluded .vtt. "audio.live_chat.json.part-Frag114", // the .part rule used endsWith "audio.live_chat.json.part-Frag514", "audio.live_chat.json.part", "audio.live_chat.json", "audio.info.json", "audio.tmp-2760235.mp3", // controller/transcode.ts scratch, mid-transcode "audio.m4a.part", // an unfinished download, not a finalized file "audio.m4a.part.good", // audio-check snapshot "audio.m4a.part.testing", "audio.f251.webm.part", // yt-dlp per-format fragment "transcript.en.vtt", "metadata.info.json", "audio.", // degenerate "audio", ]; describe("isRealAudioFile", () => { it("accepts finalized audio. for every format on disk", () => { for (const name of REAL_AUDIO) { assert.equal(isRealAudioFile(name), true, name); } }); it("accepts wrong-format containers, because those are cleanup targets", () => { // bulk-actions.spec.ts renames audio.m4a -> audio.webm and requires the // sweep to delete it, so webm/mp4 must stay in the list even though they // are containers rather than audio-only formats. for (const name of ["audio.webm", "audio.mkv", "audio.m4a", "audio.opus"]) { assert.equal(isRealAudioFile(name), true, name); } }); it("rejects every non-media name that used to slip through", () => { for (const name of NOT_AUDIO) { assert.equal(isRealAudioFile(name), false, name); } }); it("is case-insensitive on the extension", () => { assert.equal(isRealAudioFile("audio.MP3"), true); }); }); describe("isSourceMediaFile / findSourceMedia", () => { it("accepts a finalized container", () => { assert.equal(isSourceMediaFile("source-media.mp4"), true); assert.equal(isSourceMediaFile("source-media.mkv"), true); }); it("rejects yt-dlp's postprocessor scratch", () => { // The ONLY source-media.* file in the whole corpus is this one, and it is // the corrupt 4 GiB truncated container. The old denylist had no .temp. // rule, so it read as finalized media — and finalizeAppExtraction would // transcode from it and then move it into the saved-video store as the // permanent copy. assert.equal(isSourceMediaFile("source-media.temp.mp4"), false); assert.equal(isSourceMediaFile("source-media.f137.mp4"), false); assert.equal(isSourceMediaFile("source-media.mp4.part"), false); assert.equal(isSourceMediaFile("source-media.info.json"), false); }); it("never picks the scratch file over the real container", () => { assert.equal( findSourceMedia(["source-media.temp.mp4", "source-media.mp4"]), "source-media.mp4", ); // Same answer whichever order readdir happened to return them in. assert.equal( findSourceMedia(["source-media.mp4", "source-media.temp.mp4"]), "source-media.mp4", ); }); it("returns null when only scratch is present", () => { assert.equal(findSourceMedia(["source-media.temp.mp4"]), null); }); }); describe("isPartAudioFile", () => { it("accepts a genuine interrupted audio download", () => { assert.equal(isPartAudioFile("audio.m4a.part"), true); assert.equal(isPartAudioFile("audio.webm.part"), true); }); it("rejects live-chat sidecars and audio-check snapshots", () => { for (const name of [ "audio.live_chat.json.part", "audio.live_chat.json.part-Frag114", "audio.m4a.part.good", "audio.m4a.part.testing", "audio.tmp-2760235.mp3.part", "audio.mp3", ]) { assert.equal(isPartAudioFile(name), false, name); } }); }); describe("audioFilesToRemove", () => { // The delete set. A real video dir mid-transcode, with a subtitle leak and a // live-chat fragment sitting next to the audio. const DIR = [ "audio.mp3", "audio.m4a", "audio.en-orig.vtt", "audio.live_chat.json.part-Frag114", "audio.info.json", "audio.tmp-2760235.mp3", "metadata.info.json", "transcript.en.vtt", ]; it("removes only the two real audio files", () => { assert.deepEqual(audioFilesToRemove(DIR), ["audio.mp3", "audio.m4a"]); }); it("keeps the channel's target format under wrongFormatOnly", () => { assert.deepEqual( audioFilesToRemove(DIR, { targetAudioFile: "audio.mp3", wrongFormatOnly: true, }), ["audio.m4a"], ); }); it("is a subset of the raw entries for any input (monotonicity)", () => { const out = audioFilesToRemove(DIR); for (const name of out) assert.ok(DIR.includes(name), name); }); }); describe("isVideoContainer", () => { it("keys off the extension, not the base name", () => { assert.equal(isVideoContainer("source-media.mkv"), true); assert.equal(isVideoContainer("audio.mp4"), true); assert.equal(isVideoContainer("audio.mp3"), false); assert.equal(isVideoContainer("noext"), false); }); }); describe("pickPreferredAudio", () => { it("prefers mp3 when reading", () => { assert.equal( pickPreferredAudio(["audio.opus", "audio.mp3", "audio.m4a"]), "audio.mp3", ); }); it("falls back to a stable sort, not readdir order", () => { assert.equal(pickPreferredAudio(["audio.wav", "audio.aac"]), "audio.aac"); assert.equal(pickPreferredAudio(["audio.aac", "audio.wav"]), "audio.aac"); }); it("returns null for no candidates", () => { assert.equal(pickPreferredAudio([]), null); }); });