import { test } from "node:test"; import assert from "node:assert/strict"; import { chmod, mkdir, mkdtemp, readFile, readdir, rm, writeFile, } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import { downloadOneManaged, sourceFetchFailure, type ManagedDownloadOpts, } from "./downloadOneManaged"; import type { DownloadOutcomeRecord } from "../lib/downloadOutcome"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/forceMedia.test.ts // // THE WHOLE-RECORDING FETCH DOWNLOADS ANYWAY (release 10 slice N). On a // youtube-handling channel the only pass that fetches media is attempt 3, the // no-subs fallback, gated on "no transcript and no captions" — so "Persist // source video" and `fetch_clip` with `full: true` were a silent no-op on any // video that already had a transcript: two subtitle passes, no file. // `forceMedia` opens the gate; with a transcript on disk the forced pass // persists the container and does nothing else. The yt-dlp here is a temp // node script that logs every argv and writes what the real one would; no // network. const ID = "H64QQZuw-aA"; const VIDEO = `https://www.youtube.com/watch?v=${ID}`; // The fake: the metadata prefetch writes meta.json's content as the info json; // the subtitle pass writes nothing (the transcript is seeded); the media pass // (the `source-media` output template) writes the container. Every spawn // appends its argv, one JSON line, to `argv.log`. const FAKE_YTDLP = `#!/usr/bin/env node const fs = require("node:fs"); const path = require("node:path"); const argv = process.argv.slice(2); const root = process.env.FAKE_ROOT; fs.appendFileSync(path.join(root, "argv.log"), JSON.stringify(argv) + "\\n"); const has = (f) => argv.includes(f); const arg = (f) => { const i = argv.indexOf(f); return i < 0 ? undefined : argv[i + 1]; }; const info = arg("--load-info-json"); const id = info ? path.basename(path.dirname(info)) : "${ID}"; const dir = path.join("data", id); fs.mkdirSync(dir, { recursive: true }); if (has("--skip-download") && has("--write-info-json") && has("--no-write-subs")) { fs.copyFileSync(path.join(root, "meta.json"), path.join(dir, "metadata.info.json")); process.exit(0); } if (has("--skip-download") && has("--write-auto-subs")) { console.log("DLOM_ARCHIVE youtube " + id); process.exit(0); } if ((arg("-o") || "").includes("source-media")) { if (fs.existsSync(path.join(root, "media-fail"))) { // A media pass that dies mid-download: the partial yt-dlp leaves, then a 403. fs.writeFileSync(path.join(dir, "source-media.f137.mp4.part"), "half a container"); console.error("ERROR: [download] Got error: HTTP Error 403: Forbidden"); process.exit(1); } fs.writeFileSync(path.join(dir, "source-media.mp4"), "container bytes for " + id); console.log("DLOM_ARCHIVE youtube " + id); process.exit(0); } console.error("fake: unknown invocation " + argv.join(" ")); process.exit(2); `; // ffmpeg for the app extraction: writes its last argument (the temp output). const FAKE_FFMPEG = `#!/bin/sh for last; do :; done echo "extracted audio" > "$last" `; type Run = { argvs: string[][]; record: Awaited>; log: string; videoDir: string; storeDir: string; files: string[]; // The channel's download archive after the run. archive: string; }; async function runWith(opts: { // The info json the prefetch writes. meta: Record; // Seed a transcript before the run. transcript?: { name: string; body: string }; // Seed an info json before the run (so the prefetch is a REWRITE). priorMeta?: Record; config?: Partial; managed?: Partial; // The media pass (the source-media output) fails with a 403 after writing a // partial. mediaFail?: boolean; }): Promise Promise }> { const root = await mkdtemp(path.join(tmpdir(), "force-media-")); const ytdlp = path.join(root, "fake-ytdlp.cjs"); const ffmpeg = path.join(root, "fake-ffmpeg.sh"); await writeFile(ytdlp, FAKE_YTDLP); await writeFile(ffmpeg, FAKE_FFMPEG); await chmod(ytdlp, 0o755); await chmod(ffmpeg, 0o755); await writeFile(path.join(root, "meta.json"), JSON.stringify(opts.meta)); if (opts.mediaFail) await writeFile(path.join(root, "media-fail"), ""); process.env.FAKE_ROOT = root; const paths = { transcriptsDir: root, channelsDir: path.join(root, "channels"), savedVideosDir: path.join(root, "saved-videos"), ytdlpBin: ytdlp, ffmpegBin: ffmpeg, ffprobeBin: path.join(root, "no-ffprobe"), } as Paths; const videoDir = path.join(paths.channelsDir, "c", "data", ID); await mkdir(videoDir, { recursive: true }); if (opts.transcript) { await writeFile(path.join(videoDir, opts.transcript.name), opts.transcript.body); } if (opts.priorMeta) { await writeFile( path.join(videoDir, "metadata.info.json"), JSON.stringify(opts.priorMeta), ); } let log = ""; const record = await downloadOneManaged({ channelSlug: "c", channelConfig: { handling: "youtube", url: "https://www.youtube.com/@c/videos", ...opts.config, } as ChannelConfig, paths, videoUrl: VIDEO, onLog: (s) => { log += s; }, signal: new AbortController().signal, ...opts.managed, }); const argvs = (await readFile(path.join(root, "argv.log"), "utf8")) .split("\n") .filter(Boolean) .map((l) => JSON.parse(l) as string[]); return { argvs, record, log, videoDir, storeDir: path.join(paths.savedVideosDir, "c", ID), files: (await readdir(videoDir)).sort(), archive: await readFile(path.join(paths.channelsDir, "c", "archive"), "utf8").catch(() => ""), cleanup: () => rm(root, { recursive: true, force: true }), }; } const META = { id: ID, title: "A video", duration: 60, webpage_url: VIDEO, subtitles: {}, automatic_captions: { en: [{ url: "https://example/cap" }] }, }; const TRANSCRIPT = { name: "transcript.json", body: '{"segments":[{"text":"kept"}]}' }; const FORCED: Partial = { keepSourceVideoOverride: true, forceMedia: true, persistOrigin: { requestedBy: "mcp", requestedAt: "2026-09-26T00:00:00.000Z" }, }; test("forceMedia with a transcript on disk: the source is fetched and persisted, nothing else", async () => { const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: FORCED }); try { // Prefetch, the subtitle pass as every re-download runs it, then the media. assert.equal(r.argvs.length, 3); assert.ok(r.argvs[1].includes("--skip-download")); const media = r.argvs[2]; assert.ok(!media.includes("--skip-download"), media.join(" ")); assert.ok(media.includes("--load-info-json")); assert.deepEqual( media.slice(media.indexOf("-f"), media.indexOf("-f") + 2), ["-f", "bestvideo*+bestaudio/best"], ); // The transcript cannot be rewritten, even by a channel's own args: the // refusals come after everything the channel passes. assert.ok(media.indexOf("--no-write-auto-subs") > media.indexOf("--print")); assert.ok(media.includes("--no-write-subs")); assert.match(r.log, /forceMedia: downloading the source although a transcript is on disk\n/); // The pointer landed and the container is in the store. const pointer = JSON.parse( await readFile(path.join(r.videoDir, "saved-video.json"), "utf8"), ) as Record; assert.equal(pointer.keepReason, "override"); assert.equal((pointer.origin as { requestedBy: string }).requestedBy, "mcp"); assert.equal( await readFile(path.join(r.storeDir, "source-media.mp4"), "utf8"), `container bytes for ${ID}`, ); // The transcript is byte-for-byte what it was, and no audio was made. assert.equal( await readFile(path.join(r.videoDir, TRANSCRIPT.name), "utf8"), TRANSCRIPT.body, ); assert.deepEqual( r.files.filter((f) => f.startsWith("audio.") || f.startsWith("source-media")), [], ); assert.match(r.log, /Persist only: a transcript is on disk, so no audio\.mp3 is extracted/); // Not a fallback to transcription, and a clean status. assert.equal(r.record.status, "ok"); assert.equal(r.record.fellBackToTranscribe, undefined); assert.deepEqual( r.record.attempts.map((a) => [a.kind, a.ytdlpExitCode]), [ ["metadata-prefetch", 0], ["primary", 0], ["no-subs-fallback", 0], ], ); } finally { await r.cleanup(); } }); test("without forceMedia the same video is untouched: two subtitle-shaped passes, no file (today's behaviour)", async () => { const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: { keepSourceVideoOverride: true }, }); try { assert.equal(r.argvs.length, 2); assert.ok(r.argvs.every((a) => a.includes("--skip-download"))); assert.ok(!r.files.includes("saved-video.json")); assert.doesNotMatch(r.log, /forceMedia/); } finally { await r.cleanup(); } }); test("forceMedia with captions listed but no transcript: today's fallback in full (download, extract, persist)", async () => { const r = await runWith({ meta: META, managed: FORCED }); try { assert.equal(r.argvs.length, 3); const media = r.argvs[2]; assert.ok(!media.includes("--skip-download")); // No transcript to protect, so no refusals — the args are the fallback's. assert.ok(!media.includes("--no-write-auto-subs")); assert.match(r.log, /forceMedia: downloading the source although captions exist\n/); assert.ok(r.files.includes("saved-video.json")); assert.ok(r.files.includes("audio.mp3"), r.files.join(",")); assert.equal(r.record.fellBackToTranscribe, true); } finally { await r.cleanup(); } }); test("no transcript and no captions: the fallback runs as it always has, forced or not", async () => { const noCaps = { ...META, automatic_captions: {} }; const r = await runWith({ meta: noCaps, managed: FORCED }); try { assert.equal(r.argvs.length, 3); assert.match(r.log, /No subs available for .*falling back to audio download \+ whisper \(keeping source video\)/); assert.doesNotMatch(r.log, /forceMedia/); assert.ok(r.files.includes("audio.mp3")); } finally { await r.cleanup(); } }); test("forceMedia with a transcript but a plan that persists nothing fetches nothing", async () => { const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: { forceMedia: true, keepSourceVideoOverride: false }, }); try { assert.equal(r.argvs.length, 2); assert.match(r.log, /forceMedia: a transcript is on disk and this download keeps no source video/); assert.equal(r.record.status, "ok"); } finally { await r.cleanup(); } }); test("the prefetch's rewrite of metadata.info.json lands in metadata.history.json, with who asked", async () => { const r = await runWith({ meta: { ...META, title: "Renamed upstream", view_count: 12 }, priorMeta: { ...META, view_count: 10 }, transcript: TRANSCRIPT, managed: FORCED, }); try { const h = JSON.parse( await readFile(path.join(r.videoDir, "metadata.history.json"), "utf8"), ) as { entries: Array> }; assert.equal(h.entries.length, 1); const e = h.entries[0]; assert.equal(e.by, "prefetch"); assert.equal(e.requestedBy, "mcp"); assert.deepEqual(e.changed, { title: { from: "A video", to: "Renamed upstream" } }); assert.deepEqual(e.counters, { view_count: [10, 12] }); } finally { await r.cleanup(); } }); // A FAILED MEDIA PASS OVER A TRANSCRIPT IS NOT A FAILED DOWNLOAD (release 11 // slice O3; release 10 slice N's review L2). The subtitle pass succeeded and the // transcript is on disk; the extra the caller asked for — the source — failed. // The status stays the subtitle pass's (the video page said "Download failed" // on a video whose transcript is fine), the failed attempt is recorded, and the // caller that asked for the source learns it through sourceFetchFailure. test("forceMedia over a transcript whose media pass fails: the download stays ok, only the media attempt failed", async () => { const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: FORCED, mediaFail: true }); try { assert.equal(r.argvs.length, 3); assert.equal(r.record.status, "ok"); assert.equal(r.record.failureClass, undefined); assert.deepEqual( r.record.attempts.map((a) => [a.kind, a.ytdlpExitCode]), [ ["metadata-prefetch", 0], ["primary", 0], ["no-subs-fallback", 1], ], ); assert.match(r.record.attempts[2].error ?? "", /HTTP Error 403: Forbidden/); assert.match(sourceFetchFailure(r.record) ?? "", /HTTP Error 403: Forbidden/); assert.match( r.log, /forceMedia: the source download failed \(yt-dlp exit 1\); the transcript on disk is untouched and the download stays ok\. Left for a retry to resume: source-media\.f137\.mp4\.part \(16 B\)\.\n/, ); // The transcript is untouched and still counts: the video stays in the // archive as the subtitle pass left it. assert.equal(await readFile(path.join(r.videoDir, TRANSCRIPT.name), "utf8"), TRANSCRIPT.body); assert.match(r.archive, new RegExp(`youtube ${ID}`)); // Nothing persisted; the partial is left for a retry to resume. assert.ok(!r.files.includes("saved-video.json")); assert.ok(r.files.includes("source-media.f137.mp4.part"), r.files.join(",")); // The sidecar on disk says the same as the returned record. const onDisk = JSON.parse( await readFile(path.join(r.videoDir, "download-outcome.json"), "utf8"), ) as DownloadOutcomeRecord; assert.equal(onDisk.status, "ok"); } finally { await r.cleanup(); } }); test("sourceFetchFailure: a failed download, a failed forced pass, and the clean cases", () => { const base = { videoId: ID, startedAt: "t0", finishedAt: "t1" }; const attempt = (kind: DownloadOutcomeRecord["attempts"][number]["kind"], code: number | null, error?: string) => ({ n: 1, kind, handling: "youtube" as const, usedCookies: false, ytdlpExitCode: code, ...(error ? { error } : {}), }); assert.equal( sourceFetchFailure({ ...base, status: "failed", attempts: [attempt("primary", 1, "ERROR: gone")] }), "ERROR: gone", ); assert.equal( sourceFetchFailure({ ...base, status: "failed-corrupt-source", attempts: [] }), "the download ended failed-corrupt-source", ); assert.equal( sourceFetchFailure({ ...base, status: "ok", attempts: [attempt("primary", 0), attempt("no-subs-fallback", 1)], }), "yt-dlp exited 1", ); assert.equal( sourceFetchFailure({ ...base, status: "ok", attempts: [attempt("primary", 0), attempt("no-subs-fallback", 0)], }), null, ); assert.equal( sourceFetchFailure({ ...base, status: "ok-with-cookies", attempts: [attempt("auth-retry", 0)] }), null, ); // The fallback fetched and persisted the source, then inline whisper failed // and ended the download `failed`: the SOURCE is there (review low 1). assert.equal( sourceFetchFailure({ ...base, status: "failed", attempts: [attempt("primary", 0), attempt("no-subs-fallback", 0)], }), null, ); });