commit 19c214306feda8c4c433ca33bdebfc085393adfd
parent 6469058b22b8e3646f5954e8723f6892da7589aa
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 26 Sep 2026 16:42:53 -0400
common: forceMedia — "Persist source video", the whole-recording fetch and "Persist kept now" download the source on a youtube-handling video that already has a transcript or captions; with a transcript on disk the pass persists the container only (no subs, no audio, no whisper); the prefetch, the audio-check primary and download-one-audio run inside the metadata history wrap
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 499 insertions(+), 46 deletions(-)
diff --git a/common/controller/persistKept.ts b/common/controller/persistKept.ts
@@ -15,8 +15,9 @@ import { downloadOneManaged } from "../ytdlp/downloadOneManaged";
// after those videos had already been downloaded audio-only.
//
// For each kept video whose source isn't saved yet, it re-fetches the container
-// via downloadOneManaged with keepSourceVideoOverride (which app-extracts audio
-// and moves the container into the saved store), without disturbing the existing
+// via downloadOneManaged with keepSourceVideoOverride + forceMedia (which moves
+// the container into the saved store; a youtube-handling video that already has
+// a transcript gets no audio extracted), without disturbing the existing
// transcript. Mirrors redownloadToArchiveAction (videoActions.ts) but loops over
// the whole window. Videos already saved, or whose URL can't be resolved, are
// skipped rather than failing the pass.
@@ -130,6 +131,10 @@ export async function persistKept({
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
keepSourceVideoOverride: true,
+ // "Persist kept" means persist: on a youtube-handling channel the media
+ // pass is otherwise skipped for any video with a transcript or
+ // captions, which is every kept one (release 10 slice N).
+ forceMedia: true,
});
result.persisted += 1;
} catch (e) {
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -39,6 +39,7 @@ import {
} from "../lib/downloadOutcome";
import { writeDownloadOutcome } from "../lib/downloadOutcome-server";
import { recordAvailability } from "../lib/availability-server";
+import { withMetadataHistory } from "../lib/metadataHistory-server";
import {
loadRawMetadata,
loadRawMetadataFromDir,
@@ -141,6 +142,16 @@ export type ManagedDownloadOpts = {
// Per-run override: extract audio now and discard the container even for a
// video the keep-latest rule would otherwise persist (the save-disk backfill).
extractImmediately?: boolean;
+ // THE CALLER WANTS THE MEDIA (release 10 slice N): a transcript or captions
+ // on disk are not a reason to skip the download. Only youtube handling ever
+ // skipped it — its media pass is the no-subs fallback (attempt 3), gated on
+ // "no transcript and no captions" — so this opens that gate. With a
+ // transcript on disk the forced pass touches no transcript and extracts no
+ // audio: it persists the container and nothing else (see attempt 3). Set by
+ // "Persist source video", the whole-recording fetch and "Persist kept now";
+ // never by the download lane, re-acquire or import. Transcribe handling
+ // already downloads, so it is unaffected.
+ forceMedia?: boolean;
// Per-run override of channelConfig.audioFormat for the extracted audio.
audioFormatOverride?: AudioFormat;
// Resolved yt-dlp `-f` download-format preset (override > channel > global),
@@ -235,6 +246,11 @@ async function finalizeAppExtraction(opts: {
category: PersistenceDecision["category"];
// Threaded straight onto the pointer; see ManagedDownloadOpts.persistOrigin.
origin?: SavedVideoOrigin;
+ // PERSIST ONLY when false: move the container, extract nothing. A forced
+ // media download of a video that already has a transcript (attempt 3,
+ // `keepTranscript`) wants the source kept, and an audio.<fmt> beside an
+ // existing transcript is bytes nothing will read. Default true.
+ extractAudio?: boolean;
onLog: (s: string) => void;
signal: AbortSignal;
}): Promise<void> {
@@ -246,25 +262,37 @@ async function finalizeAppExtraction(opts: {
);
return;
}
- try {
- await transcodeAudio({
- paths: opts.paths,
- videoDir: opts.videoDir,
- sourceFilename: source,
- targetFormat: opts.fmt,
- onLog: opts.onLog,
- signal: opts.signal,
- });
- } catch (err) {
+ if (opts.extractAudio === false) {
opts.onLog(
- `App extraction failed (${source} -> audio.${opts.fmt}): ${(err as Error).message}. Keeping the source container.\n`,
+ `Persist only: a transcript is on disk, so no audio.${opts.fmt} is extracted from ${source}.\n`,
);
- return;
- }
- if (!opts.persist) {
- await rm(path.join(opts.videoDir, source), { force: true });
- opts.onLog(`Discarded source container ${source} (audio-only).\n`);
- return;
+ if (!opts.persist) {
+ // Unreachable from attempt 3 today (it does not download when the plan
+ // persists nothing), kept so the option means one thing on its own.
+ opts.onLog(`Not persisting ${source}; it stays in the data dir.\n`);
+ return;
+ }
+ } else {
+ try {
+ await transcodeAudio({
+ paths: opts.paths,
+ videoDir: opts.videoDir,
+ sourceFilename: source,
+ targetFormat: opts.fmt,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ } catch (err) {
+ opts.onLog(
+ `App extraction failed (${source} -> audio.${opts.fmt}): ${(err as Error).message}. Keeping the source container.\n`,
+ );
+ return;
+ }
+ if (!opts.persist) {
+ await rm(path.join(opts.videoDir, source), { force: true });
+ opts.onLog(`Discarded source container ${source} (audio-only).\n`);
+ return;
+ }
}
// Persist: move the container into the saved-video store + write a pointer.
const storeDir = savedVideoDir(
@@ -655,10 +683,21 @@ async function runManagedDownload(
"--",
opts.videoUrl,
];
- const prefetchRes = await runOneYtdlp(
- opts,
- channelDir,
- buildPrefetchArgs(prefetchCookies),
+ // EVERY PREFETCH REWRITES metadata.info.json (no existence check — the
+ // filters decide on today's metadata), so each spawn runs inside the
+ // history wrap: what moved since the last version is appended to
+ // metadata.history.json. The file yt-dlp writes is still the file.
+ const prefetchHistory = {
+ by: "prefetch" as const,
+ ...(opts.persistOrigin?.requestedBy
+ ? { requestedBy: opts.persistOrigin.requestedBy }
+ : {}),
+ onLog: opts.onLog,
+ };
+ const prefetchRes = await withMetadataHistory(
+ videoDir,
+ prefetchHistory,
+ () => runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchCookies)),
);
lastFullTail = prefetchRes.stderrTail;
const prefetchAvail = attemptSucceeded(prefetchRes.exitCode)
@@ -694,10 +733,11 @@ async function runManagedDownload(
opts.onLog(
`Metadata prefetch auth-required (${prefetchAvail}); retrying with --cookies-from-browser ${prefetchRetryCookies}\n`,
);
- const retryRes = await runOneYtdlp(
- opts,
- channelDir,
- buildPrefetchArgs(prefetchRetryCookies),
+ const retryRes = await withMetadataHistory(
+ videoDir,
+ prefetchHistory,
+ () =>
+ runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchRetryCookies)),
);
lastFullTail = retryRes.stderrTail;
const retryAvail = attemptSucceeded(retryRes.exitCode)
@@ -1027,16 +1067,34 @@ async function runManagedDownload(
// output to data/<canonicalId>/ for every platform, so the canonical id is
// the dir.
const expectedVideoIdHint = canonicalId;
- const audioOutcome = await runAudioCheckedYtdlp({
- paths: opts.paths,
- channelDir,
- channelConfig: opts.channelConfig,
- ytdlpArgs: primaryArgs,
- onLog: opts.onLog,
- signal: opts.signal,
- archiveMarker: ARCHIVE_MARKER,
- expectedVideoIdHint,
- });
+ const runAudioCheck = () =>
+ runAudioCheckedYtdlp({
+ paths: opts.paths,
+ channelDir,
+ channelConfig: opts.channelConfig,
+ ytdlpArgs: primaryArgs,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ archiveMarker: ARCHIVE_MARKER,
+ expectedVideoIdHint,
+ });
+ // The audio-checked primary writes the info json itself (it re-extracts on
+ // purpose), so it is the second rewrite of the file in one download and
+ // gets its own history entry. Only when the dir is known up front — an
+ // unidentifiable URL lands in data/%(id)s/, found only afterwards.
+ const audioOutcome = canonicalId
+ ? await withMetadataHistory(
+ path.join(channelDir, "data", canonicalId),
+ {
+ by: "audio-check",
+ ...(opts.persistOrigin?.requestedBy
+ ? { requestedBy: opts.persistOrigin.requestedBy }
+ : {}),
+ onLog: opts.onLog,
+ },
+ runAudioCheck,
+ )
+ : await runAudioCheck();
primaryRes = {
exitCode: audioOutcome.ytdlpExitCode,
stderrTail: audioOutcome.stderrTail,
@@ -1225,6 +1283,11 @@ async function runManagedDownload(
}
// ---------- Attempt 3: no-subs fallback (youtube handling only) ----------
+ // Also the youtube-handling MEDIA download a `forceMedia` caller asked for:
+ // this is the only pass that fetches media for a youtube-handling channel,
+ // and its own gate ("no transcript and no captions") is exactly the reason
+ // "Persist source video" and the whole-recording fetch used to do nothing on
+ // a video that already had a transcript — two subtitle passes, no file.
if (
lastSucceeded &&
opts.channelConfig.handling === "youtube" &&
@@ -1232,10 +1295,40 @@ async function runManagedDownload(
) {
const hasTranscript = await hasAnyTranscriptOnDisk(videoDir);
const noCaptions = await metadataReportsNoCaptions(videoDir);
- if (!hasTranscript && noCaptions) {
+ const noSubsFallback = !hasTranscript && noCaptions;
+ // Forced only when the fallback would NOT have run on its own: a video with
+ // no transcript and no captions takes today's path whoever asked.
+ const forced = !noSubsFallback && opts.forceMedia === true;
+ // THE TRANSCRIPT ON DISK IS NEVER TOUCHED. A forced pass over a video that
+ // has one fetches the source and persists it, and does nothing else: no
+ // subtitles (refused on the command line, after the channel's own args, so
+ // a channel carrying --write-auto-subs cannot overwrite the transcript), no
+ // audio extraction, no transcription, no short-audio verdict on audio it
+ // never made. With NO transcript on disk (captions listed, none fetched —
+ // a language the channel's sub-langs does not match) the forced pass is
+ // today's fallback in full: download, extract, transcribe inline if
+ // configured, persist.
+ const keepTranscript = forced && hasTranscript;
+ if (keepTranscript && !plan.persist) {
+ // PERSIST-ONLY WITH NOTHING TO PERSIST. Every forceMedia caller sets the
+ // keep-source override, so the plan persists; a caller that did not would
+ // be asking for a download with no destination — the transcript rules
+ // out audio, and the plan rules out the container.
opts.onLog(
- `No subs available for ${videoId}; falling back to audio download + whisper${plan.persist ? " (keeping source video)" : ""}.\n`,
+ `forceMedia: a transcript is on disk and this download keeps no source video (${plan.reason}); nothing to fetch.\n`,
);
+ } else if (noSubsFallback || forced) {
+ if (forced) {
+ opts.onLog(
+ `forceMedia: downloading the source although ${
+ hasTranscript ? "a transcript is on disk" : "captions exist"
+ }\n`,
+ );
+ } else {
+ opts.onLog(
+ `No subs available for ${videoId}; falling back to audio download + whisper${plan.persist ? " (keeping source video)" : ""}.\n`,
+ );
+ }
const fallbackConfig: ChannelConfig = {
...opts.channelConfig,
handling: "transcribe",
@@ -1267,6 +1360,9 @@ async function runManagedDownload(
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
...channelConfigArgs(fallbackConfig, fallbackCookieOverride),
+ // yt-dlp takes the LAST occurrence of an option, so the refusals go
+ // after the channel's own args (the chat-only pass's rule).
+ ...(keepTranscript ? ["--no-write-subs", "--no-write-auto-subs"] : []),
...sourceArgs(
opts.videoUrl,
path.join(videoDir, "metadata.info.json"),
@@ -1290,7 +1386,27 @@ async function runManagedDownload(
});
if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine;
- if (attemptSucceeded(fallbackRes.exitCode)) {
+ if (attemptSucceeded(fallbackRes.exitCode) && keepTranscript) {
+ // Persist only. Not a fallback to transcription — the transcript is
+ // the one already on disk — so `fellBackToTranscribe` stays unset and
+ // the status stays what the subtitle pass made it.
+ if (plan.extractionMode === "app") {
+ await finalizeAppExtraction({
+ paths: opts.paths,
+ channelSlug: opts.channelSlug,
+ channelConfig: opts.channelConfig,
+ videoDir,
+ videoId,
+ fmt,
+ persist: plan.persist,
+ category: plan.category,
+ origin: opts.persistOrigin,
+ extractAudio: false,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ }
+ } else if (attemptSucceeded(fallbackRes.exitCode)) {
fellBackToTranscribe = true;
// In app mode, extract audio.<fmt> from the downloaded source container
// (and keep/discard it) before any inline whisper can read the audio.
diff --git a/common/ytdlp/forceMedia.test.ts b/common/ytdlp/forceMedia.test.ts
@@ -0,0 +1,302 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ chmod,
+ mkdir,
+ mkdtemp,
+ readFile,
+ readdir,
+ rm,
+ writeFile,
+} from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { downloadOneManaged, type ManagedDownloadOpts } from "./downloadOneManaged";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/forceMedia.test.ts
+//
+// THE WHOLE-RECORDING FETCH DOWNLOADS ANYWAY (release 10 slice N). On a
+// youtube-handling channel the only pass that fetches media is attempt 3, the
+// no-subs fallback, gated on "no transcript and no captions" — so "Persist
+// source video" and `fetch_clip` with `full: true` were a silent no-op on any
+// video that already had a transcript: two subtitle passes, no file.
+// `forceMedia` opens the gate; with a transcript on disk the forced pass
+// persists the container and does nothing else. The yt-dlp here is a temp
+// node script that logs every argv and writes what the real one would; no
+// network.
+
+const ID = "H64QQZuw-aA";
+const VIDEO = `https://www.youtube.com/watch?v=${ID}`;
+
+// The fake: the metadata prefetch writes meta.json's content as the info json;
+// the subtitle pass writes nothing (the transcript is seeded); the media pass
+// (the `source-media` output template) writes the container. Every spawn
+// appends its argv, one JSON line, to `argv.log`.
+const FAKE_YTDLP = `#!/usr/bin/env node
+const fs = require("node:fs");
+const path = require("node:path");
+const argv = process.argv.slice(2);
+const root = process.env.FAKE_ROOT;
+fs.appendFileSync(path.join(root, "argv.log"), JSON.stringify(argv) + "\\n");
+const has = (f) => argv.includes(f);
+const arg = (f) => { const i = argv.indexOf(f); return i < 0 ? undefined : argv[i + 1]; };
+const info = arg("--load-info-json");
+const id = info ? path.basename(path.dirname(info)) : "${ID}";
+const dir = path.join("data", id);
+fs.mkdirSync(dir, { recursive: true });
+if (has("--skip-download") && has("--write-info-json") && has("--no-write-subs")) {
+ fs.copyFileSync(path.join(root, "meta.json"), path.join(dir, "metadata.info.json"));
+ process.exit(0);
+}
+if (has("--skip-download") && has("--write-auto-subs")) {
+ console.log("DLOM_ARCHIVE youtube " + id);
+ process.exit(0);
+}
+if ((arg("-o") || "").includes("source-media")) {
+ fs.writeFileSync(path.join(dir, "source-media.mp4"), "container bytes for " + id);
+ console.log("DLOM_ARCHIVE youtube " + id);
+ process.exit(0);
+}
+console.error("fake: unknown invocation " + argv.join(" "));
+process.exit(2);
+`;
+
+// ffmpeg for the app extraction: writes its last argument (the temp output).
+const FAKE_FFMPEG = `#!/bin/sh
+for last; do :; done
+echo "extracted audio" > "$last"
+`;
+
+type Run = {
+ argvs: string[][];
+ record: Awaited<ReturnType<typeof downloadOneManaged>>;
+ log: string;
+ videoDir: string;
+ storeDir: string;
+ files: string[];
+};
+
+async function runWith(opts: {
+ // The info json the prefetch writes.
+ meta: Record<string, unknown>;
+ // Seed a transcript before the run.
+ transcript?: { name: string; body: string };
+ // Seed an info json before the run (so the prefetch is a REWRITE).
+ priorMeta?: Record<string, unknown>;
+ config?: Partial<ChannelConfig>;
+ managed?: Partial<ManagedDownloadOpts>;
+}): Promise<Run & { cleanup: () => Promise<void> }> {
+ const root = await mkdtemp(path.join(tmpdir(), "force-media-"));
+ const ytdlp = path.join(root, "fake-ytdlp.cjs");
+ const ffmpeg = path.join(root, "fake-ffmpeg.sh");
+ await writeFile(ytdlp, FAKE_YTDLP);
+ await writeFile(ffmpeg, FAKE_FFMPEG);
+ await chmod(ytdlp, 0o755);
+ await chmod(ffmpeg, 0o755);
+ await writeFile(path.join(root, "meta.json"), JSON.stringify(opts.meta));
+ process.env.FAKE_ROOT = root;
+ const paths = {
+ transcriptsDir: root,
+ channelsDir: path.join(root, "channels"),
+ savedVideosDir: path.join(root, "saved-videos"),
+ ytdlpBin: ytdlp,
+ ffmpegBin: ffmpeg,
+ ffprobeBin: path.join(root, "no-ffprobe"),
+ } as Paths;
+ const videoDir = path.join(paths.channelsDir, "c", "data", ID);
+ await mkdir(videoDir, { recursive: true });
+ if (opts.transcript) {
+ await writeFile(path.join(videoDir, opts.transcript.name), opts.transcript.body);
+ }
+ if (opts.priorMeta) {
+ await writeFile(
+ path.join(videoDir, "metadata.info.json"),
+ JSON.stringify(opts.priorMeta),
+ );
+ }
+ let log = "";
+ const record = await downloadOneManaged({
+ channelSlug: "c",
+ channelConfig: {
+ handling: "youtube",
+ url: "https://www.youtube.com/@c/videos",
+ ...opts.config,
+ } as ChannelConfig,
+ paths,
+ videoUrl: VIDEO,
+ onLog: (s) => {
+ log += s;
+ },
+ signal: new AbortController().signal,
+ ...opts.managed,
+ });
+ const argvs = (await readFile(path.join(root, "argv.log"), "utf8"))
+ .split("\n")
+ .filter(Boolean)
+ .map((l) => JSON.parse(l) as string[]);
+ return {
+ argvs,
+ record,
+ log,
+ videoDir,
+ storeDir: path.join(paths.savedVideosDir, "c", ID),
+ files: (await readdir(videoDir)).sort(),
+ cleanup: () => rm(root, { recursive: true, force: true }),
+ };
+}
+
+const META = {
+ id: ID,
+ title: "A video",
+ duration: 60,
+ webpage_url: VIDEO,
+ subtitles: {},
+ automatic_captions: { en: [{ url: "https://example/cap" }] },
+};
+const TRANSCRIPT = { name: "transcript.json", body: '{"segments":[{"text":"kept"}]}' };
+const FORCED: Partial<ManagedDownloadOpts> = {
+ keepSourceVideoOverride: true,
+ forceMedia: true,
+ persistOrigin: { requestedBy: "mcp", requestedAt: "2026-09-26T00:00:00.000Z" },
+};
+
+test("forceMedia with a transcript on disk: the source is fetched and persisted, nothing else", async () => {
+ const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: FORCED });
+ try {
+ // Prefetch, the subtitle pass as every re-download runs it, then the media.
+ assert.equal(r.argvs.length, 3);
+ assert.ok(r.argvs[1].includes("--skip-download"));
+ const media = r.argvs[2];
+ assert.ok(!media.includes("--skip-download"), media.join(" "));
+ assert.ok(media.includes("--load-info-json"));
+ assert.deepEqual(
+ media.slice(media.indexOf("-f"), media.indexOf("-f") + 2),
+ ["-f", "bestvideo*+bestaudio/best"],
+ );
+ // The transcript cannot be rewritten, even by a channel's own args: the
+ // refusals come after everything the channel passes.
+ assert.ok(media.indexOf("--no-write-auto-subs") > media.indexOf("--print"));
+ assert.ok(media.includes("--no-write-subs"));
+ assert.match(r.log, /forceMedia: downloading the source although a transcript is on disk\n/);
+
+ // The pointer landed and the container is in the store.
+ const pointer = JSON.parse(
+ await readFile(path.join(r.videoDir, "saved-video.json"), "utf8"),
+ ) as Record<string, unknown>;
+ assert.equal(pointer.keepReason, "override");
+ assert.equal((pointer.origin as { requestedBy: string }).requestedBy, "mcp");
+ assert.equal(
+ await readFile(path.join(r.storeDir, "source-media.mp4"), "utf8"),
+ `container bytes for ${ID}`,
+ );
+ // The transcript is byte-for-byte what it was, and no audio was made.
+ assert.equal(
+ await readFile(path.join(r.videoDir, TRANSCRIPT.name), "utf8"),
+ TRANSCRIPT.body,
+ );
+ assert.deepEqual(
+ r.files.filter((f) => f.startsWith("audio.") || f.startsWith("source-media")),
+ [],
+ );
+ assert.match(r.log, /Persist only: a transcript is on disk, so no audio\.mp3 is extracted/);
+ // Not a fallback to transcription, and a clean status.
+ assert.equal(r.record.status, "ok");
+ assert.equal(r.record.fellBackToTranscribe, undefined);
+ assert.deepEqual(
+ r.record.attempts.map((a) => [a.kind, a.ytdlpExitCode]),
+ [
+ ["metadata-prefetch", 0],
+ ["primary", 0],
+ ["no-subs-fallback", 0],
+ ],
+ );
+ } finally {
+ await r.cleanup();
+ }
+});
+
+test("without forceMedia the same video is untouched: two subtitle-shaped passes, no file (today's behaviour)", async () => {
+ const r = await runWith({
+ meta: META,
+ transcript: TRANSCRIPT,
+ managed: { keepSourceVideoOverride: true },
+ });
+ try {
+ assert.equal(r.argvs.length, 2);
+ assert.ok(r.argvs.every((a) => a.includes("--skip-download")));
+ assert.ok(!r.files.includes("saved-video.json"));
+ assert.doesNotMatch(r.log, /forceMedia/);
+ } finally {
+ await r.cleanup();
+ }
+});
+
+test("forceMedia with captions listed but no transcript: today's fallback in full (download, extract, persist)", async () => {
+ const r = await runWith({ meta: META, managed: FORCED });
+ try {
+ assert.equal(r.argvs.length, 3);
+ const media = r.argvs[2];
+ assert.ok(!media.includes("--skip-download"));
+ // No transcript to protect, so no refusals — the args are the fallback's.
+ assert.ok(!media.includes("--no-write-auto-subs"));
+ assert.match(r.log, /forceMedia: downloading the source although captions exist\n/);
+ assert.ok(r.files.includes("saved-video.json"));
+ assert.ok(r.files.includes("audio.mp3"), r.files.join(","));
+ assert.equal(r.record.fellBackToTranscribe, true);
+ } finally {
+ await r.cleanup();
+ }
+});
+
+test("no transcript and no captions: the fallback runs as it always has, forced or not", async () => {
+ const noCaps = { ...META, automatic_captions: {} };
+ const r = await runWith({ meta: noCaps, managed: FORCED });
+ try {
+ assert.equal(r.argvs.length, 3);
+ assert.match(r.log, /No subs available for .*falling back to audio download \+ whisper \(keeping source video\)/);
+ assert.doesNotMatch(r.log, /forceMedia/);
+ assert.ok(r.files.includes("audio.mp3"));
+ } finally {
+ await r.cleanup();
+ }
+});
+
+test("forceMedia with a transcript but a plan that persists nothing fetches nothing", async () => {
+ const r = await runWith({
+ meta: META,
+ transcript: TRANSCRIPT,
+ managed: { forceMedia: true, keepSourceVideoOverride: false },
+ });
+ try {
+ assert.equal(r.argvs.length, 2);
+ assert.match(r.log, /forceMedia: a transcript is on disk and this download keeps no source video/);
+ assert.equal(r.record.status, "ok");
+ } finally {
+ await r.cleanup();
+ }
+});
+
+test("the prefetch's rewrite of metadata.info.json lands in metadata.history.json, with who asked", async () => {
+ const r = await runWith({
+ meta: { ...META, title: "Renamed upstream", view_count: 12 },
+ priorMeta: { ...META, view_count: 10 },
+ transcript: TRANSCRIPT,
+ managed: FORCED,
+ });
+ try {
+ const h = JSON.parse(
+ await readFile(path.join(r.videoDir, "metadata.history.json"), "utf8"),
+ ) as { entries: Array<Record<string, unknown>> };
+ assert.equal(h.entries.length, 1);
+ const e = h.entries[0];
+ assert.equal(e.by, "prefetch");
+ assert.equal(e.requestedBy, "mcp");
+ assert.deepEqual(e.changed, { title: { from: "A video", to: "Renamed upstream" } });
+ assert.deepEqual(e.counters, { view_count: [10, 12] });
+ } finally {
+ await r.cleanup();
+ }
+});
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -30,6 +30,7 @@ import {
type ResolvedCookiePolicy,
} from "../lib/cookiePolicy";
import { resolveEffectiveAvailability } from "../lib/availability-server";
+import { withMetadataHistory } from "../lib/metadataHistory-server";
import {
settledByTitleFilterIds,
upsertMetadataScan,
@@ -288,8 +289,8 @@ export function outputArgsForUrl(
opts: { mediaName?: string } = {},
): string[] {
const mediaName = opts.mediaName ?? "audio";
- const id = extractVideoId(url);
- if (id && /^[\w.-]+$/.test(id) && id !== "." && id !== "..") {
+ const id = dataDirIdForUrl(url);
+ if (id) {
return [
"-o",
`data/${id}/${mediaName}.%(ext)s`,
@@ -314,6 +315,14 @@ export function outputArgsForUrl(
return OUTPUT_ARGS;
}
+// The data/<id>/ directory outputArgsForUrl pins a URL's output to, or null
+// when it falls back to yt-dlp's %(id)s (no canonical id, or one that is not a
+// safe directory name) — i.e. the directory is only known afterwards.
+export function dataDirIdForUrl(url: string): string | null {
+ const id = extractVideoId(url);
+ return id && /^[\w.-]+$/.test(id) && id !== "." && id !== ".." ? id : null;
+}
+
// Thrown by enumeratePlaylistUrls when a listing is rate-limited part-way.
//
// A full enumeration that the platform rate-limited part-way through. What it
@@ -1377,7 +1386,20 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> {
"--",
opts.singleVideoUrl,
];
- await runChildAndStream(opts, root, args);
+ // --write-info-json rewrites the video's metadata.info.json; the history
+ // keeps what moved (lib/metadataHistory-server.ts). Only when the output dir
+ // is pinned — a %(id)s dir is known only after the run.
+ const dirId = dataDirIdForUrl(opts.singleVideoUrl);
+ const run = () => runChildAndStream(opts, root, args);
+ if (dirId) {
+ await withMetadataHistory(
+ path.join(root, "data", dirId),
+ { by: "download-one", onLog: opts.onLog },
+ run,
+ );
+ } else {
+ await run();
+ }
await safeBackfillAvailability(opts);
}
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -234,9 +234,12 @@ export async function downloadVideoPipelineAction(
// Re-fetch an already-downloaded video purely to archive its SOURCE container,
// without disturbing the existing transcript. Forces the persistence rule on
-// (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video,
-// app-extracts audio, and keeps the container (Phase 3 moves it to the saved
-// store). Works on a video already in the archive — downloadOneManaged has no
+// (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video
+// and keeps the container (Phase 3 moves it to the saved store), and forces the
+// media download itself (forceMedia) on a youtube-handling channel, whose only
+// media pass is otherwise skipped when a transcript or captions exist; there,
+// with a transcript on disk, no audio is extracted — the container is the
+// point. Works on a video already in the archive — downloadOneManaged has no
// archive prefilter, so it always re-downloads.
export async function redownloadToArchiveAction(
slug: string,
@@ -300,6 +303,11 @@ async function archiveSourceVideo(
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
keepSourceVideoOverride: true,
+ // The operator (or the tool behind a whole-recording fetch) asked for
+ // the SOURCE: a transcript or captions on disk are not a reason to
+ // skip it. Without this a youtube-handling video with a transcript
+ // got two subtitle passes and no file (release 10 slice N).
+ forceMedia: true,
persistOrigin,
});
safeRevalidate([