Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 19c214306feda8c4c433ca33bdebfc085393adfd
parent 6469058b22b8e3646f5954e8723f6892da7589aa
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 26 Sep 2026 16:42:53 -0400

common: forceMedia — "Persist source video", the whole-recording fetch and "Persist kept now" download the source on a youtube-handling video that already has a transcript or captions; with a transcript on disk the pass persists the container only (no subs, no audio, no whisper); the prefetch, the audio-check primary and download-one-audio run inside the metadata history wrap

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/persistKept.ts | 9+++++++--
Mcommon/ytdlp/downloadOneManaged.ts | 192+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----------------
Acommon/ytdlp/forceMedia.test.ts | 302++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 28+++++++++++++++++++++++++---
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 14+++++++++++---
5 files changed, 499 insertions(+), 46 deletions(-)

diff --git a/common/controller/persistKept.ts b/common/controller/persistKept.ts @@ -15,8 +15,9 @@ import { downloadOneManaged } from "../ytdlp/downloadOneManaged"; // after those videos had already been downloaded audio-only. // // For each kept video whose source isn't saved yet, it re-fetches the container -// via downloadOneManaged with keepSourceVideoOverride (which app-extracts audio -// and moves the container into the saved store), without disturbing the existing +// via downloadOneManaged with keepSourceVideoOverride + forceMedia (which moves +// the container into the saved store; a youtube-handling video that already has +// a transcript gets no audio extracted), without disturbing the existing // transcript. Mirrors redownloadToArchiveAction (videoActions.ts) but loops over // the whole window. Videos already saved, or whose URL can't be resolved, are // skipped rather than failing the pass. @@ -130,6 +131,10 @@ export async function persistKept({ globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, keepSourceVideoOverride: true, + // "Persist kept" means persist: on a youtube-handling channel the media + // pass is otherwise skipped for any video with a transcript or + // captions, which is every kept one (release 10 slice N). + forceMedia: true, }); result.persisted += 1; } catch (e) { diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -39,6 +39,7 @@ import { } from "../lib/downloadOutcome"; import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { recordAvailability } from "../lib/availability-server"; +import { withMetadataHistory } from "../lib/metadataHistory-server"; import { loadRawMetadata, loadRawMetadataFromDir, @@ -141,6 +142,16 @@ export type ManagedDownloadOpts = { // Per-run override: extract audio now and discard the container even for a // video the keep-latest rule would otherwise persist (the save-disk backfill). extractImmediately?: boolean; + // THE CALLER WANTS THE MEDIA (release 10 slice N): a transcript or captions + // on disk are not a reason to skip the download. Only youtube handling ever + // skipped it — its media pass is the no-subs fallback (attempt 3), gated on + // "no transcript and no captions" — so this opens that gate. With a + // transcript on disk the forced pass touches no transcript and extracts no + // audio: it persists the container and nothing else (see attempt 3). Set by + // "Persist source video", the whole-recording fetch and "Persist kept now"; + // never by the download lane, re-acquire or import. Transcribe handling + // already downloads, so it is unaffected. + forceMedia?: boolean; // Per-run override of channelConfig.audioFormat for the extracted audio. audioFormatOverride?: AudioFormat; // Resolved yt-dlp `-f` download-format preset (override > channel > global), @@ -235,6 +246,11 @@ async function finalizeAppExtraction(opts: { category: PersistenceDecision["category"]; // Threaded straight onto the pointer; see ManagedDownloadOpts.persistOrigin. origin?: SavedVideoOrigin; + // PERSIST ONLY when false: move the container, extract nothing. A forced + // media download of a video that already has a transcript (attempt 3, + // `keepTranscript`) wants the source kept, and an audio.<fmt> beside an + // existing transcript is bytes nothing will read. Default true. + extractAudio?: boolean; onLog: (s: string) => void; signal: AbortSignal; }): Promise<void> { @@ -246,25 +262,37 @@ async function finalizeAppExtraction(opts: { ); return; } - try { - await transcodeAudio({ - paths: opts.paths, - videoDir: opts.videoDir, - sourceFilename: source, - targetFormat: opts.fmt, - onLog: opts.onLog, - signal: opts.signal, - }); - } catch (err) { + if (opts.extractAudio === false) { opts.onLog( - `App extraction failed (${source} -> audio.${opts.fmt}): ${(err as Error).message}. Keeping the source container.\n`, + `Persist only: a transcript is on disk, so no audio.${opts.fmt} is extracted from ${source}.\n`, ); - return; - } - if (!opts.persist) { - await rm(path.join(opts.videoDir, source), { force: true }); - opts.onLog(`Discarded source container ${source} (audio-only).\n`); - return; + if (!opts.persist) { + // Unreachable from attempt 3 today (it does not download when the plan + // persists nothing), kept so the option means one thing on its own. + opts.onLog(`Not persisting ${source}; it stays in the data dir.\n`); + return; + } + } else { + try { + await transcodeAudio({ + paths: opts.paths, + videoDir: opts.videoDir, + sourceFilename: source, + targetFormat: opts.fmt, + onLog: opts.onLog, + signal: opts.signal, + }); + } catch (err) { + opts.onLog( + `App extraction failed (${source} -> audio.${opts.fmt}): ${(err as Error).message}. Keeping the source container.\n`, + ); + return; + } + if (!opts.persist) { + await rm(path.join(opts.videoDir, source), { force: true }); + opts.onLog(`Discarded source container ${source} (audio-only).\n`); + return; + } } // Persist: move the container into the saved-video store + write a pointer. const storeDir = savedVideoDir( @@ -655,10 +683,21 @@ async function runManagedDownload( "--", opts.videoUrl, ]; - const prefetchRes = await runOneYtdlp( - opts, - channelDir, - buildPrefetchArgs(prefetchCookies), + // EVERY PREFETCH REWRITES metadata.info.json (no existence check — the + // filters decide on today's metadata), so each spawn runs inside the + // history wrap: what moved since the last version is appended to + // metadata.history.json. The file yt-dlp writes is still the file. + const prefetchHistory = { + by: "prefetch" as const, + ...(opts.persistOrigin?.requestedBy + ? { requestedBy: opts.persistOrigin.requestedBy } + : {}), + onLog: opts.onLog, + }; + const prefetchRes = await withMetadataHistory( + videoDir, + prefetchHistory, + () => runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchCookies)), ); lastFullTail = prefetchRes.stderrTail; const prefetchAvail = attemptSucceeded(prefetchRes.exitCode) @@ -694,10 +733,11 @@ async function runManagedDownload( opts.onLog( `Metadata prefetch auth-required (${prefetchAvail}); retrying with --cookies-from-browser ${prefetchRetryCookies}\n`, ); - const retryRes = await runOneYtdlp( - opts, - channelDir, - buildPrefetchArgs(prefetchRetryCookies), + const retryRes = await withMetadataHistory( + videoDir, + prefetchHistory, + () => + runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchRetryCookies)), ); lastFullTail = retryRes.stderrTail; const retryAvail = attemptSucceeded(retryRes.exitCode) @@ -1027,16 +1067,34 @@ async function runManagedDownload( // output to data/<canonicalId>/ for every platform, so the canonical id is // the dir. const expectedVideoIdHint = canonicalId; - const audioOutcome = await runAudioCheckedYtdlp({ - paths: opts.paths, - channelDir, - channelConfig: opts.channelConfig, - ytdlpArgs: primaryArgs, - onLog: opts.onLog, - signal: opts.signal, - archiveMarker: ARCHIVE_MARKER, - expectedVideoIdHint, - }); + const runAudioCheck = () => + runAudioCheckedYtdlp({ + paths: opts.paths, + channelDir, + channelConfig: opts.channelConfig, + ytdlpArgs: primaryArgs, + onLog: opts.onLog, + signal: opts.signal, + archiveMarker: ARCHIVE_MARKER, + expectedVideoIdHint, + }); + // The audio-checked primary writes the info json itself (it re-extracts on + // purpose), so it is the second rewrite of the file in one download and + // gets its own history entry. Only when the dir is known up front — an + // unidentifiable URL lands in data/%(id)s/, found only afterwards. + const audioOutcome = canonicalId + ? await withMetadataHistory( + path.join(channelDir, "data", canonicalId), + { + by: "audio-check", + ...(opts.persistOrigin?.requestedBy + ? { requestedBy: opts.persistOrigin.requestedBy } + : {}), + onLog: opts.onLog, + }, + runAudioCheck, + ) + : await runAudioCheck(); primaryRes = { exitCode: audioOutcome.ytdlpExitCode, stderrTail: audioOutcome.stderrTail, @@ -1225,6 +1283,11 @@ async function runManagedDownload( } // ---------- Attempt 3: no-subs fallback (youtube handling only) ---------- + // Also the youtube-handling MEDIA download a `forceMedia` caller asked for: + // this is the only pass that fetches media for a youtube-handling channel, + // and its own gate ("no transcript and no captions") is exactly the reason + // "Persist source video" and the whole-recording fetch used to do nothing on + // a video that already had a transcript — two subtitle passes, no file. if ( lastSucceeded && opts.channelConfig.handling === "youtube" && @@ -1232,10 +1295,40 @@ async function runManagedDownload( ) { const hasTranscript = await hasAnyTranscriptOnDisk(videoDir); const noCaptions = await metadataReportsNoCaptions(videoDir); - if (!hasTranscript && noCaptions) { + const noSubsFallback = !hasTranscript && noCaptions; + // Forced only when the fallback would NOT have run on its own: a video with + // no transcript and no captions takes today's path whoever asked. + const forced = !noSubsFallback && opts.forceMedia === true; + // THE TRANSCRIPT ON DISK IS NEVER TOUCHED. A forced pass over a video that + // has one fetches the source and persists it, and does nothing else: no + // subtitles (refused on the command line, after the channel's own args, so + // a channel carrying --write-auto-subs cannot overwrite the transcript), no + // audio extraction, no transcription, no short-audio verdict on audio it + // never made. With NO transcript on disk (captions listed, none fetched — + // a language the channel's sub-langs does not match) the forced pass is + // today's fallback in full: download, extract, transcribe inline if + // configured, persist. + const keepTranscript = forced && hasTranscript; + if (keepTranscript && !plan.persist) { + // PERSIST-ONLY WITH NOTHING TO PERSIST. Every forceMedia caller sets the + // keep-source override, so the plan persists; a caller that did not would + // be asking for a download with no destination — the transcript rules + // out audio, and the plan rules out the container. opts.onLog( - `No subs available for ${videoId}; falling back to audio download + whisper${plan.persist ? " (keeping source video)" : ""}.\n`, + `forceMedia: a transcript is on disk and this download keeps no source video (${plan.reason}); nothing to fetch.\n`, ); + } else if (noSubsFallback || forced) { + if (forced) { + opts.onLog( + `forceMedia: downloading the source although ${ + hasTranscript ? "a transcript is on disk" : "captions exist" + }\n`, + ); + } else { + opts.onLog( + `No subs available for ${videoId}; falling back to audio download + whisper${plan.persist ? " (keeping source video)" : ""}.\n`, + ); + } const fallbackConfig: ChannelConfig = { ...opts.channelConfig, handling: "transcribe", @@ -1267,6 +1360,9 @@ async function runManagedDownload( "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(fallbackConfig, fallbackCookieOverride), + // yt-dlp takes the LAST occurrence of an option, so the refusals go + // after the channel's own args (the chat-only pass's rule). + ...(keepTranscript ? ["--no-write-subs", "--no-write-auto-subs"] : []), ...sourceArgs( opts.videoUrl, path.join(videoDir, "metadata.info.json"), @@ -1290,7 +1386,27 @@ async function runManagedDownload( }); if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine; - if (attemptSucceeded(fallbackRes.exitCode)) { + if (attemptSucceeded(fallbackRes.exitCode) && keepTranscript) { + // Persist only. Not a fallback to transcription — the transcript is + // the one already on disk — so `fellBackToTranscribe` stays unset and + // the status stays what the subtitle pass made it. + if (plan.extractionMode === "app") { + await finalizeAppExtraction({ + paths: opts.paths, + channelSlug: opts.channelSlug, + channelConfig: opts.channelConfig, + videoDir, + videoId, + fmt, + persist: plan.persist, + category: plan.category, + origin: opts.persistOrigin, + extractAudio: false, + onLog: opts.onLog, + signal: opts.signal, + }); + } + } else if (attemptSucceeded(fallbackRes.exitCode)) { fellBackToTranscribe = true; // In app mode, extract audio.<fmt> from the downloaded source container // (and keep/discard it) before any inline whisper can read the audio. diff --git a/common/ytdlp/forceMedia.test.ts b/common/ytdlp/forceMedia.test.ts @@ -0,0 +1,302 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + chmod, + mkdir, + mkdtemp, + readFile, + readdir, + rm, + writeFile, +} from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { downloadOneManaged, type ManagedDownloadOpts } from "./downloadOneManaged"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/forceMedia.test.ts +// +// THE WHOLE-RECORDING FETCH DOWNLOADS ANYWAY (release 10 slice N). On a +// youtube-handling channel the only pass that fetches media is attempt 3, the +// no-subs fallback, gated on "no transcript and no captions" — so "Persist +// source video" and `fetch_clip` with `full: true` were a silent no-op on any +// video that already had a transcript: two subtitle passes, no file. +// `forceMedia` opens the gate; with a transcript on disk the forced pass +// persists the container and does nothing else. The yt-dlp here is a temp +// node script that logs every argv and writes what the real one would; no +// network. + +const ID = "H64QQZuw-aA"; +const VIDEO = `https://www.youtube.com/watch?v=${ID}`; + +// The fake: the metadata prefetch writes meta.json's content as the info json; +// the subtitle pass writes nothing (the transcript is seeded); the media pass +// (the `source-media` output template) writes the container. Every spawn +// appends its argv, one JSON line, to `argv.log`. +const FAKE_YTDLP = `#!/usr/bin/env node +const fs = require("node:fs"); +const path = require("node:path"); +const argv = process.argv.slice(2); +const root = process.env.FAKE_ROOT; +fs.appendFileSync(path.join(root, "argv.log"), JSON.stringify(argv) + "\\n"); +const has = (f) => argv.includes(f); +const arg = (f) => { const i = argv.indexOf(f); return i < 0 ? undefined : argv[i + 1]; }; +const info = arg("--load-info-json"); +const id = info ? path.basename(path.dirname(info)) : "${ID}"; +const dir = path.join("data", id); +fs.mkdirSync(dir, { recursive: true }); +if (has("--skip-download") && has("--write-info-json") && has("--no-write-subs")) { + fs.copyFileSync(path.join(root, "meta.json"), path.join(dir, "metadata.info.json")); + process.exit(0); +} +if (has("--skip-download") && has("--write-auto-subs")) { + console.log("DLOM_ARCHIVE youtube " + id); + process.exit(0); +} +if ((arg("-o") || "").includes("source-media")) { + fs.writeFileSync(path.join(dir, "source-media.mp4"), "container bytes for " + id); + console.log("DLOM_ARCHIVE youtube " + id); + process.exit(0); +} +console.error("fake: unknown invocation " + argv.join(" ")); +process.exit(2); +`; + +// ffmpeg for the app extraction: writes its last argument (the temp output). +const FAKE_FFMPEG = `#!/bin/sh +for last; do :; done +echo "extracted audio" > "$last" +`; + +type Run = { + argvs: string[][]; + record: Awaited<ReturnType<typeof downloadOneManaged>>; + log: string; + videoDir: string; + storeDir: string; + files: string[]; +}; + +async function runWith(opts: { + // The info json the prefetch writes. + meta: Record<string, unknown>; + // Seed a transcript before the run. + transcript?: { name: string; body: string }; + // Seed an info json before the run (so the prefetch is a REWRITE). + priorMeta?: Record<string, unknown>; + config?: Partial<ChannelConfig>; + managed?: Partial<ManagedDownloadOpts>; +}): Promise<Run & { cleanup: () => Promise<void> }> { + const root = await mkdtemp(path.join(tmpdir(), "force-media-")); + const ytdlp = path.join(root, "fake-ytdlp.cjs"); + const ffmpeg = path.join(root, "fake-ffmpeg.sh"); + await writeFile(ytdlp, FAKE_YTDLP); + await writeFile(ffmpeg, FAKE_FFMPEG); + await chmod(ytdlp, 0o755); + await chmod(ffmpeg, 0o755); + await writeFile(path.join(root, "meta.json"), JSON.stringify(opts.meta)); + process.env.FAKE_ROOT = root; + const paths = { + transcriptsDir: root, + channelsDir: path.join(root, "channels"), + savedVideosDir: path.join(root, "saved-videos"), + ytdlpBin: ytdlp, + ffmpegBin: ffmpeg, + ffprobeBin: path.join(root, "no-ffprobe"), + } as Paths; + const videoDir = path.join(paths.channelsDir, "c", "data", ID); + await mkdir(videoDir, { recursive: true }); + if (opts.transcript) { + await writeFile(path.join(videoDir, opts.transcript.name), opts.transcript.body); + } + if (opts.priorMeta) { + await writeFile( + path.join(videoDir, "metadata.info.json"), + JSON.stringify(opts.priorMeta), + ); + } + let log = ""; + const record = await downloadOneManaged({ + channelSlug: "c", + channelConfig: { + handling: "youtube", + url: "https://www.youtube.com/@c/videos", + ...opts.config, + } as ChannelConfig, + paths, + videoUrl: VIDEO, + onLog: (s) => { + log += s; + }, + signal: new AbortController().signal, + ...opts.managed, + }); + const argvs = (await readFile(path.join(root, "argv.log"), "utf8")) + .split("\n") + .filter(Boolean) + .map((l) => JSON.parse(l) as string[]); + return { + argvs, + record, + log, + videoDir, + storeDir: path.join(paths.savedVideosDir, "c", ID), + files: (await readdir(videoDir)).sort(), + cleanup: () => rm(root, { recursive: true, force: true }), + }; +} + +const META = { + id: ID, + title: "A video", + duration: 60, + webpage_url: VIDEO, + subtitles: {}, + automatic_captions: { en: [{ url: "https://example/cap" }] }, +}; +const TRANSCRIPT = { name: "transcript.json", body: '{"segments":[{"text":"kept"}]}' }; +const FORCED: Partial<ManagedDownloadOpts> = { + keepSourceVideoOverride: true, + forceMedia: true, + persistOrigin: { requestedBy: "mcp", requestedAt: "2026-09-26T00:00:00.000Z" }, +}; + +test("forceMedia with a transcript on disk: the source is fetched and persisted, nothing else", async () => { + const r = await runWith({ meta: META, transcript: TRANSCRIPT, managed: FORCED }); + try { + // Prefetch, the subtitle pass as every re-download runs it, then the media. + assert.equal(r.argvs.length, 3); + assert.ok(r.argvs[1].includes("--skip-download")); + const media = r.argvs[2]; + assert.ok(!media.includes("--skip-download"), media.join(" ")); + assert.ok(media.includes("--load-info-json")); + assert.deepEqual( + media.slice(media.indexOf("-f"), media.indexOf("-f") + 2), + ["-f", "bestvideo*+bestaudio/best"], + ); + // The transcript cannot be rewritten, even by a channel's own args: the + // refusals come after everything the channel passes. + assert.ok(media.indexOf("--no-write-auto-subs") > media.indexOf("--print")); + assert.ok(media.includes("--no-write-subs")); + assert.match(r.log, /forceMedia: downloading the source although a transcript is on disk\n/); + + // The pointer landed and the container is in the store. + const pointer = JSON.parse( + await readFile(path.join(r.videoDir, "saved-video.json"), "utf8"), + ) as Record<string, unknown>; + assert.equal(pointer.keepReason, "override"); + assert.equal((pointer.origin as { requestedBy: string }).requestedBy, "mcp"); + assert.equal( + await readFile(path.join(r.storeDir, "source-media.mp4"), "utf8"), + `container bytes for ${ID}`, + ); + // The transcript is byte-for-byte what it was, and no audio was made. + assert.equal( + await readFile(path.join(r.videoDir, TRANSCRIPT.name), "utf8"), + TRANSCRIPT.body, + ); + assert.deepEqual( + r.files.filter((f) => f.startsWith("audio.") || f.startsWith("source-media")), + [], + ); + assert.match(r.log, /Persist only: a transcript is on disk, so no audio\.mp3 is extracted/); + // Not a fallback to transcription, and a clean status. + assert.equal(r.record.status, "ok"); + assert.equal(r.record.fellBackToTranscribe, undefined); + assert.deepEqual( + r.record.attempts.map((a) => [a.kind, a.ytdlpExitCode]), + [ + ["metadata-prefetch", 0], + ["primary", 0], + ["no-subs-fallback", 0], + ], + ); + } finally { + await r.cleanup(); + } +}); + +test("without forceMedia the same video is untouched: two subtitle-shaped passes, no file (today's behaviour)", async () => { + const r = await runWith({ + meta: META, + transcript: TRANSCRIPT, + managed: { keepSourceVideoOverride: true }, + }); + try { + assert.equal(r.argvs.length, 2); + assert.ok(r.argvs.every((a) => a.includes("--skip-download"))); + assert.ok(!r.files.includes("saved-video.json")); + assert.doesNotMatch(r.log, /forceMedia/); + } finally { + await r.cleanup(); + } +}); + +test("forceMedia with captions listed but no transcript: today's fallback in full (download, extract, persist)", async () => { + const r = await runWith({ meta: META, managed: FORCED }); + try { + assert.equal(r.argvs.length, 3); + const media = r.argvs[2]; + assert.ok(!media.includes("--skip-download")); + // No transcript to protect, so no refusals — the args are the fallback's. + assert.ok(!media.includes("--no-write-auto-subs")); + assert.match(r.log, /forceMedia: downloading the source although captions exist\n/); + assert.ok(r.files.includes("saved-video.json")); + assert.ok(r.files.includes("audio.mp3"), r.files.join(",")); + assert.equal(r.record.fellBackToTranscribe, true); + } finally { + await r.cleanup(); + } +}); + +test("no transcript and no captions: the fallback runs as it always has, forced or not", async () => { + const noCaps = { ...META, automatic_captions: {} }; + const r = await runWith({ meta: noCaps, managed: FORCED }); + try { + assert.equal(r.argvs.length, 3); + assert.match(r.log, /No subs available for .*falling back to audio download \+ whisper \(keeping source video\)/); + assert.doesNotMatch(r.log, /forceMedia/); + assert.ok(r.files.includes("audio.mp3")); + } finally { + await r.cleanup(); + } +}); + +test("forceMedia with a transcript but a plan that persists nothing fetches nothing", async () => { + const r = await runWith({ + meta: META, + transcript: TRANSCRIPT, + managed: { forceMedia: true, keepSourceVideoOverride: false }, + }); + try { + assert.equal(r.argvs.length, 2); + assert.match(r.log, /forceMedia: a transcript is on disk and this download keeps no source video/); + assert.equal(r.record.status, "ok"); + } finally { + await r.cleanup(); + } +}); + +test("the prefetch's rewrite of metadata.info.json lands in metadata.history.json, with who asked", async () => { + const r = await runWith({ + meta: { ...META, title: "Renamed upstream", view_count: 12 }, + priorMeta: { ...META, view_count: 10 }, + transcript: TRANSCRIPT, + managed: FORCED, + }); + try { + const h = JSON.parse( + await readFile(path.join(r.videoDir, "metadata.history.json"), "utf8"), + ) as { entries: Array<Record<string, unknown>> }; + assert.equal(h.entries.length, 1); + const e = h.entries[0]; + assert.equal(e.by, "prefetch"); + assert.equal(e.requestedBy, "mcp"); + assert.deepEqual(e.changed, { title: { from: "A video", to: "Renamed upstream" } }); + assert.deepEqual(e.counters, { view_count: [10, 12] }); + } finally { + await r.cleanup(); + } +}); diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -30,6 +30,7 @@ import { type ResolvedCookiePolicy, } from "../lib/cookiePolicy"; import { resolveEffectiveAvailability } from "../lib/availability-server"; +import { withMetadataHistory } from "../lib/metadataHistory-server"; import { settledByTitleFilterIds, upsertMetadataScan, @@ -288,8 +289,8 @@ export function outputArgsForUrl( opts: { mediaName?: string } = {}, ): string[] { const mediaName = opts.mediaName ?? "audio"; - const id = extractVideoId(url); - if (id && /^[\w.-]+$/.test(id) && id !== "." && id !== "..") { + const id = dataDirIdForUrl(url); + if (id) { return [ "-o", `data/${id}/${mediaName}.%(ext)s`, @@ -314,6 +315,14 @@ export function outputArgsForUrl( return OUTPUT_ARGS; } +// The data/<id>/ directory outputArgsForUrl pins a URL's output to, or null +// when it falls back to yt-dlp's %(id)s (no canonical id, or one that is not a +// safe directory name) — i.e. the directory is only known afterwards. +export function dataDirIdForUrl(url: string): string | null { + const id = extractVideoId(url); + return id && /^[\w.-]+$/.test(id) && id !== "." && id !== ".." ? id : null; +} + // Thrown by enumeratePlaylistUrls when a listing is rate-limited part-way. // // A full enumeration that the platform rate-limited part-way through. What it @@ -1377,7 +1386,20 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { "--", opts.singleVideoUrl, ]; - await runChildAndStream(opts, root, args); + // --write-info-json rewrites the video's metadata.info.json; the history + // keeps what moved (lib/metadataHistory-server.ts). Only when the output dir + // is pinned — a %(id)s dir is known only after the run. + const dirId = dataDirIdForUrl(opts.singleVideoUrl); + const run = () => runChildAndStream(opts, root, args); + if (dirId) { + await withMetadataHistory( + path.join(root, "data", dirId), + { by: "download-one", onLog: opts.onLog }, + run, + ); + } else { + await run(); + } await safeBackfillAvailability(opts); } diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -234,9 +234,12 @@ export async function downloadVideoPipelineAction( // Re-fetch an already-downloaded video purely to archive its SOURCE container, // without disturbing the existing transcript. Forces the persistence rule on -// (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video, -// app-extracts audio, and keeps the container (Phase 3 moves it to the saved -// store). Works on a video already in the archive — downloadOneManaged has no +// (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video +// and keeps the container (Phase 3 moves it to the saved store), and forces the +// media download itself (forceMedia) on a youtube-handling channel, whose only +// media pass is otherwise skipped when a transcript or captions exist; there, +// with a transcript on disk, no audio is extracted — the container is the +// point. Works on a video already in the archive — downloadOneManaged has no // archive prefilter, so it always re-downloads. export async function redownloadToArchiveAction( slug: string, @@ -300,6 +303,11 @@ async function archiveSourceVideo( globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, keepSourceVideoOverride: true, + // The operator (or the tool behind a whole-recording fetch) asked for + // the SOURCE: a transcript or captions on disk are not a reason to + // skip it. Without this a youtube-handling video with a transcript + // got two subtitle passes and no file (release 10 slice N). + forceMedia: true, persistOrigin, }); safeRevalidate([