import { removeMediaFile } from "../lib/mediaTier-server"; import { tierVideoDir } from "../lib/mediaTier-server"; import path from "node:path"; import { access, appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises"; import { createWriteStream, type Dirent, type WriteStream } from "node:fs"; import { execa } from "execa"; import { AUTH_RETRY_CLASSES, classifyDownloadFailure, hasNonRateLimitSubtitleFailure, hasSubtitleRateLimit, parseUnavailableFromStderr, } from "../lib/availability"; import { DEFAULT_COOKIE_MODE, alwaysCookies, authRetryCookies, type ResolvedCookiePolicy, } from "../lib/cookiePolicy"; import { type AudioFormat, type ChannelConfig, } from "../lib/channelConfig"; import { resolvePersistenceDecision, type PersistenceDecision, } from "./persistencePlan"; import { isInKeepWindow, uploadKeyFor, type KeepWindow, } from "../controller/keptVideos"; import { isDoNotClean } from "../lib/doNotClean-server"; import { transcodeAudio } from "../controller/transcode"; import { findSourceMedia } from "../lib/videoStatus"; import { savedVideoDir, savedVideoFormatFromLine, type SavedVideoFormat, type SavedVideoOrigin, } from "../lib/savedVideo"; import { persistSourceVideo } from "../lib/savedVideo-server"; import { type AudioCheckAttemptStats, type DownloadAttempt, type DownloadOutcomeRecord, type DownloadOutcomeStatus, } from "../lib/downloadOutcome"; import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { formatBytes } from "../lib/format"; import { recordAvailability } from "../lib/availability-server"; import { withMetadataHistory } from "../lib/metadataHistory-server"; import { ensureWaybackProvenance } from "../lib/wayback-server"; import { parseWaybackUrl } from "../lib/wayback"; import { loadRawMetadata, loadRawMetadataFromDir, platformFromMetadata, isLivestreamMetadata, } from "../lib/transcripts-server"; import { evaluateDownloadFilters } from "../lib/downloadFilters"; import { normalizeLiveChat } from "../controller/normalizeLiveChat"; import { LIVE_CHAT_FILENAME } from "../lib/videoStatus"; import { upsertMetadataScan, type MetadataScanEntry, } from "../controller/metadataScanStore"; import { detectPlatform, type Platform } from "../lib/platform"; import { downloadArchiveOrgManaged, type ArchiveOrgDownloadDeps, } from "../controller/archiveOrgDownload"; import { finalizeAppExtraction } from "./finalizeAppExtraction"; import { probeMediaDurationSec } from "./ffprobeDuration"; import { isShortAudio, INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC, } from "../lib/transcriptCoverage"; import type { Paths } from "../lib/paths"; import { transcribeWithWorker } from "../controller/transcribeOne"; import { extractVideoId, outputArgsForUrl } from "./runYtdlp"; import { channelExtraArgs, channelPaceSeconds } from "./channelArgs"; import { ARCHIVE_MARKER, FORMAT_MARKER, runOneYtdlp as runOneYtdlpRaw, type AttemptOutcome, } from "./runOneYtdlp"; import { runAudioCheckedYtdlp } from "./audioCheckedDownload"; import { DEFAULT_SOURCE_VIDEO_QUALITY, LAST_RESORT_FORMAT_SELECTOR, VIDEO_720_MAX_HEIGHT, VIDEO_720_PRESET, resolveDownloadFormatSelector, sourceVideoFormatSelector, type DownloadFormatPreset, type SourceVideoQuality, } from "./downloadFormat"; import { DOWNLOAD_PROGRESS_TEMPLATE } from "../jobs/progressParsers"; // The `--print` archive marker below implies `--quiet`, which otherwise // suppresses every extraction log and the [download] progress lines. Re-enable // full output and emit a structured, throttled progress line we can parse // directly into the per-video progress bars (human log readability is secondary // to reliable progress parsing). --progress-delta keeps it to ~1 line/sec, // matching the /jobs poll. // EXPORTED because the clip-window fetch (ytdlp/fetchWindowManaged.ts) reuses // it verbatim: the progress lines the /jobs bars are parsed from come from this // template, and a second fetch path with its own idea of it would show no // progress at all. export const FULL_LOG_PROGRESS_ARGS = [ "--no-quiet", "--progress", "--newline", "--progress-delta", "1", "--progress-template", DOWNLOAD_PROGRESS_TEMPLATE, ]; export type ManagedDownloadOpts = { channelSlug: string; channelConfig: ChannelConfig; paths: Paths; videoUrl: string; onLog: (s: string) => void; signal: AbortSignal; // Resolved cookie policy (value + mode), already collapsed channel-over- // global by the caller (resolveCookiePolicy). Governs which attempts pass // --cookies-from-browser: "always" on every invocation, "when-required" // (the default when omitted) only on auth-retry passes, "defer" never — // failures are recorded for the Needs-cookies bucket instead. cookiePolicy?: ResolvedCookiePolicy; // When false, suppress appending to the channel archive file. Mirrors // `ignoreArchive` from the playlist-level callers. appendArchive?: boolean; // When true, run whisper inline immediately after the no-subs fallback's // audio download succeeds. When false (default), leave the audio for the // next "Transcribe missing" pass. Resolved from the site setting by the // playlist-level caller. inlineTranscribeOnFallback?: boolean; // Resolved global SiteSettings.skipLiveDownloads (default true). Combined // with the per-channel override (channelConfig.skipLiveDownloads) inside the // download filters. Plumbed from the playlist-level caller alongside the // other resolved settings. globalSkipLiveDownloads?: boolean; // Batch collector for a title-filter rejection's metadata. Supplied by // runManagedDownloads so a batch writes the channel-level metadata-scan store // ONCE rather than once per non-matching video. Absent (a one-off download) // means "write it yourself" — see the call site. onFilterRejected?: (id: string, entry: MetadataScanEntry) => void; // ---- Per-download persistence rule (Phase 2) ---- // The channel's keep-latest window as a cutoff, computed once per run by the // caller (computeKeepWindow). This video's membership is decided against its // own upload date — newer-than-the-cutoff videos persist their source. keepWindow?: KeepWindow; // Per-run override: force-keep (true) / force-discard (false) the source video. keepSourceVideoOverride?: boolean; // Who asked for this container, recorded on the saved-video pointer. Set by // the full-source variant of the clip-window fetch, so "why is this 4 GB // file here" has an answer months later. Never changes `keepReason` — that // stays override/pin so the retention prune leaves it alone. persistOrigin?: SavedVideoOrigin; // Per-run override: extract audio now and discard the container even for a // video the keep-latest rule would otherwise persist (the save-disk backfill). extractImmediately?: boolean; // THE CALLER WANTS THE MEDIA (release 10 slice N): a transcript or captions // on disk are not a reason to skip the download. Only youtube handling ever // skipped it — its media pass is the no-subs fallback (attempt 3), gated on // "no transcript and no captions" — so this opens that gate. With a // transcript on disk the forced pass touches no transcript and extracts no // audio: it persists the container and nothing else (see attempt 3). Set by // "Persist source video", the whole-recording fetch and "Persist kept now"; // never by the download lane, re-acquire or import. Transcribe handling // already downloads, so it is unaffected. forceMedia?: boolean; // Per-run override of channelConfig.audioFormat for the extracted audio. audioFormatOverride?: AudioFormat; // Resolved yt-dlp `-f` download-format preset (override > channel > global), // already collapsed by the playlist-level caller. Expanded into a concrete // selector here, where the per-video source platform is known (so "auto" // picks `original` for Odysee). Defaults to "auto" when omitted. downloadFormatPreset?: DownloadFormatPreset; // The quality of the source container a PERSIST keeps (a pass whose plan // persists in app mode: "Persist source video", the whole-recording fetch, // "Persist kept now", a keep-latest download). Resolved by the caller through // resolveSourceVideoQuality (override > channel > global). Unset or // "original" = `bestvideo*+bestaudio/best`, as every persist has always been; // "video_720" = the ≤720p H.264 selector (downloadFormat.ts). Never touches // the audio-only selector above. persistFormatPreset?: SourceVideoQuality; // Test seam: what the archive.org download (controller/archiveOrgDownload.ts) // talks to. Every production caller passes none. archiveOrgDeps?: ArchiveOrgDownloadDeps; }; // When `reuseInfoJson` is true, the real download reuses the metadata the // prefetch pass already wrote (fed in via sourceArgs as --load-info-json), so // we drop --write-info-json (the file is already on disk). When false, the // legacy single-call behavior: write the info json during this download. // // A SUBTITLE 429 IS A WARNING HERE, NOT THE END (release 17, slice RL). Under // yt-dlp's default, a subtitle file it cannot fetch raises: the video fails, // and — reading from --load-info-json — yt-dlp re-extracts from the URL once and // asks the throttled endpoint again. `--ignore-errors` makes it report the // failure as `WARNING: Unable to download video subtitles for …` and carry on // (exit 0, the line still in the log for the classifier); the caller then sees // "no transcript" and fetches the media (attempt 3). Any other error is still // an ERROR and a non-zero exit. `--sleep-subtitles` is the pace before each // subtitle request — the request YouTube throttles — never below `-t sleep`'s 5. function youtubeHandlingArgs( config: ChannelConfig, reuseInfoJson = false, ): string[] { const args = [ "--write-auto-subs", "--write-subs", "--sub-langs", config.subLangs ?? "en.*,live_chat", ]; if (!reuseInfoJson) args.push("--write-info-json"); args.push( "--skip-download", "-t", "sleep", "--sleep-subtitles", String(Math.max(YOUTUBE_SLEEP_SUBTITLES_FLOOR, channelPaceSeconds(config))), "--ignore-errors", ); return args; } // `-t sleep`'s own `--sleep-subtitles`: the adaptive pace only ever raises it. const YOUTUBE_SLEEP_SUBTITLES_FLOOR = 5; // Format/extract args for a transcribe-handling download under a resolved // persistence plan. "ytdlp" mode is the legacy path: yt-dlp extracts the audio // itself (-x) and optionally keeps its bestaudio source via -k. "app" mode omits // -x so yt-dlp leaves the source container for the app to extract from // (finalizeAppExtraction) — pulling a full video when persisting, bestaudio // otherwise. function audioFormatSelectionArgs( plan: PersistenceDecision, fmt: AudioFormat, config: ChannelConfig, selector: string, persistSelector: string, ): string[] { if (plan.extractionMode === "ytdlp") { const args = ["-f", selector, "-x", "--audio-format", fmt]; if (config.keepSourceVideo) args.push("-k"); return args; } // Persisting in app mode keeps the full source container (always a complete // video), so the truncation-prone audio-only selector doesn't apply: the // persist quality's selector does (sourceVideoFormatSelector). The format // yt-dlp actually took is printed for the saved-video pointer. return plan.persist ? ["-f", persistSelector, "--print", PERSIST_FORMAT_PRINT] : ["-f", selector]; } // After the video, the downloaded format: `requested_downloads[0]` is the // merged (or single) format. A MERGED download (YouTube's split video+audio) // carries no height of its own there — yt-dlp prints NA — so each field falls // back to the selected video stream (`requested_formats[0]`), then the top // level. runOneYtdlp scrapes it as `formatLine`; savedVideoFormatFromLine // parses it. const PERSIST_FORMAT_PRINT = `after_video:${FORMAT_MARKER} ` + `%(requested_downloads.0.height,requested_formats.0.height,height)s ` + `%(requested_downloads.0.vcodec,requested_formats.0.vcodec,vcodec)s ` + `%(requested_downloads.0.format_id)s`; // The log line for a persisted container's format, and the LOUD one for a // "video_720" persist that the last-resort rung answered: a file above 720p // (or one whose height yt-dlp did not report) is never silent. export function persistedFormatLog( format: SavedVideoFormat, formatLine: string | null | undefined, ): string { if (!formatLine) { return `Source format: not reported by yt-dlp (quality ${format.preset}).\n`; } const formatId = formatLine?.trim().split(/\s+/)[2]; const what = `${format.height ? `${format.height}p` : "unknown height"}` + `${format.vcodec ? ` ${format.vcodec}` : ""}` + `${formatId && formatId !== "NA" ? ` (format ${formatId})` : ""}`; if ( format.preset === VIDEO_720_PRESET && (format.height === undefined || format.height > VIDEO_720_MAX_HEIGHT) ) { return ( `Source quality ${VIDEO_720_PRESET}: no format at or under ${VIDEO_720_MAX_HEIGHT}p matched ` + `any rung; the last resort (${LAST_RESORT_FORMAT_SELECTOR}) took ${what}.\n` ); } return `Source format: ${what} (quality ${format.preset}).\n`; } // The media output + info-json + format args for a transcribe-handling download. // In app mode the main output is source-media. (distinct from audio. // so it's never treated as cleanable audio); otherwise the historical audio.. // reuseInfoJson drops --write-info-json (the prefetch already wrote it). function transcribeMediaArgs( url: string, config: ChannelConfig, plan: PersistenceDecision, fmt: AudioFormat, reuseInfoJson: boolean, selector: string, persistSelector: string, ): string[] { const mediaName = plan.extractionMode === "app" ? "source-media" : "audio"; return [ ...outputArgsForUrl(url, { mediaName }), ...(reuseInfoJson ? [] : ["--write-info-json"]), ...audioFormatSelectionArgs(plan, fmt, config, selector, persistSelector), ]; } // The source a download attempt reads from: either the prefetched info json // (no positional URL needed) or the URL itself. Mirrors how the no-subs // fallback has always fed yt-dlp a --load-info-json instead of a URL. function sourceArgs(url: string, loadInfoJson: string | null): string[] { return loadInfoJson ? ["--load-info-json", loadInfoJson] : ["--", url]; } // Audio-checked mode: we own the audio extraction, so yt-dlp must NOT run // the ExtractAudio postprocessor. We always pass `-c` so resume picks up // where the last validated snapshot left off. function transcribeHandlingArgsForAudioCheck( _config: ChannelConfig, selector: string, ): string[] { return [ "--write-info-json", "-f", selector, "-c", ]; } // Kept as a local name because every call site below reads `channelConfigArgs`; // the body lives in ytdlp/channelArgs.ts so the clip-window fetch shares it. const channelConfigArgs = channelExtraArgs; // A TITLE-FILTER REJECTION MUST NOT LEAVE A VIDEO DIRECTORY BEHIND. // // The prefetch writes `data//metadata.info.json` BEFORE the filters get to // look at it — that is the whole point of the split — so by the time the // operator's own "not this one" is known, the directory exists. And a directory // holding a metadata.info.json is not a neutral leftover: `buildIndex.ts` // admits ANY such dir to the LMDB index and to the published site (transcript // or not), and `deriveChannelSets` reads the dir NAME as "ever fetched", which // is what takes a vanished video out of `missingNeverFetched`. The metadata // scan creates none of them for exactly these two reasons // (controller/metadataScanStore.ts); a rejection that ran through the // downloader must not create them either, or the same channel gets both // answers depending on which path reached the video first. // // THE BUCKET DOES NOT COME FROM THIS DIRECTORY. `skippedByTitleFilter` is // derived in the snapshot from the metadata-scan store against the channel's // CURRENT filter (channelSnapshot.ts, settledIdsFrom) — the entry recorded // beside this call — so removing the dir costs the count nothing. The retryable // `skippedByFilter` bucket IS read off `download-outcome.json`, which is why // every OTHER filter (skip-live) still writes one: that skip says "we will try // again", and a video with no directory and no outcome would silently leave it. // // ── IT IS AN ALLOW-LIST, AND IT HAS TO BE ──────────────────────────────────── // // This function deletes a directory recursively, so the question it must answer // is "is EVERYTHING in here mine?", not "is anything in here one of the few // things I thought to check for". The deny-list it replaced named // isVideoDownloaded, `clips/` and `saved-video.json` — and would therefore have // deleted a chat-only corpus member (`transcript.live_chat.json` + // `live_chat.cues.json`), a foreign-language `transcript.es.vtt`, a resumable // `audio.mp3.part`, a `diarization.json` or a `digest.json`. That is not a // hypothetical: the chat-only branch below calls this whenever its chat pass // did NOT come back with a file, so a chat re-fetch that failed, was aborted // mid-write or hit a cooldown would have deleted the chat that was already // there, and an `importVideoAction` on an existing chat-only id would have done // the same. // // So: the dir goes only when every entry is something THIS pass wrote or could // have found from a previous run of itself. Anything else — any file, any // subdirectory, anything a future feature adds — keeps it, and the caller falls // back to writing the outcome sidecar exactly as it always did. Being wrong in // this direction costs one metadata stub; being wrong in the other costs bytes // nobody can enumerate. const PREFETCH_OWN_FILES: ReadonlySet = new Set([ // What the prefetch pass itself writes: --write-info-json under the // `infojson:` output template (outputArgsForUrl). "metadata.info.json", // The managed download's own per-video log, opened before the prefetch runs. "download.log", // A sidecar from an EARLIER managed attempt on this same id — including the // one a previous rejection wrote before this rule existed. It records what // happened, never what is on disk, so it is ours to drop with the rest. "download-outcome.json", // NOT metadata.history.json, deliberately: a history means an earlier // metadata.info.json was here before this pass, and it is the one record of // what the source used to say (lib/metadataHistory-server.ts). ]); async function discardPrefetchDir( videoDir: string, onLog: (line: string) => void, ): Promise { let entries: Dirent[]; try { entries = await readdir(videoDir, { withFileTypes: true }); } catch { // No directory at all — the legacy single-call path never made one — or it // is unreadable. Either way there is nothing of ours to remove, and a // `rm -rf` on a path we could not list is the last thing to do about it. return false; } for (const entry of entries) { // A DIRECTORY IS NEVER OURS. `clips/` is the one that exists today (media // another tool asked this editor for, invisible to every video-dir // enumerator); the rule is shaped so the next one needs no edit here. if (!entry.isFile() || !PREFETCH_OWN_FILES.has(entry.name)) return false; } try { await rm(videoDir, { recursive: true, force: true }); return true; } catch (err) { onLog( `Could not remove the prefetch directory ${videoDir}: ${(err as Error).message}\n`, ); return false; } } // Exported for the unit test ONLY. The rule above is a delete, and the // difference between the allow-list and the deny-list it replaced is invisible // to every end-to-end path that does not happen to have the right file on disk // — which is exactly the shape of bug that reaches production. See // downloadOneManaged.test.ts. export const __discardPrefetchDirForTest = discardPrefetchDir; // The one-invocation runner moved to ytdlp/runOneYtdlp.ts (the clip-window // fetch needs the same log tee, stderr tail and archive scrape). This wrapper // keeps the ManagedDownloadOpts-shaped call sites below unchanged. function runOneYtdlp( opts: ManagedDownloadOpts, cwd: string, args: string[], ): Promise { return runOneYtdlpRaw( { ytdlpBin: opts.paths.ytdlpBin, onLog: opts.onLog, signal: opts.signal }, cwd, args, ); } // ONE EXTRA yt-dlp PASS THAT FETCHES A LIVESTREAM'S CHAT AND NOTHING ELSE. // // Reached only from the filter branch below, for a channel whose // `rejectedLivestreams` is "chat-only". The media is not wanted; the chat is. // // WHAT THE ARGUMENT ORDER IS FOR. yt-dlp takes the LAST occurrence of an // option, so the refusals go AFTER the channel's own extra args — a channel // that configured `--write-thumbnail` or `--write-auto-subs` must not be able // to turn this pass back into a partial download of things nobody asked for. // Same rule fetchWindowManaged relies on, and for the same reason. // // --no-write-info-json, NOT the absence of --write-info-json: the metadata is // ALREADY on disk from the prefetch, and it has to stay there — a video dir // with no metadata.info.json is invisible to buildIndex, so the chat would be // fetched and then never published. This pass must neither rewrite it nor // delete it. // // No archive line is appended. An archive id means "downloaded" to // verifyTranscripts and to the sync walk, and this video is not. async function fetchLiveChatOnly( opts: ManagedDownloadOpts, channelDir: string, cookies: string | undefined, ): Promise { return runOneYtdlp(opts, channelDir, [ "--ignore-config", "--restrict-filenames", ...channelConfigArgs(opts.channelConfig, cookies), "--skip-download", "--write-subs", "--no-write-auto-subs", "--sub-langs", "live_chat", "--no-write-info-json", "--no-write-description", "--no-write-thumbnail", "--no-download-archive", ...outputArgsForUrl(opts.videoUrl), "--", opts.videoUrl, ]); } // Did the chat pass actually leave a chat behind? A stream with chat replay // disabled makes yt-dlp exit 0 and write nothing, and a directory holding only // a metadata.info.json is the leftover this slice exists to stop creating. async function hasLiveChatOnDisk(videoDir: string): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); return entries.includes(LIVE_CHAT_FILENAME); } function attemptSucceeded(exitCode: number | null): boolean { // yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop). return exitCode === 0 || exitCode === 101; } // DID A DOWNLOAD THAT WAS ASKED FOR THE SOURCE GET IT? (release 11 slice O3.) // For "Persist source video" and the whole-recording fetch: null when nothing // failed, else yt-dlp's reason. Two ways to fail: the download failed outright // (the status says so), or the forced media pass over a transcript failed — // which leaves the status as the subtitle pass made it, so only the last // attempt (the n: 3 media pass) says so. Reading the status alone ended those // jobs `done` with no file. // // WHEN THE LAST ATTEMPT IS THE MEDIA PASS, IT ALONE ANSWERS (review low 1). That // pass IS the source fetch; a later step's status is not about the source. A // fallback whose container came down and was persisted, and whose inline // whisper then failed, ends the download `failed` — and reading the status // first called that source "not downloaded". export function sourceFetchFailure(record: DownloadOutcomeRecord): string | null { const last = record.attempts.at(-1); const reason = (fallback: string) => last?.error?.trim() || fallback; if (last?.kind === "no-subs-fallback") { return attemptSucceeded(last.ytdlpExitCode) ? null : reason(`yt-dlp exited ${last.ytdlpExitCode ?? "without a code"}`); } if (record.status === "failed" || record.status === "failed-corrupt-source") { return reason(`the download ended ${record.status}`); } return null; } // yt-dlp's partial source container(s) in a video dir — `source-media..part` // and its fragments — with their sizes, for the failed-media log line. async function sourceMediaPartials(videoDir: string): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); const out: string[] = []; for (const name of entries.sort()) { if (!name.startsWith("source-media.") || !/\.part(?:-Frag\d+)?$/.test(name)) continue; const st = await stat(path.join(videoDir, name)).catch(() => null); out.push(st ? `${name} (${formatBytes(st.size)})` : name); } return out; } async function hasAnyTranscriptOnDisk(videoDir: string): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); return entries.some((e) => { if (e === "transcript.json") return true; const m = e.match( /^transcript\.([^.]+)\.(?:vtt|json|json3|srv1|srv2|srv3)$/, ); if (!m) return false; // live_chat is YouTube's chat replay, not a transcript of speech — don't // let its presence suppress the no-subs fallback to audio + whisper. return m[1] !== "live_chat"; }); } async function metadataReportsNoCaptions( videoDir: string, ): Promise { let raw: string; try { raw = await readFile( path.join(videoDir, "metadata.info.json"), "utf8", ); } catch { return false; } let parsed: { subtitles?: Record; automatic_captions?: Record; }; try { parsed = JSON.parse(raw); } catch { return false; } const subs = parsed.subtitles ?? {}; const auto = parsed.automatic_captions ?? {}; if (typeof subs !== "object" || typeof auto !== "object") return false; // live_chat is YouTube's chat replay, not a caption track. yt-dlp places // it under `subtitles`, so a video whose only listed track is live_chat // should be treated as having no captions for fallback purposes. const realSubs = Object.keys(subs).filter((k) => k !== "live_chat"); const realAuto = Object.keys(auto).filter((k) => k !== "live_chat"); return realSubs.length === 0 && realAuto.length === 0; } function trimError(stderrTail: string): string | undefined { const trimmed = stderrTail.trim().split("\n").slice(-3).join("\n"); return trimmed || undefined; } async function appendArchiveLine( channelDir: string, line: string, ): Promise { await mkdir(channelDir, { recursive: true }); await appendFile(path.join(channelDir, "archive"), line + "\n"); } export async function downloadOneManaged( opts: ManagedDownloadOpts, ): Promise { const startedAt = new Date().toISOString(); const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug); await mkdir(channelDir, { recursive: true }); // We pin yt-dlp's output to data// (see outputArgsForUrl), so we // know the dir before yt-dlp runs. Tee every log line into a per-video // download.log next to the sidecar (overwritten per managed download, like // download-outcome.json). Falls back to no-log when the URL has no canonical // id (outputArgsForUrl then uses %(id)s and the reconcile pass repairs it). const canonicalId = extractVideoId(opts.videoUrl); let logStream: WriteStream | null = null; if (canonicalId) { const dir = path.join(channelDir, "data", canonicalId); try { await mkdir(dir, { recursive: true }); const stream = createWriteStream(path.join(dir, "download.log")); logStream = stream; const base = opts.onLog; opts = { ...opts, onLog: (s) => { try { stream.write(s); } catch { /* logging is best-effort */ } base(s); }, }; } catch { /* per-video log is best-effort; fall back to opts.onLog */ } } try { // AN archive.org RECORD IS NOT A yt-dlp DOWNLOAD: its file comes over // BitTorrent or straight from archive.org, verified against the item's // checksums, and the record is built from the item's metadata // (controller/archiveOrgDownload.ts). Same log, same outcome sidecar. if (detectPlatform(opts.videoUrl) === "archiveorg") { return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps); } const outcome = await runManagedDownload(opts, channelDir, startedAt, canonicalId); await recordWaybackProvenance(opts, channelDir, canonicalId); return outcome; } finally { logStream?.end(); } } // A WAYBACK CAPTURE IS A COPY (lib/wayback.ts): once the record exists — its // metadata.info.json written by the prefetch or the download — the // `wayback.json` sidecar says of what. Built from the URL alone, so it costs no // request; never fails the download. async function recordWaybackProvenance( opts: ManagedDownloadOpts, channelDir: string, canonicalId: string | null, ): Promise { if (!canonicalId || !parseWaybackUrl(opts.videoUrl)) return; const videoDir = path.join(channelDir, "data", canonicalId); try { await access(path.join(videoDir, "metadata.info.json")); } catch { return; } try { await ensureWaybackProvenance(videoDir, opts.videoUrl, { onLog: opts.onLog }); } catch (err) { opts.onLog(`Wayback provenance not written: ${(err as Error).message}\n`); } } async function runManagedDownload( opts: ManagedDownloadOpts, channelDir: string, startedAt: string, canonicalId: string | null, ): Promise { const attempts: DownloadAttempt[] = []; let status: DownloadOutcomeStatus = "failed"; let fellBackToTranscribe = false; // Cookie policy for every yt-dlp spawn in this download. An omitted policy // behaves like the historical default: no prophylactic cookies, retry-only // (and with no value configured the retries degrade to no-ops). const cookiePolicy: ResolvedCookiePolicy = opts.cookiePolicy ?? { cookies: undefined, mode: DEFAULT_COOKIE_MODE, }; // Set when a failed metadata prefetch was recovered by its cookie retry: the // real download then gets cookies immediately (skipping a doomed cookie-less // attempt) and a success is reported as "ok-with-cookies". let prefetchNeededCookies = false; // Set when the download-time duration guard trips: the measured shortfall, // recorded on the outcome so the UI can explain it without re-probing. let shortAudioInfo: NonNullable | undefined; let lastArchiveLine: string | null = null; // The format the last media pass printed (FORMAT_MARKER), for the pointer. let lastFormatLine: string | null = null; // Full (untruncated) stderr tail of the most recent attempt, so a failed // download can be classified (rate_limit/network) against everything yt-dlp // printed — not just the last 3 lines stored on the attempt record. let lastFullTail = ""; // ---------- Attempt 0: metadata prefetch + app-level filters ---------- // Split the per-video download into a cheap metadata-only pass followed by // the real download. The metadata lets app-level filters (e.g. skip-live) // decide before we commit to a full download, and the real attempts reuse it // via --load-info-json (sourceArgs) so they don't re-extract. Only possible // when we know the canonical data// dir up front; for unidentifiable // URLs we fall back to the legacy single-call path (positional URL, // --write-info-json, no prefetch). The audio-check primary deliberately does // NOT reuse the info json — its long, restart-heavy download would outlive // yt-dlp's pinned (expirable) format URLs — so it re-extracts even though the // prefetch ran. let infoJsonPath: string | null = null; // This video's recency key (upload_date), captured from the prefetched // metadata so the keep-latest cutoff can classify it even before it's on disk. let videoUploadKey = canonicalId ? uploadKeyFor(undefined, canonicalId) : ""; // Source platform for the "auto" download-format selector. Seeded from the // channel config / URL and refined to the metadata extractor once prefetched // (e.g. lbry -> odysee), which is the authoritative signal. let resolvedPlatform: Platform | null = opts.channelConfig.platform ?? detectPlatform(opts.videoUrl); if (canonicalId && !opts.signal.aborted) { const videoDir = path.join(channelDir, "data", canonicalId); // In "always" mode the prefetch (like every other invocation) carries the // configured cookies up front. const prefetchCookies = alwaysCookies(cookiePolicy); const buildPrefetchArgs = (cookies: string | undefined) => [ "--ignore-config", "--restrict-filenames", ...outputArgsForUrl(opts.videoUrl), "--write-info-json", "--skip-download", "--no-write-subs", "--no-write-auto-subs", ...channelConfigArgs(opts.channelConfig, cookies), "--", opts.videoUrl, ]; // EVERY PREFETCH REWRITES metadata.info.json (no existence check — the // filters decide on today's metadata), so each spawn runs inside the // history wrap: what moved since the last version is appended to // metadata.history.json. The file yt-dlp writes is still the file. const prefetchHistory = { by: "prefetch" as const, ...(opts.persistOrigin?.requestedBy ? { requestedBy: opts.persistOrigin.requestedBy } : {}), onLog: opts.onLog, }; const prefetchRes = await withMetadataHistory( videoDir, prefetchHistory, () => runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchCookies)), ); lastFullTail = prefetchRes.stderrTail; const prefetchAvail = attemptSucceeded(prefetchRes.exitCode) ? undefined : parseUnavailableFromStderr(prefetchRes.stderrTail); attempts.push({ n: 0, kind: "metadata-prefetch", handling: opts.channelConfig.handling, usedCookies: Boolean(prefetchCookies), ytdlpExitCode: prefetchRes.exitCode, availabilityClass: prefetchAvail, error: attemptSucceeded(prefetchRes.exitCode) ? undefined : trimError(prefetchRes.stderrTail), }); // Prefetch auth retry: a prefetch that failed with an auth/age error is // re-run once with cookies (when the mode allows it and the failed pass // didn't already use them). A recovery here lets the real download start // with cookies immediately instead of burning a doomed cookie-less // attempt first. Defer mode lands here with no retry cookies, so the // failure stands and feeds the Needs-cookies bucket. const prefetchRetryCookies = authRetryCookies(cookiePolicy); if ( !attemptSucceeded(prefetchRes.exitCode) && prefetchAvail !== undefined && AUTH_RETRY_CLASSES.has(prefetchAvail) && prefetchRetryCookies !== undefined && !prefetchCookies && !opts.signal.aborted ) { opts.onLog( `Metadata prefetch auth-required (${prefetchAvail}); retrying with --cookies-from-browser ${prefetchRetryCookies}\n`, ); const retryRes = await withMetadataHistory( videoDir, prefetchHistory, () => runOneYtdlp(opts, channelDir, buildPrefetchArgs(prefetchRetryCookies)), ); lastFullTail = retryRes.stderrTail; const retryAvail = attemptSucceeded(retryRes.exitCode) ? undefined : parseUnavailableFromStderr(retryRes.stderrTail); attempts.push({ n: 0, kind: "metadata-prefetch-auth-retry", handling: opts.channelConfig.handling, usedCookies: true, ytdlpExitCode: retryRes.exitCode, availabilityClass: retryAvail, error: attemptSucceeded(retryRes.exitCode) ? undefined : trimError(retryRes.stderrTail), }); if (attemptSucceeded(retryRes.exitCode)) { prefetchNeededCookies = true; } } // A RATE-LIMITED PREFETCH ENDS THE VIDEO HERE (release 10, L2 review). // Every failed prefetch used to fall through to the real download, so a // 429, a bot check or YouTube's soft block ("…isn't available, try again // later") cost one more request into the same refusal before the batch // could back off. That request is only skipped for a `rate_limit` class: // any other failure (a 403, a removed video, an auth gate) keeps today's // flow, where attempt 1 and its own auth retry still get their chance. The // record carries `failureClass: "rate_limit"`, so the batch's cooldown and // abort and the runner's per-video deferral follow exactly as before. const lastPrefetch = attempts.at(-1); if ( lastPrefetch && !attemptSucceeded(lastPrefetch.ytdlpExitCode) && classifyDownloadFailure(lastFullTail, lastPrefetch.availabilityClass) === "rate_limit" ) { opts.onLog( `Metadata prefetch for ${canonicalId} was rate-limited by the source; ` + `not attempting the download (the platform backs off instead).\n`, ); return writeOutcome(opts, videoDir, { videoId: canonicalId, status: "failed", startedAt, attempts, lastFullTail, }); } const metaPath = path.join(videoDir, "metadata.info.json"); const metadata = await loadRawMetadata(metaPath); // Only wire --load-info-json into the real attempts when we actually have // the metadata file; a failed prefetch falls through to the legacy path so // the existing auth-retry logic still gets a chance. if (metadata) infoJsonPath = metaPath; if (metadata) resolvedPlatform = platformFromMetadata(metadata); videoUploadKey = uploadKeyFor(metadata?.upload_date, canonicalId); const decision = evaluateDownloadFilters({ metadata, channelConfig: opts.channelConfig, settings: { skipLiveDownloads: opts.globalSkipLiveDownloads ?? true }, onLog: opts.onLog, }); if (decision?.skip) { opts.onLog( `Skipping ${canonicalId}: ${decision.reason} [filter=${decision.filter}]\n`, ); // ── CHAT ONLY, AND IT RUNS BEFORE THE STORE IS WRITTEN ─────────────── // // The operator asked for this livestream's chat and not its media, so // this is the one rejection that FETCHES something — and the one that // keeps its prefetch directory. A chat-only dir holds the // metadata.info.json (without it buildIndex never sees the video and the // chat is published nowhere) plus transcript.live_chat.json and its // normalized live_chat.cues.json, beside the outcome sidecar and the log // every managed download leaves. Every "is it downloaded?" predicate // tests whisper, an English VTT or audio, so it still answers no — which // is right: this is a chat track, not a download. // // IT RUNS FIRST because its result belongs in the scan entry below. A // stream whose chat replay is OFF has to be remembered somewhere, or the // id sits in chatOnlyPending forever and every runner restart re-prefetches // it and re-runs a pass that will never return anything. let chatOnlyFetched = false; let chatOnlyUnavailable = false; if (decision.chatOnly && !opts.signal.aborted) { opts.onLog( `Fetching the live chat for ${canonicalId} (media skipped by the download filter).\n`, ); const chatRes = await fetchLiveChatOnly( opts, channelDir, alwaysCookies(cookiePolicy) ?? (prefetchNeededCookies ? authRetryCookies(cookiePolicy) : undefined), ); attempts.push({ n: 1, kind: "live-chat-only", handling: opts.channelConfig.handling, usedCookies: Boolean(alwaysCookies(cookiePolicy)), ytdlpExitCode: chatRes.exitCode, error: attemptSucceeded(chatRes.exitCode) ? undefined : trimError(chatRes.stderrTail), }); if (attemptSucceeded(chatRes.exitCode)) { chatOnlyFetched = await hasLiveChatOnDisk(videoDir); if (chatOnlyFetched) { // The CUES sidecar, not just the raw file: buildIndex prefers // live_chat.cues.json when it is fresh, so normalizing here makes // the video index-ready the moment it lands instead of waiting for // the corpus-wide normalize pass. try { await normalizeLiveChat({ videoDir, channelSlug: opts.channelSlug, ...(opts.channelConfig.name ? { configName: opts.channelConfig.name } : {}), log: (m: string) => opts.onLog(`${m}\n`), }); } catch (err) { opts.onLog( `Could not normalize the live chat: ${(err as Error).message}\n`, ); } } else { // A CLEAN PASS THAT FOUND NOTHING IS AN ANSWER, not a retry. Chat // replay was off for this stream, or the source has since dropped // it; asking again tomorrow gets the same nothing. Recorded on the // scan entry below so the video leaves chatOnlyPending for good and // settles as the ordinary filtered-out livestream it is. // // A FAILED pass (non-zero exit, a cooldown, an abort) is NOT this: // it leaves the flag alone and the id stays pending, which is the // whole reason this is gated on the exit code. chatOnlyUnavailable = true; opts.onLog( `No live chat was available for ${canonicalId}; recording that so it is not asked for again.\n`, ); } } } // A title-filter rejection feeds the metadata-scan store, so this video is // settled from here on WITHOUT a second metadata fetch. The store is the // one place a settled verdict is derived from; the outcome below records // only what happened, exactly as skip-live's does. A metadata-less skip // (the fail-closed branch) writes nothing here — there is nothing to // store, and it must stay retryable. if (decision.filter === "titleFilter" && metadata) { const entry = { title: metadata.title ?? "", description: metadata.description ?? "", uploadDate: metadata.upload_date ?? "", ...(metadata.live_status ? { liveStatus: metadata.live_status } : {}), ...(typeof metadata.duration === "number" ? { duration: metadata.duration } : {}), ...(chatOnlyUnavailable ? { noLiveChat: true } : {}), scannedAt: new Date().toISOString(), }; // A BATCH COLLECTS; A SINGLE VIDEO WRITES. The store is channel-level, // so writing it from inside a per-video loop is a full load + stringify // + rename per rejection — on a filtered channel that is one rewrite of // the whole file per non-matching video. The batch runner passes a // collector and writes once when it is done; a one-off download (no // collector) writes for itself, because nothing else will. if (opts.onFilterRejected) { opts.onFilterRejected(canonicalId, entry); } else { try { await upsertMetadataScan( opts.paths, opts.channelSlug, { entries: { [canonicalId]: entry } }, new Date().toISOString(), ); } catch (err) { opts.onLog( `Failed to record the metadata scan entry: ${(err as Error).message}\n`, ); } } } const finishedAt = new Date().toISOString(); const record: DownloadOutcomeRecord = { videoId: canonicalId, webpageUrl: opts.videoUrl, status: chatOnlyFetched ? "chat-only" : "skipped-filtered", startedAt, finishedAt, attempts, filter: { name: decision.filter, reason: decision.reason }, }; // The operator's own rejection takes its prefetch directory with it — // see discardPrefetchDir for why a metadata-only dir is not a neutral // leftover, and why the rule is an allow-list. Every other filter's skip // is RETRYABLE and its outcome sidecar is what `skippedByFilter` derives // from, so only this one discards. // // A fetched chat is not "mine to delete" either way — the allow-list // refuses the directory the moment transcript.live_chat.json is in it — // so this condition is a shortcut past a readdir, not the safety rail. const discarded = decision.filter === "titleFilter" && metadata && !chatOnlyFetched ? await discardPrefetchDir(videoDir, opts.onLog) : false; if (discarded) { opts.onLog( `Removed the metadata-only directory for ${canonicalId}: a filtered video is not a member of this corpus.\n`, ); } else { try { await mkdir(videoDir, { recursive: true }); // THE MEDIA TIER'S HOOK (release 17): what this run finalised moves // into channels//media when the channel has one. Never throws. await tierVideoDir(videoDir, { onLog: opts.onLog }); await writeDownloadOutcome(videoDir, record); } catch (err) { opts.onLog( `Failed to write download-outcome.json: ${(err as Error).message}\n`, ); } } return record; } } // Audio-check re-extracts fresh format URLs, so it never reuses the prefetch. const reuseInfoJson = infoJsonPath !== null; // ---------- Per-download persistence decision ---------- // Resolve the channel keep-latest rule (plus per-run overrides) into a concrete // plan: whether to keep the source video and who extracts the audio. Only // affects transcribe-handling downloads (and the youtube no-subs fallback, // which switches to transcribe). Cheap: a set/cutoff compare + one stat. const fmt: AudioFormat = opts.audioFormatOverride ?? opts.channelConfig.audioFormat ?? "mp3"; // The concrete yt-dlp `-f` selector, expanded from the resolved preset against // the source platform (so "auto" downloads `original` for Odysee). const downloadFormatSelector = resolveDownloadFormatSelector( opts.downloadFormatPreset ?? "auto", resolvedPlatform, ); // The container a persisting pass keeps. Platform-independent: a persist // wants a whole video, which "auto"'s Odysee rule is not about. const persistQuality = opts.persistFormatPreset ?? DEFAULT_SOURCE_VIDEO_QUALITY; const persistSelector = sourceVideoFormatSelector(persistQuality); const pinned = canonicalId ? await isDoNotClean(path.join(channelDir, "data", canonicalId)) : false; const plan = resolvePersistenceDecision({ inWindow: isInKeepWindow(videoUploadKey, opts.keepWindow), pinned, channelExtractionMode: opts.channelConfig.extractionMode, overrides: { keepSourceVideoOverride: opts.keepSourceVideoOverride, extractImmediately: opts.extractImmediately, }, }); // ---------- Attempt 1: primary ---------- const audioCheckEnabled = opts.channelConfig.handling === "transcribe" && opts.channelConfig.audioCheck?.enabled === true; // The shared media args for the primary + auth-retry attempts. Audio-check owns // its own args; youtube handling downloads subtitles (the persistence plan only // applies to the no-subs fallback below). For transcribe handling the plan // chooses audio-only vs. keep-source-video and yt-dlp vs. app extraction. const mediaArgs = opts.channelConfig.handling === "youtube" ? [ ...outputArgsForUrl(opts.videoUrl), ...youtubeHandlingArgs(opts.channelConfig, reuseInfoJson), ] : transcribeMediaArgs( opts.videoUrl, opts.channelConfig, plan, fmt, reuseInfoJson, downloadFormatSelector, persistSelector, ); if (opts.channelConfig.handling === "transcribe") { opts.onLog( `Persistence: ${plan.persist ? "keep source video" : "audio-only"} via ${plan.extractionMode} extraction (${plan.reason}).\n`, ); if (audioCheckEnabled && plan.persist) { opts.onLog( `Note: this video qualifies for source-video persistence, but the channel uses audio-check; persistence is skipped for audio-checked downloads.\n`, ); } } // Cookies for the real download attempts: always-mode passes them on every // invocation; otherwise a cookie-recovered prefetch means this video needs // them, so don't burn a doomed cookie-less attempt first. const primaryCookies = alwaysCookies(cookiePolicy) ?? (prefetchNeededCookies ? cookiePolicy.cookies : undefined); let primaryRes: AttemptOutcome; let audioCheckStats: AudioCheckAttemptStats | undefined; let audioCheckCorruptSource = false; // Complete download (yt-dlp exit 0) whose final probe stayed malformed after // one re-download. Terminal, kept on disk, NOT a failure (no retry/backoff). let audioCheckCorruptFullSource = false; // Orchestrator resolves data// by scanning the on-disk tree. We use // this in preference to extractVideoId(url), which doesn't know yt-dlp's // internal id (e.g. Odysee claim hashes vs URL slugs). let audioCheckVideoDir: string | null = null; if (audioCheckEnabled) { const primaryArgs = [ "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...outputArgsForUrl(opts.videoUrl), ...transcribeHandlingArgsForAudioCheck( opts.channelConfig, downloadFormatSelector, ), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(opts.channelConfig, primaryCookies), "--", opts.videoUrl, ]; // Hint the orchestrator at which data// subdir this launch will write // to, so its .part discovery and final-file resolution don't latch onto // stale .parts from prior interrupted attempts. outputArgsForUrl pins the // output to data// for every platform, so the canonical id is // the dir. const expectedVideoIdHint = canonicalId; const runAudioCheck = () => runAudioCheckedYtdlp({ paths: opts.paths, channelDir, channelConfig: opts.channelConfig, ytdlpArgs: primaryArgs, onLog: opts.onLog, signal: opts.signal, archiveMarker: ARCHIVE_MARKER, expectedVideoIdHint, }); // The audio-checked primary writes the info json itself (it re-extracts on // purpose), so it is the second rewrite of the file in one download and // gets its own history entry. Only when the dir is known up front — an // unidentifiable URL lands in data/%(id)s/, found only afterwards. const audioOutcome = canonicalId ? await withMetadataHistory( path.join(channelDir, "data", canonicalId), { by: "audio-check", ...(opts.persistOrigin?.requestedBy ? { requestedBy: opts.persistOrigin.requestedBy } : {}), onLog: opts.onLog, }, runAudioCheck, ) : await runAudioCheck(); primaryRes = { exitCode: audioOutcome.ytdlpExitCode, stderrTail: audioOutcome.stderrTail, archiveLine: audioOutcome.archiveLine, }; audioCheckStats = { checkpoints: audioOutcome.checkpoints.length, rollbacks: audioOutcome.rollbacks, restarts: audioOutcome.restarts, ...(audioOutcome.finalProbeVerdict ? { finalProbeVerdict: audioOutcome.finalProbeVerdict } : {}), }; audioCheckCorruptSource = audioOutcome.kind === "failed-corrupt-source"; audioCheckCorruptFullSource = audioOutcome.kind === "corrupt-full-source"; audioCheckVideoDir = audioOutcome.videoDir; } else { const primaryArgs = [ "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...mediaArgs, "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(opts.channelConfig, primaryCookies), ...sourceArgs(opts.videoUrl, infoJsonPath), ]; primaryRes = await runOneYtdlp(opts, channelDir, primaryArgs); } lastFullTail = primaryRes.stderrTail; const primaryAvail = attemptSucceeded(primaryRes.exitCode) ? undefined : parseUnavailableFromStderr(primaryRes.stderrTail); attempts.push({ n: 1, kind: audioCheckEnabled ? "audio-checked-primary" : "primary", handling: opts.channelConfig.handling, usedCookies: Boolean(primaryCookies), ytdlpExitCode: primaryRes.exitCode, availabilityClass: primaryAvail, error: attemptSucceeded(primaryRes.exitCode) ? undefined : trimError(primaryRes.stderrTail), ...(audioCheckStats ? { audioCheck: audioCheckStats } : {}), }); if (primaryRes.archiveLine) lastArchiveLine = primaryRes.archiveLine; if (primaryRes.formatLine) lastFormatLine = primaryRes.formatLine; // ---------- A subtitle 429 is not a failed download (release 17, RL) ---------- // YouTube's timedtext endpoint refuses per video while the media requests // succeed. When the subtitle fetch is all that failed, the download goes on // to the media and the record carries `subs_rate_limit` (a success): the // caller defers the video's SUBTITLES, never the platform. // - youtube handling runs the primary with --ignore-errors, so yt-dlp // exits 0 with the failure as a WARNING; the media pass below fetches // the audio when no transcript track arrived. // - any other primary that died on its subtitles alone (a channel whose // own args ask for subtitles) is run ONCE more with every subtitle // refused: a second spawn, the media, no timedtext request. // - ONLY a rate limit (review H1). --ignore-errors turns EVERY subtitle // failure into a WARNING and exit 0 — a 403/404/5xx, a failed live_chat // replay, a file it could not write. Any of those keeps today's meaning: // the attempt failed (not archived, classified from the tail), exactly // as the exit 1 it used to be. let subsRateLimited = false; let subtitleFailedOtherwise = false; if ( !audioCheckEnabled && !audioCheckCorruptSource && !audioCheckCorruptFullSource ) { if ( attemptSucceeded(primaryRes.exitCode) && hasNonRateLimitSubtitleFailure(primaryRes.stderrTail) ) { subtitleFailedOtherwise = true; const last = attempts.at(-1); if (last) { last.error = trimError(primaryRes.stderrTail); last.availabilityClass = parseUnavailableFromStderr(primaryRes.stderrTail); } opts.onLog( `The subtitle fetch for ${canonicalId ?? opts.videoUrl} failed (not a rate limit); ` + `the download is a failed attempt, as before.\n`, ); } else if ( attemptSucceeded(primaryRes.exitCode) && hasSubtitleRateLimit(primaryRes.stderrTail) ) { subsRateLimited = true; } else if ( !attemptSucceeded(primaryRes.exitCode) && !opts.signal.aborted && classifyDownloadFailure(primaryRes.stderrTail, primaryAvail) === "subs_rate_limit" ) { subsRateLimited = true; opts.onLog( `The subtitle fetch for ${canonicalId ?? opts.videoUrl} was rate-limited (HTTP 429) and nothing else failed; ` + `downloading without subtitles — they are deferred, the platform is not backed off.\n`, ); const noSubsRes = await runOneYtdlp(opts, channelDir, [ "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...mediaArgs, "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(opts.channelConfig, primaryCookies), // After the channel's own args: yt-dlp keeps the last occurrence. "--no-write-subs", "--no-write-auto-subs", ...sourceArgs(opts.videoUrl, infoJsonPath), ]); lastFullTail = noSubsRes.stderrTail; attempts.push({ n: 1, kind: "primary-without-subs", handling: opts.channelConfig.handling, usedCookies: Boolean(primaryCookies), ytdlpExitCode: noSubsRes.exitCode, availabilityClass: attemptSucceeded(noSubsRes.exitCode) ? undefined : parseUnavailableFromStderr(noSubsRes.stderrTail), error: attemptSucceeded(noSubsRes.exitCode) ? undefined : trimError(noSubsRes.stderrTail), }); if (noSubsRes.archiveLine) lastArchiveLine = noSubsRes.archiveLine; if (noSubsRes.formatLine) lastFormatLine = noSubsRes.formatLine; primaryRes = noSubsRes; if (!attemptSucceeded(noSubsRes.exitCode)) subsRateLimited = false; } } let lastSucceeded = !audioCheckCorruptSource && !audioCheckCorruptFullSource && !subtitleFailedOtherwise && attemptSucceeded(primaryRes.exitCode); if (lastSucceeded) { // "ok-with-cookies" keeps meaning "cookies were NEEDED" (the prefetch only // succeeded with them) — always-mode prophylactic cookies on a clean // success stay a plain "ok". status = audioCheckEnabled ? "ok-audio-checked" : prefetchNeededCookies ? "ok-with-cookies" : "ok"; } else if (audioCheckCorruptSource) { status = "failed-corrupt-source"; } else if (audioCheckCorruptFullSource) { // Kept file, terminal: not "ok" (no usable audio), not a failure (the // bytes are on disk; re-downloading is futile). Drops into its own bucket. status = "corrupt-full-source"; } // ---------- Attempt 2: auth retry ---------- // Off in defer mode (authRetryCookies -> undefined), so the failure is // recorded as-is and feeds the Needs-cookies bucket. Also skipped when the // failed attempt already used cookies — retrying identically is futile. const downloadRetryCookies = authRetryCookies(cookiePolicy); const shouldAuthRetry = !lastSucceeded && primaryAvail !== undefined && AUTH_RETRY_CLASSES.has(primaryAvail) && downloadRetryCookies !== undefined && !primaryCookies; if (shouldAuthRetry && !opts.signal.aborted) { opts.onLog( `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${downloadRetryCookies}\n`, ); const retryArgs = [ "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...mediaArgs, "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(opts.channelConfig, downloadRetryCookies), ...sourceArgs(opts.videoUrl, infoJsonPath), ]; const retryRes = await runOneYtdlp(opts, channelDir, retryArgs); lastFullTail = retryRes.stderrTail; const retryAvail = attemptSucceeded(retryRes.exitCode) ? undefined : parseUnavailableFromStderr(retryRes.stderrTail); attempts.push({ n: 2, kind: "auth-retry", handling: opts.channelConfig.handling, usedCookies: true, ytdlpExitCode: retryRes.exitCode, availabilityClass: retryAvail, error: attemptSucceeded(retryRes.exitCode) ? undefined : trimError(retryRes.stderrTail), }); if (retryRes.archiveLine) lastArchiveLine = retryRes.archiveLine; if (retryRes.formatLine) lastFormatLine = retryRes.formatLine; if (attemptSucceeded(retryRes.exitCode)) { lastSucceeded = true; status = "ok-with-cookies"; } } const videoDir = audioCheckVideoDir ?? path.join(channelDir, "data", canonicalId ?? "unknown"); const videoId = path.basename(videoDir); // The pointer's `format`, read off the media pass that just ran, logged as it // is recorded (persistedFormatLog). Only a persisting plan keeps a container. const persistedFormat = (): SavedVideoFormat | null => { if (!plan.persist) return null; const format = savedVideoFormatFromLine(lastFormatLine, persistQuality); const line = persistedFormatLog(format, lastFormatLine); if (line) opts.onLog(line); return format; }; // Download-time duration guard: probe the produced audio. and compare its // actual length to the metadata duration. A large shortfall means the source // served a truncated stream (e.g. a CDN-truncated HLS rung) even though yt-dlp // exited 0. Returns the shortfall metrics when tripped, else null. Skips // livestreams (unreliable durations), short videos, and unmeasurable files // (probe failure -> null -> no false positive). The file is left on disk; the // caller marks the download failed-short-audio so it isn't transcribed. const probeShortAudio = async (): Promise< NonNullable | null > => { const meta = await loadRawMetadataFromDir(videoDir); const expected = meta?.duration; if ( !meta || isLivestreamMetadata(meta) || typeof expected !== "number" || expected < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC ) { return null; } const audioDurationSec = await probeMediaDurationSec({ ffprobeBin: opts.paths.ffprobeBin, file: path.join(videoDir, `audio.${fmt}`), signal: opts.signal, onLog: opts.onLog, }); if (!isShortAudio(audioDurationSec, expected, { isLivestream: false })) { return null; } const a = audioDurationSec as number; return { audioDurationSec: Math.round(a * 10) / 10, expectedDurationSec: expected, coverage: Math.round((a / expected) * 1000) / 1000, }; }; // ---------- App-side extraction (transcribe handling) ---------- // After a successful non-audio-check transcribe download in app mode, produce // audio. from the downloaded source-media container and keep or discard it // per the persistence plan. (Audio-check owns its own extraction; youtube // handling extracts inside the no-subs fallback below.) if ( lastSucceeded && !audioCheckEnabled && opts.channelConfig.handling === "transcribe" && plan.extractionMode === "app" && !opts.signal.aborted ) { await finalizeAppExtraction({ paths: opts.paths, channelSlug: opts.channelSlug, channelConfig: opts.channelConfig, videoDir, videoId, fmt, persist: plan.persist, category: plan.category, origin: opts.persistOrigin, format: persistedFormat(), onLog: opts.onLog, signal: opts.signal, }); } // ---------- Attempt 3: no-subs fallback (youtube handling only) ---------- // Also the youtube-handling MEDIA download a `forceMedia` caller asked for: // this is the only pass that fetches media for a youtube-handling channel, // and its own gate ("no transcript and no captions") is exactly the reason // "Persist source video" and the whole-recording fetch used to do nothing on // a video that already had a transcript — two subtitle passes, no file. if ( lastSucceeded && opts.channelConfig.handling === "youtube" && !opts.signal.aborted ) { const hasTranscript = await hasAnyTranscriptOnDisk(videoDir); const noCaptions = await metadataReportsNoCaptions(videoDir); // A subtitle 429 left no transcript: the media is fetched exactly as for a // video with no captions, and the subtitles wait (release 17, slice RL). const subsDeferredFallback = subsRateLimited && !hasTranscript; const noSubsFallback = !hasTranscript && (noCaptions || subsDeferredFallback); // Forced only when the fallback would NOT have run on its own: a video with // no transcript and no captions takes today's path whoever asked. const forced = !noSubsFallback && opts.forceMedia === true; // THE TRANSCRIPT ON DISK IS NEVER TOUCHED. A forced pass over a video that // has one fetches the source and persists it, and does nothing else: no // subtitles (refused on the command line, after the channel's own args, so // a channel carrying --write-auto-subs cannot overwrite the transcript), no // audio extraction, no transcription, no short-audio verdict on audio it // never made. With NO transcript on disk (captions listed, none fetched — // a language the channel's sub-langs does not match) the forced pass is // today's fallback in full: download, extract, transcribe inline if // configured, persist. const keepTranscript = forced && hasTranscript; if (keepTranscript && !plan.persist) { // PERSIST-ONLY WITH NOTHING TO PERSIST. Every forceMedia caller sets the // keep-source override, so the plan persists; a caller that did not would // be asking for a download with no destination — the transcript rules // out audio, and the plan rules out the container. opts.onLog( `forceMedia: a transcript is on disk and this download keeps no source video (${plan.reason}); nothing to fetch.\n`, ); } else if (noSubsFallback || forced) { if (forced) { opts.onLog( `forceMedia: downloading the source although ${ hasTranscript ? "a transcript is on disk" : "captions exist" }\n`, ); } else if (subsDeferredFallback) { opts.onLog( `The subtitles for ${videoId} were rate-limited (HTTP 429); downloading the media anyway — ` + `the subtitles are deferred for download-missing-subs, the platform is not backed off${plan.persist ? " (keeping source video)" : ""}.\n`, ); } else { opts.onLog( `No subs available for ${videoId}; falling back to audio download + whisper${plan.persist ? " (keeping source video)" : ""}.\n`, ); } const fallbackConfig: ChannelConfig = { ...opts.channelConfig, handling: "transcribe", }; // Cookies for the fallback: always-mode passes them like every other // invocation; otherwise only when this video demonstrably needed them // (an attempt succeeded with cookies -> "ok-with-cookies"). const fallbackCookieOverride = alwaysCookies(cookiePolicy) ?? (status === "ok-with-cookies" ? cookiePolicy.cookies : undefined); // Feed yt-dlp the metadata it already wrote during the primary // attempt instead of re-querying the extractor — saves a network // round-trip per video, which adds up across batch runs and helps // dodge rate limits we'd otherwise burn on info we already have. The // persistence plan applies here too, so a kept youtube video that lacks // captions still archives its source container. const fallbackArgs = [ "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...transcribeMediaArgs( opts.videoUrl, fallbackConfig, plan, fmt, true, downloadFormatSelector, persistSelector, ), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(fallbackConfig, fallbackCookieOverride), // yt-dlp takes the LAST occurrence of an option, so the refusals go // after the channel's own args (the chat-only pass's rule). // A deferred subtitle fetch is not asked again in the same minute. ...(keepTranscript || subsDeferredFallback ? ["--no-write-subs", "--no-write-auto-subs"] : []), ...sourceArgs( opts.videoUrl, path.join(videoDir, "metadata.info.json"), ), ]; const fallbackRes = await runOneYtdlp(opts, channelDir, fallbackArgs); lastFullTail = fallbackRes.stderrTail; const fallbackAvail = attemptSucceeded(fallbackRes.exitCode) ? undefined : parseUnavailableFromStderr(fallbackRes.stderrTail); attempts.push({ n: 3, kind: "no-subs-fallback", handling: "transcribe", usedCookies: Boolean(fallbackCookieOverride), ytdlpExitCode: fallbackRes.exitCode, availabilityClass: fallbackAvail, error: attemptSucceeded(fallbackRes.exitCode) ? undefined : trimError(fallbackRes.stderrTail), }); if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine; if (fallbackRes.formatLine) lastFormatLine = fallbackRes.formatLine; if (attemptSucceeded(fallbackRes.exitCode) && keepTranscript) { // Persist only. Not a fallback to transcription — the transcript is // the one already on disk — so `fellBackToTranscribe` stays unset and // the status stays what the subtitle pass made it. if (plan.extractionMode === "app") { await finalizeAppExtraction({ paths: opts.paths, channelSlug: opts.channelSlug, channelConfig: opts.channelConfig, videoDir, videoId, fmt, persist: plan.persist, category: plan.category, origin: opts.persistOrigin, format: persistedFormat(), extractAudio: false, onLog: opts.onLog, signal: opts.signal, }); } } else if (attemptSucceeded(fallbackRes.exitCode)) { fellBackToTranscribe = true; // In app mode, extract audio. from the downloaded source container // (and keep/discard it) before any inline whisper can read the audio. if (plan.extractionMode === "app") { await finalizeAppExtraction({ paths: opts.paths, channelSlug: opts.channelSlug, channelConfig: opts.channelConfig, videoDir, videoId, fmt, persist: plan.persist, category: plan.category, origin: opts.persistOrigin, format: persistedFormat(), onLog: opts.onLog, signal: opts.signal, }); } const shortBeforeInline = await probeShortAudio(); if (shortBeforeInline) { shortAudioInfo = shortBeforeInline; status = "failed-short-audio"; lastSucceeded = false; opts.onLog( `Short audio (${shortBeforeInline.audioDurationSec}s of ` + `${shortBeforeInline.expectedDurationSec}s); skipping whisper and ` + `keeping the file for re-download with a different format.\n`, ); } else if (opts.inlineTranscribeOnFallback) { // Inline whisper: matches whisperVideoAction's shape. Routes through // the worker pool so the inline transcription respects worker config // and slot limits like any other. try { await transcribeWithWorker({ paths: opts.paths, videoDir, videoId, audioFilename: `audio.${fmt}`, onLog: opts.onLog, signal: opts.signal, }); status = "ok-auto-transcribed"; } catch (err) { opts.onLog( `Whisper failed after no-subs fallback: ${(err as Error).message}\n`, ); status = "failed"; lastSucceeded = false; } } else { opts.onLog( `Audio downloaded for ${videoId}; skipping inline whisper (inlineTranscribeOnFallback=off). Run "Transcribe missing" to transcribe.\n`, ); // status remains whatever the primary attempt set; the // fellBackToTranscribe flag distinguishes from a normal "ok" // so the UI and downstream jobs can tell what happened. } } else if (keepTranscript) { // THE MEDIA FAILED, THE DOWNLOAD DID NOT (release 11 slice O3). A // persist-only pass over a transcript is an extra the caller asked for, // not the download: the subtitle pass succeeded and the transcript is // on disk, so the status stays the subtitle pass's and only the failed // media attempt is recorded (n: 3 above, with its error). Marking the // whole download failed put "Download failed" on a video whose // transcript is fine. A caller that asked for the source reads the // attempt through sourceFetchFailure. yt-dlp's partial container is // left where it is — a retry resumes it — and named here, because // nothing else surfaces a source-media partial. const partials = await sourceMediaPartials(videoDir); opts.onLog( `forceMedia: the source download failed (yt-dlp exit ${fallbackRes.exitCode}); ` + `the transcript on disk is untouched and the download stays ${status}.` + (partials.length ? ` Left for a retry to resume: ${partials.join(", ")}.` : "") + "\n", ); } else { status = "failed"; lastSucceeded = false; } } } // ---------- Duration guard (short-audio) ---------- // Covers the audio-check and non-inline transcribe paths (the inline-whisper // path checked before transcribing). A truncated download is marked failed so // it isn't archived or transcribed; the short file is left on disk so it isn't // immediately re-downloaded into a loop. if ( lastSucceeded && status !== "failed-short-audio" && (opts.channelConfig.handling === "transcribe" || fellBackToTranscribe) && !opts.signal.aborted ) { const short = await probeShortAudio(); if (short) { shortAudioInfo = short; status = "failed-short-audio"; lastSucceeded = false; opts.onLog( `Short audio for ${videoId}: ${short.audioDurationSec}s of ` + `${short.expectedDurationSec}s (${Math.round(short.coverage * 100)}% ` + `of the video). The source served a truncated stream; keeping the ` + `file and flagging it. Re-download with a different format ` + `(e.g. Original) to fix.\n`, ); } } // ---------- Archive append ---------- if (lastSucceeded && opts.appendArchive !== false && lastArchiveLine) { try { await appendArchiveLine(channelDir, lastArchiveLine); } catch (err) { opts.onLog( `Archive append failed (${lastArchiveLine}): ${(err as Error).message}\n`, ); } } // ---------- Sidecar ---------- return writeOutcome(opts, videoDir, { videoId, status, startedAt, attempts, lastFullTail, fellBackToTranscribe, shortAudio: shortAudioInfo, subsRateLimited, }); } // THE SIDECAR AND THE AVAILABILITY HISTORY, written one way by every exit that // ends a download's attempts: the end of the main path, and a prefetch the // source rate-limited (release 10, L2 review). A filter-skip writes its own // record and has no failure class; it does not come through here. async function writeOutcome( opts: ManagedDownloadOpts, videoDir: string, o: { videoId: string; status: DownloadOutcomeStatus; startedAt: string; attempts: DownloadAttempt[]; lastFullTail: string; fellBackToTranscribe?: boolean; shortAudio?: DownloadOutcomeRecord["shortAudio"]; // The subtitle fetch alone answered 429 and the download went on. subsRateLimited?: boolean; }, ): Promise { const { status, attempts } = o; const finishedAt = new Date().toISOString(); // Classify a failed download against the FULL stderr tail of the last attempt // so a rate-limit logged as a WARNING (then masked by a different final error) // is still caught. Skipped for successes and filter-skips — except that a // success whose subtitles were rate-limited says so (`subs_rate_limit`), so // its caller defers the subtitles (release 17, slice RL). const isFailure = status === "failed" || status === "failed-corrupt-source"; const failureClass = isFailure ? classifyDownloadFailure(o.lastFullTail, attempts.at(-1)?.availabilityClass) : o.subsRateLimited && status.startsWith("ok") ? ("subs_rate_limit" as const) : undefined; const record: DownloadOutcomeRecord = { videoId: o.videoId, webpageUrl: opts.videoUrl, status, startedAt: o.startedAt, finishedAt, attempts, ...(failureClass ? { failureClass } : {}), ...(o.fellBackToTranscribe ? { fellBackToTranscribe: true } : {}), ...(o.shortAudio ? { shortAudio: o.shortAudio } : {}), }; // Only write the sidecar if we know which dir to put it in. If the very first // attempt failed before metadata could be written, the data/ dir may not // exist yet; in that case `mkdir -p` it so the sidecar lands somewhere. try { await mkdir(videoDir, { recursive: true }); // THE MEDIA TIER'S HOOK (release 17), after reconcileVideoDirs: what this // run finalised — yt-dlp's own `-x` output, the app's extraction, the live // chat — moves into channels//media when the channel has one, and a // relative link stays. Never throws: on a classic channel, or a media // drive that is not there, the files stay real. await tierVideoDir(videoDir, { onLog: opts.onLog }); await writeDownloadOutcome(videoDir, record); } catch (err) { opts.onLog( `Failed to write download-outcome.json: ${(err as Error).message}\n`, ); } // Capture an availability history entry when the last attempt observed a // class (e.g. a public video the uploader just deleted). Never touch the // top-level fields (updateTopLevel: false) so resolveEffectiveAvailability's // recency comparison against download-outcome.json is unchanged. const lastAttempt = attempts.at(-1); if (lastAttempt?.availabilityClass) { try { await recordAvailability( videoDir, { availability: lastAttempt.availabilityClass, observedAt: finishedAt, source: "download", }, { updateTopLevel: false }, ); } catch (err) { opts.onLog( `Failed to record availability history: ${(err as Error).message}\n`, ); } } return record; }