// RE-READ ONE VIDEO'S METADATA, AND NOTHING ELSE. // // The motivating case is a YouTube livestream that has just ended. Its VOD // starts life with one DASH-fragmented audio format and no captions; hours // later YouTube finishes processing it and the same URL offers plain https // formats and auto-captions. The metadata.info.json on disk still says what // the source said the first time, and everything downstream (the download // filters, the audio check, the subtitle bucket) decides on that file. Before // this, the only in-app ways to re-read it were a channel-wide metadata scan // (which writes the scan store, not the video's file) or a full re-download. // // WHAT IT IS: one yt-dlp spawn with `--skip-download --write-info-json`, for // one video, with exactly what the managed download's metadata prefetch // (ytdlp/downloadOneManaged.ts, attempt 0) carries — the platform's args at its // current pace, the cookie policy (`alwaysCookies` up front, one // `authRetryCookies` retry after an auth failure, none in "defer"), and the // channel's own extra args. It runs inside `withMetadataHistory` as writer // "refresh", so what moved lands in metadata.history.json like every other // rewrite. // // WHAT IT IS NOT: a download. No subtitles, no media, no archive line, no // download-outcome.json (that sidecar records download ATTEMPTS, and the video // page and the buckets read it as one), no download.log. It never deletes or // moves anything — and the negations go AFTER the channel's own args, because // yt-dlp takes the last occurrence of an option and `ytdlpExtraArgs` is free // text: a channel carrying `--write-subs` must not turn this into a subtitle // fetch. The output template is pinned to `data//` — the directory // being refreshed — rather than derived from the URL, so a source whose // URL-derived id differs from the directory name still rewrites THIS file. // // NOT FOR A RECORD yt-dlp DID NOT WRITE. An archive.org record, a Wayback copy // and a record completed from a podcast feed hold metadata their own writers // put there; a yt-dlp re-read would replace it with what the bare URL says (the // hazard the persist prefetch already has with Wayback titles). Those are // refused, by URL and by the writers in the history. // // A RATE LIMIT IS RECORDED, NOT RETRIED: the caller's `onPlatformBackoff` // writes the shared per-platform cooldown the download lane honours, and the // job fails. A clean answer settles the platform (`onPlatformClean`), as a // clean metadata scan does. // // The job around it (queue, kind, refusals) is refreshVideoMetadataJob.ts; // this module needs no registry, so a test drives it with a stub runner. import path from "node:path"; import { readFile, stat } from "node:fs/promises"; import { AUTH_RETRY_CLASSES, classifyDownloadFailure, parseUnavailableFromStderr, } from "../lib/availability"; import type { ChannelConfig } from "../lib/channelConfig"; import { DEFAULT_COOKIE_MODE, alwaysCookies, authRetryCookies, type ResolvedCookiePolicy, } from "../lib/cookiePolicy"; import type { MetadataHistoryEntry } from "../lib/metadataHistory"; import { loadMetadataHistory, withMetadataHistory, } from "../lib/metadataHistory-server"; import { detectPlatform } from "../lib/platform"; import { parseWaybackUrl } from "../lib/wayback"; import type { Paths } from "../lib/paths"; import { channelExtraArgs } from "../ytdlp/channelArgs"; import { runOneYtdlp, type AttemptOutcome, } from "../ytdlp/runOneYtdlp"; import { findVideoSourceUrl } from "./undownloadedVideos"; const INFO_JSON = "metadata.info.json"; // One path segment, the shape of every data// name. Checked before the id // reaches a path.join or an `-o` template. export function isVideoIdSegment(id: string): boolean { return ( id.length > 0 && id !== "." && id !== ".." && !/[/\\\0]/.test(id) ); } export type RefreshTarget = | { ok: true; url: string } | { ok: false; error: string }; // The refusal for an id with no directory. Exported for the tests. export function notFetchedRefusal(videoId: string, slug: string): string { return ( `"${videoId}" has not been fetched into ${slug} yet — sync, import or ` + `download it first; a refresh only re-reads a video already archived.` ); } // WHICH URL TO RE-READ, OR WHY NOT. Only a video this channel already has a // directory for: it resolves the way every per-video action does (its // metadata's webpage_url, else the playlist, else a URL rebuilt from the id // and the platform). // // NO DIRECTORY IS A REFUSAL, never one created. A data// holding a // metadata.info.json is not neutral (see PREFETCH_OWN_FILES in // ytdlp/downloadOneManaged.ts): buildIndex admits any such directory to the // index and the published site, and deriveChannelSets reads the name as // "ever fetched". The metadata scan creates none for the same reason, and // neither does this — the check is made before yt-dlp is spawned, and again // by refreshVideoMetadata itself. export async function resolveRefreshTarget( paths: Paths, slug: string, videoId: string, config: ChannelConfig, ): Promise { if (!isVideoIdSegment(videoId)) { return { ok: false, error: `"${videoId}" is not a video id (one path segment)` }; } const videoDir = path.join(paths.channelsDir, slug, "data", videoId); if (!(await isDirectory(videoDir))) { return { ok: false, error: notFetchedRefusal(videoId, slug) }; } const url = await findVideoSourceUrl(paths, slug, videoId, config); if (!url) { return { ok: false, error: "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", }; } const notYtdlp = await notYtdlpMetadata(videoDir, url); if (notYtdlp) { return { ok: false, error: `${videoId}'s metadata is not yt-dlp's to rewrite: ${notYtdlp}. ` + `A yt-dlp re-read would overwrite it.`, }; } return { ok: true, url }; } async function isDirectory(p: string): Promise { return stat(p) .then((s) => s.isDirectory()) .catch(() => false); } // The writers that complete a record yt-dlp cannot describe. A record one of // them has touched holds what they wrote — a title found for a raw Wayback // file, an archive.org item's own metadata, a podcast episode's feed entry — // and a yt-dlp re-read replaces the whole file with what the URL alone says. const NON_YTDLP_WRITERS: ReadonlySet = new Set([ "feed-backfill", "archiveorg-provenance", "archiveorg-import", "wayback-provenance", ]); // WHY THIS RECORD'S METADATA IS NOT yt-dlp's, or null when it is. Asked of // the URL first (an archive.org record is written whole from the item API // with no history entry — a first write is not a rewrite — and a Wayback // copy's title is the operator's), then of the history. async function notYtdlpMetadata( videoDir: string, url: string, ): Promise { if (detectPlatform(url) === "archiveorg") { return "an archive.org record is written from the item's metadata"; } if (parseWaybackUrl(url)) { return "a Wayback Machine copy's title and date are set from its provenance"; } const history = await loadMetadataHistory(videoDir).catch(() => null); const by = history?.entries.find((e) => NON_YTDLP_WRITERS.has(e.by))?.by; return by ? `it was completed by ${by}` : null; } // The argv of the one spawn. Exported for the test, which pins it. export function buildRefreshArgs(opts: { videoId: string; videoUrl: string; channelConfig: ChannelConfig; cookies: string | undefined; // The platform pace; default: its current one (channelExtraArgs). Tests // pass it. paceSeconds?: number; }): string[] { const dir = `data/${opts.videoId}`; return [ "--ignore-config", "--restrict-filenames", "--no-playlist", // One request per second, the floor this repo applies everywhere it // touches a source; the platform's own (adaptive) pace below raises it. "--sleep-requests", "1", ...channelExtraArgs(opts.channelConfig, opts.cookies, opts.paceSeconds), // AFTER the channel's own args, so they cannot be undone by them (see the // header). `-o` likewise: the info json lands in THIS video's directory. "--skip-download", "--write-info-json", "--no-write-subs", "--no-write-auto-subs", "--no-write-description", "--no-write-thumbnail", "--no-write-comments", "--no-download-archive", "--no-write-playlist-metafiles", "-o", `${dir}/audio.%(ext)s`, "-o", `infojson:${dir}/metadata`, "--", opts.videoUrl, ]; } // --- the closing summary ------------------------------------------------------ type Json = Record; function isObject(v: unknown): v is Json { return typeof v === "object" && v !== null && !Array.isArray(v); } function isAudioOnly(f: Json): boolean { return ( f.vcodec === "none" && typeof f.acodec === "string" && f.acodec !== "none" ); } // Fragmented = yt-dlp has to stitch it from pieces (DASH segments, HLS). A // plain https format is one ranged file, which is what a download wants. function isFragmented(f: Json): boolean { const protocol = typeof f.protocol === "string" ? f.protocol : ""; if (Array.isArray(f.fragments) && f.fragments.length > 0) return true; return !(protocol === "https" || protocol === "http"); } // `140 m4a http_dash_segments 144k`. function describeAudioFormat(f: Json): string { const kbps = typeof f.abr === "number" && f.abr > 0 ? f.abr : typeof f.tbr === "number" && f.tbr > 0 ? f.tbr : null; return [ String(f.format_id ?? "?"), String(f.ext ?? "?"), String(f.protocol ?? "?"), ...(kbps !== null ? [`${Math.round(kbps)}k`] : []), ].join(" "); } // The English track names of a caption map (en, en-US, en-orig, …), sorted. function englishTracks(map: unknown): string[] { if (!isObject(map)) return []; return Object.keys(map) .filter((k) => /^en(-|$)/i.test(k)) .sort(); } function list(values: string[]): string { return values.length ? values.join(", ") : "none"; } // The content keys that moved across every entry this refresh appended (an // auth retry can append a second), in the order they first appear. function movedKeys(entries: MetadataHistoryEntry[]): { changed: string[]; added: string[]; removed: string[]; } { const changed = new Set(); const added = new Set(); const removed = new Set(); for (const e of entries) { for (const k of Object.keys(e.changed)) changed.add(k); for (const k of Object.keys(e.added)) added.add(k); for (const k of Object.keys(e.removed)) removed.add(k); } return { changed: [...changed], added: [...added], removed: [...removed] }; } // WHAT THE SOURCE SAYS NOW, in the lines an operator waiting on a VOD reads: // is it still live, how many formats, which audio formats and whether any of // them is a plain file, which English captions exist, and what moved. // // PURE: the parsed file and the history entries arrive as arguments. export function refreshSummaryLines(input: { videoId: string; meta: Json; entries: MetadataHistoryEntry[]; // Was there a metadata.info.json before this pass? With none there is // nothing to compare against, and the history (correctly) records nothing. hadBefore: boolean; }): string[] { const { meta } = input; const formats = Array.isArray(meta.formats) ? meta.formats.filter(isObject) : []; const audio = formats.filter(isAudioOnly); const plainAudio = audio.filter((f) => !isFragmented(f)); const moved = movedKeys(input.entries); let changedLine: string; if (!input.hadBefore) { changedLine = "first metadata for this video (nothing to compare)"; } else if (input.entries.length === 0) { changedLine = "nothing (the file is byte-identical)"; } else if ( moved.changed.length + moved.added.length + moved.removed.length === 0 ) { changedLine = "no content keys (formats, URLs or counters only)"; } else { changedLine = [ ...moved.changed, ...moved.added.map((k) => `+${k}`), ...moved.removed.map((k) => `-${k}`), ].join(", "); } return [ `Metadata for ${input.videoId} now says:`, ` live_status: ${typeof meta.live_status === "string" ? meta.live_status : "(not set)"}`, ` formats: ${formats.length}`, ` audio-only formats: ${list(audio.map(describeAudioFormat))}`, ` non-fragmented audio: ${ plainAudio.length ? `yes (${plainAudio.map((f) => String(f.format_id ?? "?")).join(", ")})` : "no" }`, ` English subtitles: ${list(englishTracks(meta.subtitles))}`, ` English automatic captions: ${list(englishTracks(meta.automatic_captions))}`, ` changed: ${changedLine}`, ]; } // --- the pass ----------------------------------------------------------------- export type RefreshRunner = ( cwd: string, args: string[], ) => Promise; export type RefreshVideoMetadataOpts = { paths: Paths; slug: string; videoId: string; videoUrl: string; channelConfig: ChannelConfig; // Resolved channel-over-global (resolveCookiePolicy). Omitted = the // historical default: no prophylactic cookies, retry-only. cookiePolicy?: ResolvedCookiePolicy; onLog: (s: string) => void; signal: AbortSignal; // The source rate-limited us: record the shared platform cooldown. onPlatformBackoff?: (failureClass: "rate_limit") => Promise | void; // The source answered cleanly: settle the platform. Returns the line for the // log, or null when there was nothing to settle. onPlatformClean?: () => Promise | string | null; // Who asked, recorded on the history entry ("ops", …). Optional. requestedBy?: string; // Test seams: the spawn, and the platform pace in the argv. run?: RefreshRunner; paceSeconds?: number; }; export type RefreshVideoMetadataResult = { // The parsed file after the pass. meta: Json; entries: MetadataHistoryEntry[]; usedCookies: boolean; summary: string[]; }; function succeeded(exitCode: number | null): boolean { // yt-dlp: 0 = clean, 101 = a clean stop (break-on-existing / max-downloads). return exitCode === 0 || exitCode === 101; } function tail(stderr: string): string { return stderr.trim().split("\n").slice(-3).join(" / "); } // Throws on a failed pass (the job ends `failed` with the reason); on success // logs the summary and returns it. export async function refreshVideoMetadata( opts: RefreshVideoMetadataOpts, ): Promise { const channelDir = path.join(opts.paths.channelsDir, opts.slug); const videoDir = path.join(channelDir, "data", opts.videoId); const infoPath = path.join(videoDir, INFO_JSON); const policy: ResolvedCookiePolicy = opts.cookiePolicy ?? { cookies: undefined, mode: DEFAULT_COOKIE_MODE, }; const run: RefreshRunner = opts.run ?? ((cwd, args) => runOneYtdlp( { ytdlpBin: opts.paths.ytdlpBin, onLog: opts.onLog, signal: opts.signal }, cwd, args, )); // Never create the directory (see resolveRefreshTarget): refused before // yt-dlp is spawned, whose `-o` template would otherwise make it. if (!(await isDirectory(videoDir))) { throw new Error(notFetchedRefusal(opts.videoId, opts.slug)); } const hadBefore = await stat(infoPath) .then((s) => s.isFile()) .catch(() => false); const entries: MetadataHistoryEntry[] = []; const history = { by: "refresh" as const, ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), onLog: opts.onLog, onEntry: (e: MetadataHistoryEntry) => entries.push(e), }; const pass = (cookies: string | undefined) => withMetadataHistory(videoDir, history, () => run( channelDir, buildRefreshArgs({ videoId: opts.videoId, videoUrl: opts.videoUrl, channelConfig: opts.channelConfig, cookies, ...(opts.paceSeconds !== undefined ? { paceSeconds: opts.paceSeconds } : {}), }), ), ); opts.onLog(`Refreshing the metadata of ${opts.videoId} from ${opts.videoUrl}\n`); const firstCookies = alwaysCookies(policy); let usedCookies = Boolean(firstCookies); let outcome = await pass(firstCookies); // The prefetch's auth retry, verbatim in effect: once, with cookies, when // the failure is an auth/age gate, the mode allows it, and the failed pass // did not already carry them. "defer" has no retry cookies, so the failure // stands. if (!succeeded(outcome.exitCode) && !opts.signal.aborted) { const cls = parseUnavailableFromStderr(outcome.stderrTail); const retryCookies = authRetryCookies(policy); if ( classifyDownloadFailure(outcome.stderrTail, cls) !== "rate_limit" && AUTH_RETRY_CLASSES.has(cls) && retryCookies !== undefined && !firstCookies ) { opts.onLog( `Metadata refresh auth-required (${cls}); retrying with --cookies-from-browser ${retryCookies}\n`, ); usedCookies = true; outcome = await pass(retryCookies); } } if (opts.signal.aborted) throw new Error("Cancelled"); if (!succeeded(outcome.exitCode)) { const cls = parseUnavailableFromStderr(outcome.stderrTail); if (classifyDownloadFailure(outcome.stderrTail, cls) === "rate_limit") { await opts.onPlatformBackoff?.("rate_limit"); throw new Error( `The source rate-limited the metadata refresh of ${opts.videoId}; ` + `a per-platform cooldown has been recorded. Nothing was rewritten. ` + `(${tail(outcome.stderrTail)})`, ); } throw new Error( `yt-dlp could not read the metadata of ${opts.videoId} ` + `(exit ${outcome.exitCode ?? "null"}): ${tail(outcome.stderrTail)}`, ); } try { const line = await opts.onPlatformClean?.(); if (line) opts.onLog(line); } catch { /* shared-state write is best-effort */ } let meta: unknown; try { meta = JSON.parse(await readFile(infoPath, "utf8")); } catch (err) { throw new Error( `yt-dlp exited cleanly but ${path.relative(channelDir, infoPath)} ` + `could not be read: ${(err as Error).message}`, ); } if (!isObject(meta)) { throw new Error(`${path.relative(channelDir, infoPath)} is not a JSON object`); } const summary = refreshSummaryLines({ videoId: opts.videoId, meta, entries, hadBefore, }); opts.onLog(`${summary.join("\n")}\n`); return { meta, entries, usedCookies, summary }; }