// THE METADATA SCAN — one yt-dlp metadata pass over a channel's listed but // never-fetched videos, reading what they are CALLED so the download filter can // decide about them before a single byte of media is fetched. // // It is a catalogued operation (common/lib/operations.ts, METADATA_SCAN_OPERATION) // on the channel's platform download queue, because it contends for exactly the // resource a download does: the source's patience. It writes NO video directory // — see controller/metadataScanStore.ts for the two downstream readers that // would misread one — and no media, which is why its job kind declares // `needsMedia: false`. // // What it costs: one metadata extraction per unscanned listed video, in a single // yt-dlp process reading a batch file, with `--sleep-requests 1`. What it saves: // on a filtered channel, every one of those videos would otherwise be prefetched // by the downloader on every run, forever. import path from "node:path"; import { mkdtemp, readdir, readFile, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import { execa } from "execa"; import { classifyDownloadFailure, isBotCheck, isSoftBlock, parseUnavailableFromStderr, } from "../lib/availability"; import type { ChannelConfig } from "../lib/channelConfig"; import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters"; import { alwaysCookies, authRetryCookies, resolveCookiePolicy, } from "../lib/cookiePolicy"; import { channelExtraArgs, pacingPlatformKey } from "./channelArgs"; import { recordPlatformClean } from "../jobs/downloadBackoff"; import type { Paths } from "../lib/paths"; import { getSettings } from "../lib/settings"; import { extractVideoId } from "../lib/videoId"; import { isVideoFetched, readVideoFiles } from "../lib/videoStatus"; import type { JobProgress } from "../jobs/registry"; import { loadMetadataScan, metadataScanWanted, upsertMetadataScan, type MetadataScanEntry, type MetadataScanError, type MetadataScanRun, } from "../controller/metadataScanStore"; // yt-dlp's JSON-subset print template: one JSON object per entry, so the scan // parses lines instead of whole info jsons — and so nothing is ever written to // disk by yt-dlp itself. // `webpage_url` is in here for a reason that is not cosmetic: yt-dlp's `id` is // the NATIVE extractor id, and every consumer of this store keys by the // CANONICAL id (extractVideoId of the URL — the name data// carries). The // two coincide on YouTube and diverge everywhere else: Rumble's native id is the // embed id (`v2apmfn`) while the canonical id is the URL slug (`v2846lb`); // Twitch's native id drops the leading `v`. Keyed natively, nothing on those // platforms would ever settle and every run would re-fetch the whole listing. const PRINT_TEMPLATE = "%(.{id,title,description,upload_date,live_status,duration,webpage_url})j"; // Flush to the store every N records. A killed or rate-limited job keeps what it // learned; 25 is small enough that a run stopped early has lost almost nothing // and large enough that a 1,800-video scan rewrites the file ~72 times, not // 1,800. const FLUSH_EVERY = 25; // `ERROR: [youtube] dQw4w9WgXcQ: Video unavailable` — yt-dlp names the id it // failed on, which is the only way to attribute a per-video failure in a batch. const ERROR_LINE = /^ERROR:\s*(?:\[[^\]]+\]\s*)?([^\s:]+)\s*:\s*(.*)$/; // THE SOFT BLOCK. Measured 2026-09-20 on a raw yt-dlp pass over one channel's // playlist: after ~100 good records YouTube answered "Video unavailable" for 545 // CONSECUTIVE videos, every one of which fetched fine when probed individually // minutes later. That is throttling wearing availability's clothes, and it is // the most expensive possible failure to believe — it would write 545 `deleted` // errors, each of which then suppresses a re-scan for a day. // // So a run of these is treated as a rate limit, and the streak's ids are // DISCARDED rather than recorded: the next run retries them, which is the whole // point. Ten in a row is the threshold — a channel really can have ten dead // videos in a row, but then the eleventh onwards cost nothing to re-read, while // believing a 545-video soft block costs the channel. const SOFT_BLOCK_STREAK = 10; const UNAVAILABLE_LINE = /video unavailable/i; // The id this record must be STORED under: the canonical id derived from the // video's own URL, falling back to yt-dlp's native id when there is no usable // URL. Pure and exported so the divergence is testable without spawning // anything. export function canonicalScanId(record: { id?: unknown; webpage_url?: unknown; }): string { const url = typeof record.webpage_url === "string" ? record.webpage_url : ""; const fromUrl = url ? extractVideoId(url) : null; if (fromUrl) return fromUrl; return typeof record.id === "string" ? record.id : ""; } // yt-dlp names the NATIVE id on an `ERROR:` line, and there is no URL on it to // canonicalize. Three ways back, cheapest first: a record we already read this // run, a target whose canonical id is literally this string (the YouTube case), // or a target whose URL contains it (the Rumble/Twitch case). Failing all three // the native id is stored as-is — a wrong key for an error row is a re-read next // run, not a lost video. export function resolveScanErrorId( native: string, nativeToCanonical: ReadonlyMap, urlsById: ReadonlyMap, ): string { const known = nativeToCanonical.get(native); if (known) return known; if (urlsById.has(native)) return native; for (const [canonical, url] of urlsById) { if (url.includes(native)) return canonical; } return native; } export type MetadataScanResult = { scanned: number; errors: number; targets: number; stopped?: MetadataScanRun["stopped"]; }; export type RunMetadataScanOpts = { paths: Paths; channelSlug: string; channelConfig: ChannelConfig; onLog: (line: string) => void; signal: AbortSignal; setProgress?: (progress: JobProgress) => void; // Called when the source rate-limits us (HTTP 429, or YouTube's bot check), // so the SHARED per-platform cooldown the auto-download runner honors is // recorded. Same contract as runYtdlp's. onPlatformBackoff?: ( failureClass: "rate_limit" | "network", ) => void | Promise; }; async function readPlaylistIds(channelDir: string): Promise { const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch( () => "", ); const ids: string[] = []; const seen = new Set(); for (const line of raw.split("\n")) { const url = line.trim(); if (!url) continue; const id = extractVideoId(url); if (id && !seen.has(id)) { seen.add(id); ids.push(id); } } return ids; } // Ids that have already been fetched. A downloaded video needs no scan — its // metadata.info.json is the better record, and the snapshot reads it. // // ENOENT IS NOT "NOTHING FETCHED". This used to swallow a readdir failure and // return an empty set, which on a relocated channel whose drive is unmounted // meant the scan happily re-requested the whole channel, downloaded videos // included. The kind declares `needsMedia: true` for exactly this reason — the // target set is DERIVED from data/ — so runManagedFunction refuses the job // before it starts, and anything that slips past that throws here rather than // lying. A channel that has simply never downloaded anything has no data/ dir // at all, which is the one case that legitimately means "nothing fetched". async function fetchedIds(dataDir: string): Promise> { const out = new Set(); let names: string[]; try { names = (await readdir(dataDir, { withFileTypes: true })) .filter((e) => e.isDirectory()) .map((e) => e.name); } catch (err) { if ((err as NodeJS.ErrnoException).code === "ENOENT") return out; throw err; } for (const name of names) { const files = await readVideoFiles(path.join(dataDir, name)); if (isVideoFetched(files)) out.add(name); } return out; } // What this run will fetch: listed, not already fetched, not already scanned. // Deliberately NOT filtered by the download filter — the scan exists to find out // what the filter should say, so it cannot use the answer as its input. export async function metadataScanTargets( paths: Paths, slug: string, now: number = Date.now(), ): Promise { const channelDir = path.join(paths.channelsDir, slug); const [listed, fetched, scan] = await Promise.all([ readPlaylistIds(channelDir), fetchedIds(path.join(channelDir, "data")), loadMetadataScan(paths, slug), ]); // metadataScanWanted, not `!scan.entries[id]`: it also applies the 24 h error // cooldown, which is what the snapshot's advertised backlog // (`metadataScan.unscanned`) uses. Without it the two disagreed — the // operation page would offer a Run for work the run would not do, and a // members-only video would be re-requested on every single scan. return listed.filter( (id) => !fetched.has(id) && metadataScanWanted(scan, id, now), ); } function urlForId(id: string, channelConfig: ChannelConfig): string { // The playlist file holds real URLs; rebuilding one from an id is only safe // for YouTube. So callers pass URLs through — see runMetadataScan, which reads // the playlist file again for exactly this reason. void channelConfig; return `https://www.youtube.com/watch?v=${id}`; } async function readPlaylistUrlsById( channelDir: string, ): Promise> { const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch( () => "", ); const byId = new Map(); for (const line of raw.split("\n")) { const url = line.trim(); if (!url) continue; const id = extractVideoId(url); if (id && !byId.has(id)) byId.set(id, url); } return byId; } export async function runMetadataScan( opts: RunMetadataScanOpts, ): Promise { const { paths, channelSlug: slug, channelConfig } = opts; const channelDir = path.join(paths.channelsDir, slug); const startedAt = new Date().toISOString(); const targetIds = await metadataScanTargets(paths, slug); const urlsById = await readPlaylistUrlsById(channelDir); if (targetIds.length === 0) { opts.onLog( "Nothing to scan: every listed video is already downloaded or already in the metadata scan.\n", ); await upsertMetadataScan( paths, slug, { lastRun: { startedAt, finishedAt: new Date().toISOString(), scanned: 0, errors: 0, }, }, new Date().toISOString(), ); return { scanned: 0, errors: 0, targets: 0 }; } opts.onLog( `Scanning metadata for ${targetIds.length} listed video(s). No media is downloaded and no video directory is created.\n`, ); opts.setProgress?.({ metric: "scans", initial: 0, target: targetIds.length, current: 0, }); const policy = resolveCookiePolicy(getSettings(), channelConfig); const tmpRoot = await mkdtemp(path.join(tmpdir(), "ttb-mdscan-")); const entries: Record = {}; const errors: Record = {}; let scannedTotal = 0; let pending = 0; let stopped: MetadataScanRun["stopped"] | undefined; let stoppedMessage: string | undefined; // WHY THE CURRENT PASS STOPPED, before anyone decides what it MEANS. // // A bot check and a soft block both look like "the source is refusing us", // and the first response to that is not a cooldown — it is to try the thing // we have not tried. Measured 2026-09-20 on the first real scan of a channel: // the FIRST video answered "Sign in to confirm you're not a bot", the run // stopped after 3 s with a rate-limit cooldown, and the operator had a // configured `cookiesFromBrowser` the scan never used (it passes cookies only // in "always" mode). An hour of cooldown for a problem one retry solves. // // So a block on a COOKIE-LESS pass is a reason to retry with cookies, once. // Only a block on a pass that already carried them is a real refusal. let block: { kind: "bot-check" | "soft-block" | "rate-limit"; message: string } | null = null; const needsAuthIds: string[] = []; // Native extractor id -> canonical id, learned from the records this run // reads, so an `ERROR:` line naming a native id can be attributed correctly. const nativeToCanonical = new Map(); // "Video unavailable" errors, HELD rather than recorded until the pass ends — // so a soft block's ids are never written to the store at all, rather than // written and then deleted. See SOFT_BLOCK_STREAK. // // KEYED BY ID, NOT A LIST, because "consecutive" has to mean consecutive IN // THE BATCH and not in the order the parent happened to read two pipes. // stdout and stderr arrive independently: a record read late could break a // streak that never actually contained it, which made a real soft block look // like ten separate dead videos (and the e2e flaky under load). The verdict is // computed from the batch order after the pass, where the order is a fact. let unavailableById = new Map(); // Longest run of consecutive unavailable ids in BATCH order, over the ids this // pass was asked for. Nothing else decides whether this was a block. const longestUnavailableRun = (ids: ReadonlyArray): number => { let best = 0; let run = 0; for (const id of ids) { if (unavailableById.has(id)) { run++; if (run > best) best = run; } else if (entries[id] || errors[id]) { // A video we DID read, or that failed for another reason, genuinely // breaks the run. run = 0; } // An id with no outcome at all was never reached (the pass was killed); // it neither extends nor breaks the run. } return best; }; // These really were unavailable videos, so record them. const commitUnavailable = () => { for (const [id, message] of unavailableById) { const err: MetadataScanError = { class: "deleted", message, at: new Date().toISOString(), }; errors[id] = err; unflushedErrors[id] = err; pending++; } unavailableById = new Map(); }; // WHAT HAS NOT REACHED DISK YET. `entries`/`errors` above are the run's full // accumulator (the orchestration reads them to decide what is left to fetch); // these two are the DELTA. Sending the whole map on every flush made a scan // O(N²) in bytes written — a 1,800-video channel's 72 flushes would each // re-send everything read so far. const unflushedEntries: Record = {}; const unflushedErrors: Record = {}; // ONE WRITER AT A TIME. Flushes used to be fired mid-stream and collected in // an array, so two could overlap — and `writeMetadataScan` wrote to a tmp path // named only for the pid, so the second rename found the file already moved // and the job died with ENOENT. `pending` is also zeroed BEFORE the await, so // records arriving during a write are counted toward the next flush instead of // being re-sent by this one. let flushChain: Promise = Promise.resolve(); const flush = (force = false): Promise => { if (pending === 0) return flushChain; if (!force && pending < FLUSH_EVERY) return flushChain; const batchEntries = { ...unflushedEntries }; const batchErrors = { ...unflushedErrors }; for (const k of Object.keys(unflushedEntries)) delete unflushedEntries[k]; for (const k of Object.keys(unflushedErrors)) delete unflushedErrors[k]; pending = 0; flushChain = flushChain.then(async () => { await upsertMetadataScan( paths, slug, { entries: batchEntries, errors: batchErrors }, new Date().toISOString(), ); }); return flushChain; }; // One pass over a batch file. Returns the ids it could not read as needs_auth, // so the caller can decide whether a cookie retry is worth a second pass. const runPass = async ( ids: ReadonlyArray, cookies: string | undefined, label: string, ): Promise => { const batchFile = path.join(tmpRoot, `${label}.batch`); await writeFile( batchFile, ids.map((id) => urlsById.get(id) ?? urlForId(id, channelConfig)).join("\n") + "\n", ); const args = [ "--ignore-config", "--skip-download", "--no-warnings", "--ignore-errors", // One request per second. The scan touches every listed video of a // channel in one process; without this it is the most rate-limitable // thing the app does. "--sleep-requests", "1", "--print", PRINT_TEMPLATE, "-a", batchFile, ...channelExtraArgs(channelConfig, cookies), ]; opts.onLog(`$ ${paths.ytdlpBin} ${args.join(" ")}\n`); const child = execa(paths.ytdlpBin, args, { cwd: channelDir, cancelSignal: opts.signal, all: false, buffer: false, reject: false, }); const takeRecord = (line: string) => { const trimmed = line.trim(); if (!trimmed.startsWith("{")) return; let parsed: Record; try { parsed = JSON.parse(trimmed) as Record; } catch { return; } const id = canonicalScanId(parsed); if (!id) return; const native = typeof parsed.id === "string" ? parsed.id : ""; if (native && native !== id) nativeToCanonical.set(native, id); const entry: MetadataScanEntry = { title: typeof parsed.title === "string" ? parsed.title : "", description: typeof parsed.description === "string" ? parsed.description : "", uploadDate: typeof parsed.upload_date === "string" ? parsed.upload_date : "", ...(typeof parsed.live_status === "string" ? { liveStatus: parsed.live_status } : {}), ...(typeof parsed.duration === "number" ? { duration: parsed.duration } : {}), scannedAt: new Date().toISOString(), }; entries[id] = entry; unflushedEntries[id] = entry; scannedTotal++; pending++; opts.setProgress?.({ metric: "scans", initial: 0, target: targetIds.length, current: scannedTotal, }); }; let stdoutBuf = ""; child.stdout?.on("data", (c: Buffer) => { stdoutBuf += c.toString("utf8"); let nl: number; while ((nl = stdoutBuf.indexOf("\n")) !== -1) { takeRecord(stdoutBuf.slice(0, nl)); stdoutBuf = stdoutBuf.slice(nl + 1); } // Serialized on flushChain, so this cannot overlap the previous write. if (pending >= FLUSH_EVERY) void flush(); }); let stderrBuf = ""; // See the comment in takeError: a stop signal, not a verdict. let unavailableRun = 0; const takeError = (line: string) => { // Once this pass is blocked, the rest of the stream is that block // talking. Nothing more is recorded from it. if (block) return; const trimmed = line.trim(); if (!trimmed.startsWith("ERROR")) return; const cls = parseUnavailableFromStderr(trimmed); const failure = classifyDownloadFailure(trimmed, cls); // A batch-level signal: continuing would just hammer the source, and on a // bot check every subsequent entry fails the same way. Stop the pass and // let the caller decide whether cookies are the answer. YouTube's // explicit soft block ("…isn't available, try again later") lands here // too since release 10 — before, it was recorded as a `deleted` error on // its id and the pass kept going into it. if (failure === "rate_limit") { block = { kind: isBotCheck(trimmed) ? "bot-check" : isSoftBlock(trimmed) ? "soft-block" : "rate-limit", message: trimmed.slice(0, 300), }; child.kill("SIGTERM"); return; } const m = ERROR_LINE.exec(trimmed); const native = m?.[1] ?? ""; const id = native ? resolveScanErrorId(native, nativeToCanonical, urlsById) : ""; const message = (m?.[2] || trimmed).trim().slice(0, 300); if (!id) return; // "Video unavailable", possibly the soft block. Held, not recorded. if (UNAVAILABLE_LINE.test(message)) { unavailableById.set(id, message); // A LIVE COUNTER, used only to STOP FETCHING — not to decide anything. // stderr is ordered within itself, so this is a good enough trigger to // stop hammering a source that is refusing us; the verdict is computed // from batch order once the pass is over. unavailableRun++; if (unavailableRun >= SOFT_BLOCK_STREAK) child.kill("SIGTERM"); return; } // Any other error breaks the run — a members-only video is the case this // protects: it errors every time, it is genuinely not fetchable, and it // must not be thrown away with a soft block's ids. unavailableRun = 0; if (cls === "needs_auth") needsAuthIds.push(id); const err: MetadataScanError = { class: cls, message, at: new Date().toISOString(), }; errors[id] = err; unflushedErrors[id] = err; pending++; }; child.stderr?.on("data", (c: Buffer) => { const chunk = c.toString("utf8"); opts.onLog(chunk); stderrBuf += chunk; let nl: number; while ((nl = stderrBuf.indexOf("\n")) !== -1) { takeError(stderrBuf.slice(0, nl)); stderrBuf = stderrBuf.slice(nl + 1); } }); const result = await child; if (stdoutBuf) takeRecord(stdoutBuf); if (stderrBuf) takeError(stderrBuf); // THE VERDICT, from the batch order. A run of consecutive unavailable ids // long enough to be a block is discarded so the next run re-reads them; // anything shorter is what it looks like — videos that really are gone. if (!block) { const run = longestUnavailableRun(ids); if (run >= SOFT_BLOCK_STREAK) { block = { kind: "soft-block", message: `${run} consecutive "Video unavailable" errors — ` + `treating this as a soft block rather than ${run} deleted videos. ` + `Their ids were NOT recorded, so the next run re-reads them.`, }; } } if (block) unavailableById = new Map(); else commitUnavailable(); await flushChain; await flush(true); if (opts.signal.aborted) { stopped = "aborted"; return; } if (block) return; // --ignore-errors means a per-video failure still exits nonzero while every // readable entry was printed, so a nonzero exit is only fatal when nothing // came back at all. if ( result.exitCode !== 0 && result.exitCode !== 101 && scannedTotal === 0 && Object.keys(errors).length === 0 ) { stopped = "error"; // `exited with code undefined` is what this said when yt-dlp could not be // SPAWNED at all — a missing binary, the single most likely cause — which // told the operator nothing. execa knows the difference. stoppedMessage = result.failed ? result.shortMessage : `yt-dlp exited with code ${result.exitCode}`; throw new Error(stoppedMessage); } }; // Cookies for the retry pass, when the mode offers any. Defer mode resolves // to none, which is the point of defer: those videos wait for a deliberate // cookie run rather than being retried behind the operator's back. const retryCookies = authRetryCookies(policy); let usedCookies = false; try { const firstCookies = alwaysCookies(policy); usedCookies = Boolean(firstCookies); await runPass(targetIds, firstCookies, "main"); // THE COOKIE RETRY. A block on a cookie-LESS pass is not yet a refusal — // for a bot check it is literally "you are not signed in", and a soft block // behaves the same way on an unauthenticated session. Retry the ids we have // not read yet, once, with the cookies the operator already configured. // Only a block that survives cookies is treated as the source saying no. if ( block && !usedCookies && retryCookies !== undefined && !opts.signal.aborted ) { const remaining = targetIds.filter((id) => !entries[id]); opts.onLog( `${(block as { kind: string }).kind === "bot-check" ? "bot check" : "soft block"} without cookies — ` + `retrying the remaining ${remaining.length} ids with --cookies-from-browser ${retryCookies}.\n`, ); block = null; unavailableById = new Map(); usedCookies = true; if (remaining.length > 0) { await runPass(remaining, retryCookies, "cookie-retry"); } } // ONE cookie retry for the auth-gated ids. Skipped when the pass above // already carried cookies — it covered these ids too. const retryIds = [...new Set(needsAuthIds)].filter((id) => !entries[id]); if ( !block && stopped === undefined && !usedCookies && retryIds.length > 0 && retryCookies !== undefined && !opts.signal.aborted ) { opts.onLog( `${retryIds.length} video(s) need auth; retrying once with --cookies-from-browser ${retryCookies}.\n`, ); await runPass(retryIds, retryCookies, "auth-retry"); } } finally { await flush(true); await rm(tmpRoot, { recursive: true, force: true }); // THE RUN RECORD IS WRITTEN HERE, not after the happy path. A fatal throw // used to escape before it, so `stopped: "error"` was never persisted: the // job failed, the store still said the last run finished cleanly, and the // UI had nothing to show for it. Best-effort — a store this cannot write is // not a reason to replace the real error with a second one. try { await upsertMetadataScan( paths, slug, { lastRun: { startedAt, finishedAt: new Date().toISOString(), scanned: scannedTotal, errors: Object.keys(errors).length, ...(stopped ? { stopped } : {}), ...(stoppedMessage ? { message: stoppedMessage } : {}), }, }, new Date().toISOString(), ); } catch { /* the thrown error, if any, is the one worth reporting */ } } // A block that survived (or never had) a cookie retry is the source refusing // us. Only NOW does it become a cooldown. if (block && stopped === undefined) { stopped = "rate_limit"; stoppedMessage = (block as { message: string }).message; } if (stopped === "rate_limit") { opts.onLog( `STOPPED: the source is rate-limiting this scan. ${stoppedMessage ?? ""}\n` + `${scannedTotal} video(s) were read and saved; the rest stay unscanned. A per-platform cooldown has been recorded — run the scan again once it lapses.\n`, ); await opts.onPlatformBackoff?.("rate_limit"); } else if (stopped === undefined && scannedTotal > 0 && !opts.signal.aborted) { // A scan the source answered cleanly settles the platform like a clean // probe — its backoff and any hold clear (release 17 review H2). try { const line = await recordPlatformClean(pacingPlatformKey(channelConfig), paths); if (line) opts.onLog(line); } catch { /* shared-state write is best-effort */ } } const errorCount = Object.keys(errors).length; // Rewritten because `stopped` may only have been DECIDED above (a block that // survived its cookie retry becomes a stop after the finally has run). The // finally's write is the one that survives a throw; this is the one that // carries the verdict. await upsertMetadataScan( paths, slug, { lastRun: { startedAt, finishedAt: new Date().toISOString(), scanned: scannedTotal, errors: errorCount, ...(stopped ? { stopped } : {}), ...(stoppedMessage ? { message: stoppedMessage } : {}), }, }, new Date().toISOString(), ); // The summary the operator actually wants: not "how many did you read" but // "what does my filter now say about them". const compiled = compileDownloadFilter(channelConfig.downloadFilter); if (compiled) { const scan = await loadMetadataScan(paths, slug); let matched = 0; let settled = 0; for (const entry of Object.values(scan.entries)) { if (titleFilterRejects(compiled, entry)) settled++; else matched++; } opts.onLog( `Metadata scan complete: ${scannedTotal} scanned, ${errorCount} error(s). Download filter: ${matched} match, ${settled} filtered out.\n`, ); } else { opts.onLog( `Metadata scan complete: ${scannedTotal} scanned, ${errorCount} error(s). No download filter configured on this channel.\n`, ); } return { scanned: scannedTotal, errors: errorCount, targets: targetIds.length, ...(stopped ? { stopped } : {}), }; }