// App-level download filters: pure decisions made from a video's prefetched // metadata, evaluated by the managed downloader BEFORE it commits to a full // download. The metadata/download split (see prefetchMetadata in // downloadOneManaged.ts) exists so these filters have something to look at. // // A filter that returns `skip: true` causes the managed downloader to record a // "skipped-filtered" outcome and move on WITHOUT archiving the video, so the // next sync/download-missing retries it once the filter no longer matches. // // Adding a filter = append one entry to FILTERS. Each filter sees the same // context (metadata + channel config + resolved settings). import type { ChannelConfig, DownloadFilterConfig, RejectedLivestreamMode, } from "./channelConfig"; import type { RawMetadata } from "./transcripts-server"; export type DownloadFilterSettings = { // Global default-on toggle; per-channel config can override it. skipLiveDownloads: boolean; }; export type DownloadFilterContext = { metadata: RawMetadata | null; channelConfig: ChannelConfig; settings: DownloadFilterSettings; // Optional job log. A filter uses it only to report that it made itself // INERT (an unparseable regex on disk) — a silent no-op filter is the one // failure mode nothing downstream can explain. onLog?: (line: string) => void; }; export type DownloadFilterDecision = { skip: boolean; filter: string; reason: string; // The video's MEDIA is skipped and its live chat is wanted — see // DownloadFilterConfig.rejectedLivestreams. `skip` is still true: everything // that decides whether to fetch the media reads that and is unchanged. The // downloader is the one caller that asks the next question. chatOnly?: boolean; }; type DownloadFilter = { name: string; // Returns a skip decision, or null when the filter doesn't apply / passes. evaluate(ctx: DownloadFilterContext): DownloadFilterDecision | null; }; // Resolve the skip-live toggle: per-channel override wins over the global // default. function skipLiveEnabled(ctx: DownloadFilterContext): boolean { return ctx.channelConfig.skipLiveDownloads ?? ctx.settings.skipLiveDownloads; } // Skip videos that are currently live or scheduled/upcoming — they can't be // downloaded as a finished video yet. IMPORTANT: a finished livestream VOD // (was_live === true / live_status === "was_live") is a normal downloadable // video and must NOT be skipped. This deliberately diverges from the // `isLivestream` flag in transcripts-server.ts:summarize(), which folds // was_live in for *display* purposes — do not unify the two. const skipLive: DownloadFilter = { name: "skipLive", evaluate(ctx) { if (!skipLiveEnabled(ctx)) return null; const m = ctx.metadata; if (!m) return null; // fail-open: no metadata → don't skip const status = m.live_status ?? ""; const isLive = m.is_live === true || status === "is_live"; const isUpcoming = status === "is_upcoming"; if (!isLive && !isUpcoming) return null; return { skip: true, filter: "skipLive", reason: isUpcoming ? "video is an upcoming/scheduled livestream" : "video is currently live", }; }, }; // --------------------------------------------------------------------------- // The per-channel title/description filter. // --------------------------------------------------------------------------- export type CompiledDownloadFilter = { include: RegExp | null; exclude: RegExp | null; includeLivestreams: boolean; // See DownloadFilterConfig.rejectedLivestreams. "skip" is the default and is // what every channel written before the field did. rejectedLivestreams: RejectedLivestreamMode; }; // Compile a channel's filter. Returns null when there is no filter AND when a // pattern doesn't parse — an invalid regex makes the filter INERT rather than // failing every download on the channel. The editor form refuses to save one, // so this only fires for a hand-edited config.json. // // `includeLivestreams` alone IS a filter: it is a positive selector, so a // channel that sets only that is filtering (to livestreams). export function compileDownloadFilter( filter: DownloadFilterConfig | undefined | null, ): CompiledDownloadFilter | null { const include = filter?.include?.trim() ?? ""; const exclude = filter?.exclude?.trim() ?? ""; const includeLivestreams = filter?.includeLivestreams === true; if (!include && !exclude && !includeLivestreams) return null; // The mode is NOT a positive selector and deliberately does not make a // channel "filtered" on its own: it says what to do with a rejection, and // with nothing rejecting there is nothing for it to say. const rejectedLivestreams: RejectedLivestreamMode = filter?.rejectedLivestreams === "chat-only" ? "chat-only" : "skip"; try { return { include: include ? new RegExp(include, "i") : null, exclude: exclude ? new RegExp(exclude, "i") : null, includeLivestreams, rejectedLivestreams, }; } catch { return null; } } // What counts as a livestream for `includeLivestreams`. "was_live"/"post_live" // are finished VODs (the normal case); "is_live" and "is_upcoming" are included // so the FILTER accepts a stream that has not finished — skip-live, the next // entry in the registry, still declines to DOWNLOAD it until it ends. Two // different questions, answered by two different filters on purpose. // // "is_upcoming" is here because leaving it out was a trap rather than a // nicety: a scheduled stream scanned before it airs is stored with that status // forever (the scan never re-reads an id it already has), so an // includeLivestreams channel would settle it permanently and never download the // stream it was configured to want. Accepting it costs nothing — skip-live // declines the download as RETRYABLE, which is exactly the right state for a // video that will exist tomorrow. const LIVESTREAM_STATUSES: ReadonlySet = new Set([ "was_live", "post_live", "is_live", "is_upcoming", ]); // The metadata-scan store spells it `liveStatus`; yt-dlp's raw metadata spells // it `live_status`. Both are read here so the scan's prediction and the // download's decision cannot disagree. export type FilterableVideo = { title?: string | null; description?: string | null; liveStatus?: string | null; live_status?: string | null; }; function isLivestream(meta: FilterableVideo): boolean { const status = meta.liveStatus ?? meta.live_status ?? ""; return LIVESTREAM_STATUSES.has(status); } // The only text a download filter ever sees. Kept here so the metadata scan's // stored entries and the download-time prefetch can never be matched against // different strings. // How much description the matcher ever sees. A description can be tens of // kilobytes of links and timestamps, and a backtracking regex's cost grows with // the input — over ~1,800 stored entries, re-run on every snapshot, every sync // page and every render of the Configure stage. Two kilobytes is far more than // a title-matching pattern needs and bounds the worst case. export const DOWNLOAD_FILTER_DESCRIPTION_LIMIT = 2048; export function downloadFilterText( meta: FilterableVideo | null | undefined, ): string { const description = (meta?.description ?? "").slice( 0, DOWNLOAD_FILTER_DESCRIPTION_LIMIT, ); return `${meta?.title ?? ""}\n${description}`; } // Is this pattern safe to hand to the matcher? The operator's regex runs over // every stored title+description on every snapshot, every sync page and every // render of the Configure stage, on the server's single thread — so a // catastrophically backtracking pattern like `^(\w+\s?)*$` does not fail a // request, it hangs the editor. // // Deliberately CRUDE: a length cap and a check for a quantified group that is // itself quantified, which is the shape behind essentially every real // ReDoS. This cannot be exhaustive (deciding it in general is undecidable) and // does not try to be — it is the form's guard, refusing the patterns a person // actually types by accident. Returns null when the pattern is fine. export const DOWNLOAD_FILTER_PATTERN_MAX_LENGTH = 200; export function downloadFilterPatternProblem(pattern: string): string | null { if (pattern.length > DOWNLOAD_FILTER_PATTERN_MAX_LENGTH) { return `must be ${DOWNLOAD_FILTER_PATTERN_MAX_LENGTH} characters or fewer (this one is ${pattern.length})`; } // A group whose contents end in a quantifier, itself quantified: // (a+)+ (a*)* (\w+\s?)* (a{2,}){3,} … // `?` is in the class because `(\w+\s?)*` — "words separated by spaces", // which is what a person means when they type it — is the exact shape that // hangs, and its group ends in `?`. if (/\([^()]*[+*?}][?]?\)\s*[+*{]/.test(pattern)) { return "contains a nested quantifier (a repeated group that itself repeats, e.g. `(\\w+\\s?)*`), which can take exponential time to match"; } return null; } // Why a video passed, or that it didn't. "livestream" exists so the UI can say // how many videos a channel is keeping for a reason other than their name. // // "chat-only" IS A REJECTION with an instruction attached, not a fourth way to // pass. The video is not wanted; its live chat is. Everything that asks "is // this video wanted?" — `titleFilterRejects`, and therefore the settled set and // the download queue — must keep answering yes-it-is-rejected for it, or the // downloader would fetch the media the operator asked it not to. Only the // callers that ask the NEXT question ("and then what?") look for this value. export type DownloadFilterVerdict = | "text" | "livestream" | "rejected" | "chat-only"; // PURE, and the SINGLE matcher. Both the download-time registry entry below and // the derived settled set (controller/metadataScanStore.ts) call exactly this, // which is what makes "what the scan predicted" and "what the download did" the // same computation rather than two that merely look alike. // // The rule, in order: // 1. `exclude` always wins. A match is rejected however else it qualifies. // 2. With NO positive selector (an exclude-only filter), everything else // passes. // 3. Otherwise a video must satisfy at least one positive selector: the // `include` pattern over title + "\n" + description, or — when // `includeLivestreams` is on — being a livestream. // // THE CORNER WORTH STATING: `{ includeLivestreams: true }` with no `include` // rejects plain uploads and passes livestreams. `includeLivestreams` is a // second positive selector, not a modifier of the first, so switching it on // without an `include` is "livestreams only" rather than "everything, plus // livestreams". export function classifyAgainstFilter( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): DownloadFilterVerdict { const text = downloadFilterText(meta); if (compiled.exclude && compiled.exclude.test(text)) { return rejection(compiled, meta); } const hasPositive = Boolean(compiled.include || compiled.includeLivestreams); if (!hasPositive) return "text"; if (compiled.include && compiled.include.test(text)) return "text"; if (compiled.includeLivestreams && isLivestream(meta)) return "livestream"; return rejection(compiled, meta); } // HOW a rejection is spelled — the ONE place, so `exclude`-matched and // unmatched livestreams cannot get different answers. The operator's setting is // about what to do with a rejected livestream, not about which selector did the // rejecting. function rejection( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): DownloadFilterVerdict { return compiled.rejectedLivestreams === "chat-only" && isLivestream(meta) ? "chat-only" : "rejected"; } // IS THIS VIDEO'S MEDIA UNWANTED? True for "chat-only" too — see // DownloadFilterVerdict. This is what `settledByTitleFilterIds` and therefore // `undownloadedIds` are built on, so a chat-only video is settled exactly like // any other rejection and the download queue never offers its media. export function titleFilterRejects( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): boolean { const verdict = classifyAgainstFilter(compiled, meta); return verdict === "rejected" || verdict === "chat-only"; } // Is this one the operator wants the CHAT of? A separate question from the one // above, asked by exactly the two places that act on the answer: the downloader, // which fetches the chat instead of skipping, and the snapshot, which puts the // video in its own bucket rather than in `skippedByTitleFilter`. export function titleFilterWantsChat( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): boolean { return classifyAgainstFilter(compiled, meta) === "chat-only"; } // Why a rejection happened, for the log and the outcome record. export function titleFilterReason( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): string { const text = downloadFilterText(meta); if (compiled.exclude && compiled.exclude.test(text)) { return `title/description matches exclude /${compiled.exclude.source}/i`; } const wants: string[] = []; if (compiled.include) wants.push(`include /${compiled.include.source}/i`); if (compiled.includeLivestreams) wants.push("a livestream"); return `title/description matches neither ${wants.join(" nor ")}`; } // Per-channel include/exclude over title + description. Runs BEFORE skipLive so // a video the operator doesn't want is declined on its own terms rather than // being classified as a live-stream retry. // // FAILS OPEN on missing metadata, like skipLive, and that is a correction. // // It used to fail closed — no metadata, no match, so skip. But a prefetch // returns nothing for BATCH-LEVEL reasons far more often than for per-video // ones: a 429, a bot check, a network drop. Turning those into // `skipped-filtered` was actively harmful. The video never reached the real // attempt, so its failure was never classified (runYtdlp.ts ~937 vs ~944): // no `classifyDownloadFailure`, no `onPlatformBackoff`, no abort-on-error. A // rate-limited channel would walk its entire playlist "filtering" every video // while the source refused every request, and record a cooldown for none of it. // // Returning null costs nothing: the real download attempt runs, fails for the // reason it actually failed, and the existing machinery does its job. The // filter is not bypassed either — the download writes metadata, and the next // pass (or the metadata scan) decides with something to decide on. const titleFilter: DownloadFilter = { name: "titleFilter", evaluate(ctx) { const raw = ctx.channelConfig.downloadFilter; const configured = Boolean( raw?.include?.trim() || raw?.exclude?.trim() || raw?.includeLivestreams, ); if (!configured) return null; const compiled = compileDownloadFilter(raw); if (!compiled) { ctx.onLog?.( `Download filter is INERT: include=${JSON.stringify( raw?.include ?? "", )} exclude=${JSON.stringify( raw?.exclude ?? "", )} is not a valid regex — nothing will be filtered.\n`, ); return null; } const m = ctx.metadata; if (!m) return null; // fail open — see above if (!titleFilterRejects(compiled, m)) return null; const chatOnly = titleFilterWantsChat(compiled, m); return { skip: true, filter: "titleFilter", reason: chatOnly ? `${titleFilterReason(compiled, m)} — fetching its live chat only` : titleFilterReason(compiled, m), ...(chatOnly ? { chatOnly: true } : {}), }; }, }; const FILTERS: ReadonlyArray = [titleFilter, skipLive]; // Evaluate every filter; the first one that says "skip" wins. Returns null when // the video passes all filters. export function evaluateDownloadFilters( ctx: DownloadFilterContext, ): DownloadFilterDecision | null { for (const f of FILTERS) { const decision = f.evaluate(ctx); if (decision?.skip) return decision; } return null; }