import { test } from "node:test"; import assert from "node:assert/strict"; import type { ChannelConfig } from "./channelConfig"; import { DOWNLOAD_FILTER_DESCRIPTION_LIMIT, classifyAgainstFilter, compileDownloadFilter, downloadFilterPatternProblem, downloadFilterText, evaluateDownloadFilters, titleFilterRejects, titleFilterWantsChat, type DownloadFilterContext, } from "./downloadFilters"; import type { RawMetadata } from "./transcripts-server"; // Run with: node_modules/.bin/tsx --test common/lib/downloadFilters.test.ts const BASE_CONFIG: ChannelConfig = { handling: "youtube", name: "Test" }; function meta(over: Partial = {}): RawMetadata { return { id: "vid", title: "Synthetic vid", description: "Synthetic video vid", is_live: false, was_live: false, live_status: "not_live", ...over, }; } function ctx(over: Partial = {}): DownloadFilterContext { return { metadata: meta(), channelConfig: BASE_CONFIG, // Global skip-live default is ON, matching downloadOneManaged's fallback. settings: { skipLiveDownloads: true }, ...over, }; } function withFilter(include?: string, exclude?: string): ChannelConfig { return { ...BASE_CONFIG, downloadFilter: { ...(include ? { include } : {}), ...(exclude ? { exclude } : {}), }, }; } // The compiled filter for a pair of patterns, asserted non-null. function compiled(include?: string, exclude?: string) { const c = compileDownloadFilter(withFilter(include, exclude).downloadFilter); assert.ok(c); return c; } test("no filter configured -> nothing compiles, nothing is decided", () => { assert.equal(compileDownloadFilter(undefined), null); assert.equal(compileDownloadFilter({}), null); assert.equal(compileDownloadFilter({ include: " " }), null); assert.equal(evaluateDownloadFilters(ctx()), null); }); test("invalid regex compiles to null (inert), it never throws", () => { assert.equal(compileDownloadFilter({ include: "elf(" }), null); assert.equal(compileDownloadFilter({ exclude: "a[" }), null); }); test("titleFilterRejects: include hit, include miss, description-only hit", () => { const c = compiled("guest"); assert.equal( titleFilterRejects(c, { title: "Synthetic guestvid0001", description: "" }), false, ); assert.equal( titleFilterRejects(c, { title: "Synthetic plainvid0001", description: "" }), true, ); // The description alone can satisfy include — it is half the matched text. assert.equal( titleFilterRejects(c, { title: "Episode 12", description: "With a special guest this week", }), false, ); }); test("titleFilterRejects: exclude wins over include", () => { const c = compiled("guest", "rerun"); assert.equal( titleFilterRejects(c, { title: "guest episode (rerun)", description: "" }), true, ); assert.equal( titleFilterRejects(c, { title: "guest episode", description: "" }), false, ); }); test("matching is case-insensitive in both directions", () => { assert.equal( titleFilterRejects(compiled("GUEST"), { title: "a guest appears", description: "", }), false, ); assert.equal( titleFilterRejects(compiled(undefined, "rerun"), { title: "A RERUN", description: "", }), true, ); }); test("the matched text is title + newline + description, missing fields empty", () => { assert.equal(downloadFilterText({ title: "a", description: "b" }), "a\nb"); assert.equal(downloadFilterText({ title: "a" }), "a\n"); assert.equal(downloadFilterText(null), "\n"); // A pattern must not straddle the boundary by accident — the separator is a // newline, so /title description/ does NOT match. assert.equal( titleFilterRejects(compiled("title description"), { title: "title", description: "description", }), true, ); }); test("registry: include miss -> a skip naming the reason, with no verdict", () => { const d = evaluateDownloadFilters( ctx({ channelConfig: withFilter("guest"), metadata: meta({ title: "Synthetic plainvid0001", description: "x" }), }), ); assert.ok(d); assert.equal(d.skip, true); assert.equal(d.filter, "titleFilter"); assert.match(d.reason, /matches neither include \/guest\/i/); // Whether this video is SETTLED is derived from the metadata-scan store and // the channel's current patterns — never carried on the decision. assert.deepEqual(Object.keys(d).sort(), ["filter", "reason", "skip"]); }); test("registry: include hit downloads", () => { assert.equal( evaluateDownloadFilters( ctx({ channelConfig: withFilter("guest"), metadata: meta({ title: "Synthetic guestvid0001", description: "x" }), }), ), null, ); }); test("registry: invalid regex -> inert filter, exactly one log line", () => { const lines: string[] = []; const d = evaluateDownloadFilters( ctx({ channelConfig: withFilter("elf("), metadata: meta({ title: "anything at all", description: "" }), onLog: (l) => lines.push(l), }), ); assert.equal(d, null); assert.equal(lines.length, 1); assert.match(lines[0], /INERT/); }); test("registry: null metadata is NOT a filter decision (fail open)", () => { // A prefetch returns nothing for batch-level reasons — 429, bot check, // network — far more often than per-video ones. Calling that a filter skip // stopped the video from ever reaching the real attempt, so its failure was // never classified and no per-platform backoff was ever recorded. assert.equal( evaluateDownloadFilters( ctx({ channelConfig: withFilter("guest"), metadata: null }), ), null, ); // Including when the filter would have rejected it on its title: we do not // have the title, so there is nothing to reject it on. assert.equal( evaluateDownloadFilters( ctx({ channelConfig: { ...BASE_CONFIG, downloadFilter: { include: "guest", includeLivestreams: true }, }, metadata: null, }), ), null, ); }); test("registry: null metadata with NO filter -> no skip (fail-open)", () => { assert.equal(evaluateDownloadFilters(ctx({ metadata: null })), null); }); test("skipLive still fires, and titleFilter runs first", () => { // A live video with no title filter: skip-live decides, as before. const live = evaluateDownloadFilters( ctx({ metadata: meta({ is_live: true, live_status: "is_live" }) }), ); assert.ok(live); assert.equal(live.filter, "skipLive"); // A live video that ALSO misses the include: the title filter answers, // because it is first in the registry — the operator's "not this one" is a // better description of the video than "it is live". const both = evaluateDownloadFilters( ctx({ channelConfig: withFilter("guest"), metadata: meta({ title: "Synthetic plainvid0001", description: "", is_live: true, live_status: "is_live", }), }), ); assert.ok(both); assert.equal(both.filter, "titleFilter"); // A live video that PASSES the include still falls through to skip-live. const passThrough = evaluateDownloadFilters( ctx({ channelConfig: withFilter("guest"), metadata: meta({ title: "Synthetic guestvid0001", description: "", is_live: true, live_status: "is_live", }), }), ); assert.ok(passThrough); assert.equal(passThrough.filter, "skipLive"); // Finished VOD is untouched by either (regression guard). assert.equal( evaluateDownloadFilters( ctx({ metadata: meta({ was_live: true, live_status: "was_live" }) }), ), null, ); }); // --- includeLivestreams ------------------------------------------------------ function liveMeta(status: string, title = "Synthetic plainvid0001") { return { title, description: "", liveStatus: status }; } test("includeLivestreams is a SECOND positive selector, not a modifier", () => { // The corner: with no `include`, plain uploads are rejected and livestreams // pass. "livestreams only", not "everything plus livestreams". const only = compileDownloadFilter({ includeLivestreams: true }); assert.ok(only); assert.equal(classifyAgainstFilter(only, liveMeta("not_live")), "rejected"); assert.equal(classifyAgainstFilter(only, liveMeta("was_live")), "livestream"); // Alongside an include, either selector is enough. const both = compileDownloadFilter({ include: "guest", includeLivestreams: true, }); assert.ok(both); assert.equal( classifyAgainstFilter(both, liveMeta("not_live", "a guest appears")), "text", ); assert.equal(classifyAgainstFilter(both, liveMeta("was_live")), "livestream"); assert.equal(classifyAgainstFilter(both, liveMeta("not_live")), "rejected"); }); test("every live status that counts, and one that does not", () => { const f = compileDownloadFilter({ includeLivestreams: true }); assert.ok(f); for (const status of ["was_live", "post_live", "is_live"]) { assert.equal(classifyAgainstFilter(f, liveMeta(status)), "livestream", status); } // A SCHEDULED stream counts. Leaving it out settled it forever: the scan // never re-reads an id it already has, so the stored "is_upcoming" would // outlive the broadcast and the channel would never download the stream it // was configured to want. skip-live still declines the download, retryably. assert.equal(classifyAgainstFilter(f, liveMeta("is_upcoming")), "livestream"); assert.equal(classifyAgainstFilter(f, liveMeta("")), "rejected"); assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected"); // yt-dlp's raw spelling is read as well as the store's. assert.equal( classifyAgainstFilter(f, { title: "x", description: "", live_status: "was_live", }), "livestream", ); }); test("exclude still wins over a livestream", () => { const f = compileDownloadFilter({ includeLivestreams: true, exclude: "rerun", }); assert.ok(f); assert.equal( classifyAgainstFilter(f, liveMeta("was_live", "a rerun stream")), "rejected", ); }); test("an exclude-only filter still passes everything it does not match", () => { const f = compileDownloadFilter({ exclude: "rerun" }); assert.ok(f); assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "text"); assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text"); }); test("includeLivestreams alone makes the channel filtered at all", () => { // compileDownloadFilter returning null is what "no filter" means everywhere // downstream, so this flag has to be enough on its own. assert.notEqual(compileDownloadFilter({ includeLivestreams: true }), null); assert.equal(compileDownloadFilter({ includeLivestreams: false }), null); }); test("the registry names both selectors when it declines", () => { const d = evaluateDownloadFilters( ctx({ channelConfig: { ...BASE_CONFIG, downloadFilter: { include: "guest", includeLivestreams: true }, }, metadata: meta({ title: "Synthetic plainvid0001", live_status: "not_live" }), }), ); assert.ok(d); assert.match(d.reason, /neither include \/guest\/i nor a livestream/); }); test("an upcoming stream is accepted by the filter and still deferred by skip-live", () => { // The two filters answering the two questions, in the order the registry runs // them: the title filter says "yes, this channel wants livestreams", and // skip-live says "not yet — it has not aired". const d = evaluateDownloadFilters( ctx({ channelConfig: { ...BASE_CONFIG, downloadFilter: { includeLivestreams: true }, }, metadata: meta({ title: "Tomorrow's stream", description: "", live_status: "is_upcoming", }), }), ); assert.ok(d); assert.equal(d.filter, "skipLive"); assert.match(d.reason, /upcoming/); }); // --- the operator's regex is server-side code ------------------------------- test("a catastrophically backtracking pattern is refused", () => { // The canonical ReDoS shape, and one a person types by accident when they // mean "words separated by spaces". assert.match( downloadFilterPatternProblem("^(\\w+\\s?)*$") ?? "", /nested quantifier/, ); assert.match(downloadFilterPatternProblem("(a+)+") ?? "", /nested quantifier/); assert.match(downloadFilterPatternProblem("(a*)*b") ?? "", /nested quantifier/); assert.match( downloadFilterPatternProblem("x".repeat(201)) ?? "", /200 characters or fewer/, ); }); test("ordinary patterns are not refused", () => { for (const ok of [ "guest", "guest|special", "^Ep\\.? ?\\d+", "(guest|host) episode", "\\bQ&A\\b", "a{2,4}", ]) { assert.equal(downloadFilterPatternProblem(ok), null, ok); } }); test("the matched description is capped", () => { const long = "x".repeat(DOWNLOAD_FILTER_DESCRIPTION_LIMIT + 500) + "needle"; const text = downloadFilterText({ title: "t", description: long }); assert.equal(text.length, 1 + 1 + DOWNLOAD_FILTER_DESCRIPTION_LIMIT); // The cap is what bounds the matcher's worst case, so something past it is // genuinely not matched — that is the trade, and it is stated here. assert.equal( titleFilterRejects(compiled("needle"), { title: "t", description: long }), true, ); }); // --- rejectedLivestreams: the chat-only tier -------------------------------- // // The whole design risk of this feature is one sentence: "chat-only" is a // REJECTION with an instruction attached, not a fourth way to pass. Everything // that asks "is this video's media wanted?" must keep answering no for it, or // the downloader fetches the very media the operator said not to. test("chat-only is a rejection, and titleFilterRejects still says so", () => { const f = compileDownloadFilter({ include: "guest", rejectedLivestreams: "chat-only", }); assert.ok(f); const stream = liveMeta("was_live"); assert.equal(classifyAgainstFilter(f, stream), "chat-only"); // THE LOAD-BEARING ONE. settledIdsFrom and therefore undownloadedIds are // built on this: a chat-only video is settled exactly like any other // rejection, so the download queue never offers its media. assert.equal(titleFilterRejects(f, stream), true); assert.equal(titleFilterWantsChat(f, stream), true); }); test("it applies to livestreams ONLY, and only when configured", () => { const f = compileDownloadFilter({ include: "guest", rejectedLivestreams: "chat-only", }); assert.ok(f); // A rejected plain upload is a plain rejection: there is no chat to keep. assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected"); assert.equal(titleFilterWantsChat(f, liveMeta("not_live")), false); // A video the filter WANTS is unaffected either way. assert.equal( classifyAgainstFilter(f, liveMeta("was_live", "a guest appears")), "text", ); // The default, and every channel written before the field: absent = skip. const plain = compileDownloadFilter({ include: "guest" }); assert.ok(plain); assert.equal(plain.rejectedLivestreams, "skip"); assert.equal(classifyAgainstFilter(plain, liveMeta("was_live")), "rejected"); assert.equal(titleFilterWantsChat(plain, liveMeta("was_live")), false); }); test("an EXCLUDE-matched livestream is chat-only too, because the mode is about the rejection", () => { // The setting says what to do with a rejected livestream, not which selector // did the rejecting. One rule, in one place, so the two cannot diverge. const f = compileDownloadFilter({ exclude: "rerun", rejectedLivestreams: "chat-only", }); assert.ok(f); assert.equal( classifyAgainstFilter(f, liveMeta("was_live", "rerun of last night")), "chat-only", ); // Exclude-only is still "everything else passes" — the mode adds no filtering. assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text"); }); test("the mode alone is not a filter, and is never the reason a channel is filtered", () => { // Nothing rejecting means nothing for the mode to say, so it does not make a // channel filtered — which is also what keeps the download lane from offering // a metadata scan to a channel that has no filter at all. assert.equal(compileDownloadFilter({ rejectedLivestreams: "chat-only" }), null); }); test("evaluateDownloadFilters marks the decision, and it is still a skip", () => { const decision = evaluateDownloadFilters( ctx({ metadata: meta({ title: "Synthetic plainvid0001", live_status: "was_live" }), channelConfig: { ...BASE_CONFIG, downloadFilter: { include: "guest", rejectedLivestreams: "chat-only" }, }, }), ); assert.ok(decision); // `skip` is what every media-fetching path reads, and it is unchanged: the // downloader is the one caller that asks the next question. assert.equal(decision.skip, true); assert.equal(decision.filter, "titleFilter"); assert.equal(decision.chatOnly, true); assert.match(decision.reason, /live chat only/); });