commit eda6f556ba363f626d175ab7fd35613073ae4fd2
parent a8ebb54e143402780b170e7961f2f651c823e189
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 01:45:17 -0400
download filter: a channel can opt every livestream in
`downloadFilter.includeLivestreams` is a SECOND positive selector beside
`include`, not a modifier of it — which is the whole of the surprising corner:
switching it on with no include pattern means "livestreams only", because a
positive selector now exists and plain uploads satisfy neither. `exclude` still
wins over both. The rule lives in one function, `classifyAgainstFilter`, which
is also what the scan's prediction and the downloader's decision both call.
`is_live` counts as a livestream here on purpose: whether the FILTER wants a
stream and whether skip-live will download one that has not ended are different
questions, and they stay answered by two registry entries.
The Playlist stage splits the matched count into "by title/description" and
"livestreams". Without the split, turning the flag on is indistinguishable from
widening the regex — and the number it moves is the number of videos about to
be downloaded.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
13 files changed, 334 insertions(+), 40 deletions(-)
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -62,6 +62,11 @@ export type AudioCheckConfig = {
export type DownloadFilterConfig = {
include?: string;
exclude?: string;
+ // Opt every livestream VOD in, regardless of what it is called. A SECOND
+ // positive selector beside `include`, not a modifier of it — which is why
+ // `{ includeLivestreams: true }` alone rejects plain uploads and passes
+ // livestreams. See titleFilterRejects.
+ includeLivestreams?: boolean;
};
export type ChannelConfig = {
@@ -359,10 +364,12 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null {
const df = r.downloadFilter as Record<string, unknown>;
const include = typeof df.include === "string" ? df.include.trim() : "";
const exclude = typeof df.exclude === "string" ? df.exclude.trim() : "";
- if (include || exclude) {
+ const includeLivestreams = df.includeLivestreams === true;
+ if (include || exclude || includeLivestreams) {
config.downloadFilter = {
...(include ? { include } : {}),
...(exclude ? { exclude } : {}),
+ ...(includeLivestreams ? { includeLivestreams: true } : {}),
};
}
}
diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts
@@ -2,6 +2,7 @@ import { test } from "node:test";
import assert from "node:assert/strict";
import type { ChannelConfig } from "./channelConfig";
import {
+ classifyAgainstFilter,
compileDownloadFilter,
downloadFilterText,
evaluateDownloadFilters,
@@ -139,7 +140,7 @@ test("registry: include miss -> a skip naming the reason, with no verdict", () =
assert.ok(d);
assert.equal(d.skip, true);
assert.equal(d.filter, "titleFilter");
- assert.match(d.reason, /does not match include \/guest\/i/);
+ assert.match(d.reason, /matches neither include \/guest\/i/);
// Whether this video is SETTLED is derived from the metadata-scan store and
// the channel's current patterns — never carried on the decision.
assert.deepEqual(Object.keys(d).sort(), ["filter", "reason", "skip"]);
@@ -233,3 +234,90 @@ test("skipLive still fires, and titleFilter runs first", () => {
null,
);
});
+
+// --- includeLivestreams ------------------------------------------------------
+
+function liveMeta(status: string, title = "Synthetic plainvid0001") {
+ return { title, description: "", liveStatus: status };
+}
+
+test("includeLivestreams is a SECOND positive selector, not a modifier", () => {
+ // The corner: with no `include`, plain uploads are rejected and livestreams
+ // pass. "livestreams only", not "everything plus livestreams".
+ const only = compileDownloadFilter({ includeLivestreams: true });
+ assert.ok(only);
+ assert.equal(classifyAgainstFilter(only, liveMeta("not_live")), "rejected");
+ assert.equal(classifyAgainstFilter(only, liveMeta("was_live")), "livestream");
+
+ // Alongside an include, either selector is enough.
+ const both = compileDownloadFilter({
+ include: "guest",
+ includeLivestreams: true,
+ });
+ assert.ok(both);
+ assert.equal(
+ classifyAgainstFilter(both, liveMeta("not_live", "a guest appears")),
+ "text",
+ );
+ assert.equal(classifyAgainstFilter(both, liveMeta("was_live")), "livestream");
+ assert.equal(classifyAgainstFilter(both, liveMeta("not_live")), "rejected");
+});
+
+test("every live status that counts, and one that does not", () => {
+ const f = compileDownloadFilter({ includeLivestreams: true });
+ assert.ok(f);
+ for (const status of ["was_live", "post_live", "is_live"]) {
+ assert.equal(classifyAgainstFilter(f, liveMeta(status)), "livestream", status);
+ }
+ assert.equal(classifyAgainstFilter(f, liveMeta("is_upcoming")), "rejected");
+ assert.equal(classifyAgainstFilter(f, liveMeta("")), "rejected");
+ // yt-dlp's raw spelling is read as well as the store's.
+ assert.equal(
+ classifyAgainstFilter(f, {
+ title: "x",
+ description: "",
+ live_status: "was_live",
+ }),
+ "livestream",
+ );
+});
+
+test("exclude still wins over a livestream", () => {
+ const f = compileDownloadFilter({
+ includeLivestreams: true,
+ exclude: "rerun",
+ });
+ assert.ok(f);
+ assert.equal(
+ classifyAgainstFilter(f, liveMeta("was_live", "a rerun stream")),
+ "rejected",
+ );
+});
+
+test("an exclude-only filter still passes everything it does not match", () => {
+ const f = compileDownloadFilter({ exclude: "rerun" });
+ assert.ok(f);
+ assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "text");
+ assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text");
+});
+
+test("includeLivestreams alone makes the channel filtered at all", () => {
+ // compileDownloadFilter returning null is what "no filter" means everywhere
+ // downstream, so this flag has to be enough on its own.
+ assert.notEqual(compileDownloadFilter({ includeLivestreams: true }), null);
+ assert.equal(compileDownloadFilter({ includeLivestreams: false }), null);
+});
+
+test("the registry names both selectors when it declines", () => {
+ const d = evaluateDownloadFilters(
+ ctx({
+ channelConfig: {
+ ...BASE_CONFIG,
+ downloadFilter: { include: "guest", includeLivestreams: true },
+ },
+ metadata: meta({ title: "Synthetic plainvid0001", live_status: "not_live" }),
+ }),
+ );
+ assert.ok(d);
+ assert.match(d.reason, /neither include \/guest\/i nor a livestream/);
+});
diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts
@@ -79,60 +79,124 @@ const skipLive: DownloadFilter = {
export type CompiledDownloadFilter = {
include: RegExp | null;
exclude: RegExp | null;
+ includeLivestreams: boolean;
};
-// Compile a channel's patterns. Returns null when there is no filter AND when a
+// Compile a channel's filter. Returns null when there is no filter AND when a
// pattern doesn't parse — an invalid regex makes the filter INERT rather than
// failing every download on the channel. The editor form refuses to save one,
// so this only fires for a hand-edited config.json.
+//
+// `includeLivestreams` alone IS a filter: it is a positive selector, so a
+// channel that sets only that is filtering (to livestreams).
export function compileDownloadFilter(
filter: DownloadFilterConfig | undefined | null,
): CompiledDownloadFilter | null {
const include = filter?.include?.trim() ?? "";
const exclude = filter?.exclude?.trim() ?? "";
- if (!include && !exclude) return null;
+ const includeLivestreams = filter?.includeLivestreams === true;
+ if (!include && !exclude && !includeLivestreams) return null;
try {
return {
include: include ? new RegExp(include, "i") : null,
exclude: exclude ? new RegExp(exclude, "i") : null,
+ includeLivestreams,
};
} catch {
return null;
}
}
+// What counts as a livestream for `includeLivestreams`. "was_live"/"post_live"
+// are finished VODs (the normal case); "is_live" is included so the FILTER
+// accepts a stream in progress — skip-live, the next entry in the registry,
+// still declines to download it until it ends. Two different questions, and
+// they are answered by two different filters on purpose.
+const LIVESTREAM_STATUSES: ReadonlySet<string> = new Set([
+ "was_live",
+ "post_live",
+ "is_live",
+]);
+
+// The metadata-scan store spells it `liveStatus`; yt-dlp's raw metadata spells
+// it `live_status`. Both are read here so the scan's prediction and the
+// download's decision cannot disagree.
+export type FilterableVideo = {
+ title?: string | null;
+ description?: string | null;
+ liveStatus?: string | null;
+ live_status?: string | null;
+};
+
+function isLivestream(meta: FilterableVideo): boolean {
+ const status = meta.liveStatus ?? meta.live_status ?? "";
+ return LIVESTREAM_STATUSES.has(status);
+}
+
// The only text a download filter ever sees. Kept here so the metadata scan's
// stored entries and the download-time prefetch can never be matched against
// different strings.
export function downloadFilterText(
- meta: { title?: string | null; description?: string | null } | null | undefined,
+ meta: FilterableVideo | null | undefined,
): string {
return `${meta?.title ?? ""}\n${meta?.description ?? ""}`;
}
-// PURE, and the single matcher. `true` = this video is rejected (the filter says
-// don't download it). Exclude wins over include.
+// Why a video passed, or that it didn't. "livestream" exists so the UI can say
+// how many videos a channel is keeping for a reason other than their name.
+export type DownloadFilterVerdict = "text" | "livestream" | "rejected";
+
+// PURE, and the SINGLE matcher. Both the download-time registry entry below and
+// the derived settled set (controller/metadataScanStore.ts) call exactly this,
+// which is what makes "what the scan predicted" and "what the download did" the
+// same computation rather than two that merely look alike.
+//
+// The rule, in order:
+// 1. `exclude` always wins. A match is rejected however else it qualifies.
+// 2. With NO positive selector (an exclude-only filter), everything else
+// passes.
+// 3. Otherwise a video must satisfy at least one positive selector: the
+// `include` pattern over title + "\n" + description, or — when
+// `includeLivestreams` is on — being a livestream.
//
-// Both the download-time registry entry below and the derived settled set
-// (controller/metadataScanStore.ts:settledByTitleFilterIds) call exactly this,
-// which is what makes "what the scan predicted" and "what the download did"
-// the same computation rather than two that look alike.
+// THE CORNER WORTH STATING: `{ includeLivestreams: true }` with no `include`
+// rejects plain uploads and passes livestreams. `includeLivestreams` is a
+// second positive selector, not a modifier of the first, so switching it on
+// without an `include` is "livestreams only" rather than "everything, plus
+// livestreams".
+export function classifyAgainstFilter(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): DownloadFilterVerdict {
+ const text = downloadFilterText(meta);
+ if (compiled.exclude && compiled.exclude.test(text)) return "rejected";
+ const hasPositive = Boolean(compiled.include || compiled.includeLivestreams);
+ if (!hasPositive) return "text";
+ if (compiled.include && compiled.include.test(text)) return "text";
+ if (compiled.includeLivestreams && isLivestream(meta)) return "livestream";
+ return "rejected";
+}
+
export function titleFilterRejects(
compiled: CompiledDownloadFilter,
- meta: { title?: string | null; description?: string | null },
+ meta: FilterableVideo,
): boolean {
- const text = downloadFilterText(meta);
- if (compiled.exclude && compiled.exclude.test(text)) return true;
- if (compiled.include && !compiled.include.test(text)) return true;
- return false;
+ return classifyAgainstFilter(compiled, meta) === "rejected";
}
// Why a rejection happened, for the log and the outcome record.
-export function titleFilterReason(compiled: CompiledDownloadFilter, text: string): string {
+export function titleFilterReason(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): string {
+ const text = downloadFilterText(meta);
if (compiled.exclude && compiled.exclude.test(text)) {
return `title/description matches exclude /${compiled.exclude.source}/i`;
}
- return `title/description does not match include /${compiled.include?.source ?? ""}/i`;
+ const wants: string[] = [];
+ if (compiled.include) wants.push(`include /${compiled.include.source}/i`);
+ if (compiled.includeLivestreams) wants.push("a livestream");
+ return `title/description matches neither ${wants.join(" nor ")}`;
}
// Per-channel include/exclude over title + description. Runs BEFORE skipLive so
@@ -150,7 +214,7 @@ const titleFilter: DownloadFilter = {
evaluate(ctx) {
const raw = ctx.channelConfig.downloadFilter;
const configured = Boolean(
- raw?.include?.trim() || raw?.exclude?.trim(),
+ raw?.include?.trim() || raw?.exclude?.trim() || raw?.includeLivestreams,
);
if (!configured) return null;
const compiled = compileDownloadFilter(raw);
@@ -177,7 +241,7 @@ const titleFilter: DownloadFilter = {
return {
skip: true,
filter: "titleFilter",
- reason: titleFilterReason(compiled, downloadFilterText(m)),
+ reason: titleFilterReason(compiled, m),
};
},
};
diff --git a/editor/app/channels/[slug]/components/stages/PlaylistStage.tsx b/editor/app/channels/[slug]/components/stages/PlaylistStage.tsx
@@ -197,6 +197,17 @@ function MetadataScanSummary({ view }: { view: MetadataScanView }) {
<strong>{view.filteredOut}</strong> filtered out ·{" "}
<strong>{view.unscanned}</strong> unscanned
</p>
+ {/* The split matters when "Include every livestream" is on: without
+ it, switching that on is indistinguishable from widening the
+ regex. */}
+ <p
+ aria-label="download filter match split"
+ className="text-xs text-muted-foreground"
+ >
+ {view.matchedByText} by title/description ·{" "}
+ {view.matchedLivestreams} livestream
+ {view.matchedLivestreams === 1 ? "" : "s"}
+ </p>
{view.matched.length > 0 && (
<details className="text-sm">
<summary className="cursor-pointer text-muted-foreground">
diff --git a/editor/app/channels/[slug]/lib/metadataScanView.ts b/editor/app/channels/[slug]/lib/metadataScanView.ts
@@ -29,5 +29,10 @@ export type MetadataScanView = {
filterConfigured: boolean;
filteredOut: number;
matchedTotal: number;
+ // The split of `matchedTotal`: how many passed on their title/description and
+ // how many only because "Include every livestream" is on. Without the split,
+ // turning that switch on looks identical to widening the regex.
+ matchedByText: number;
+ matchedLivestreams: number;
matched: MetadataScanMatch[];
};
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -75,8 +75,8 @@ import {
} from "./lib/metadataScanView";
import { loadMetadataScan } from "yt-dlp-transcript-common/controller/metadataScanStore";
import {
+ classifyAgainstFilter,
compileDownloadFilter,
- titleFilterRejects,
} from "yt-dlp-transcript-common/lib/downloadFilters";
import { TranscribeStage } from "./components/stages/TranscribeStage";
import { DigestStage } from "./components/stages/DigestStage";
@@ -382,10 +382,18 @@ export default async function ChannelDetailPage({
const compiledFilter = compileDownloadFilter(config!.downloadFilter);
const matched: MetadataScanMatch[] = [];
let filteredOut = 0;
+ let matchedByText = 0;
+ let matchedLivestreams = 0;
if (compiledFilter) {
for (const [id, e] of Object.entries(scanStore.entries)) {
- if (titleFilterRejects(compiledFilter, e)) filteredOut++;
- else matched.push({ id, uploadDate: e.uploadDate, title: e.title });
+ const verdict = classifyAgainstFilter(compiledFilter, e);
+ if (verdict === "rejected") {
+ filteredOut++;
+ continue;
+ }
+ if (verdict === "livestream") matchedLivestreams++;
+ else matchedByText++;
+ matched.push({ id, uploadDate: e.uploadDate, title: e.title });
}
// Newest first, and capped: the list is an eyeball check that the
// regex caught what the operator meant, not an inventory.
@@ -406,6 +414,8 @@ export default async function ChannelDetailPage({
filterConfigured: Boolean(compiledFilter),
filteredOut,
matchedTotal: matched.length,
+ matchedByText,
+ matchedLivestreams,
matched: matched.slice(0, MATCHED_PREVIEW_LIMIT),
}}
/>
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -714,6 +714,23 @@ export function ChannelForm({
placeholder="(exclude nothing)"
hint="Case-insensitive regex, applied to title + description and winning over include. Matching videos are settled until you change this."
/>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="downloadFilterIncludeLivestreams"
+ defaultChecked={c?.downloadFilter?.includeLivestreams ?? false}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-0.5">
+ <span className="font-medium">Include every livestream</span>
+ <span className="text-xs text-muted-foreground">
+ A second way in, beside the include pattern: any livestream VOD is
+ downloaded whatever it is called. On its own — with no include
+ pattern — this means <em>livestreams only</em>: ordinary uploads
+ are then filtered out. An exclude pattern still wins over it.
+ </span>
+ </span>
+ </label>
<Field
label="Cookies from browser"
name="cookiesFromBrowser"
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -230,11 +230,14 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
);
}
}
+ const includeLivestreams =
+ formData.get("downloadFilterIncludeLivestreams") != null;
const downloadFilter: ChannelConfig["downloadFilter"] | undefined =
- downloadFilterInclude || downloadFilterExclude
+ downloadFilterInclude || downloadFilterExclude || includeLivestreams
? {
...(downloadFilterInclude ? { include: downloadFilterInclude } : {}),
...(downloadFilterExclude ? { exclude: downloadFilterExclude } : {}),
+ ...(includeLivestreams ? { includeLivestreams: true } : {}),
}
: undefined;
diff --git a/editor/app/lib/actionable/loadActionable.ts b/editor/app/lib/actionable/loadActionable.ts
@@ -186,7 +186,8 @@ export async function loadActionableSummary(
actionableMetadataScanCount(r) > 0 &&
Boolean(
r.channel.config.downloadFilter?.include?.trim() ||
- r.channel.config.downloadFilter?.exclude?.trim(),
+ r.channel.config.downloadFilter?.exclude?.trim() ||
+ r.channel.config.downloadFilter?.includeLivestreams,
),
)
.sort((a, b) => actionableMetadataScanCount(b) - actionableMetadataScanCount(a));
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -147,7 +147,10 @@ function urlSentinels(url) {
noCaptions: lower.includes("nocaptions"),
isLive: lower.includes("islive"),
isUpcoming: lower.includes("isupcoming"),
- wasLive: lower.includes("waslive"),
+ // `livevid` is the download-filter fixture's livestream: a FINISHED VOD, so
+ // it downloads normally and skip-live leaves it alone — which is what makes
+ // it a clean test of `includeLivestreams`.
+ wasLive: lower.includes("waslive") || lower.includes("livevid"),
longDuration: lower.includes("longvideo"),
odysee: lower.includes("odyseevid"),
};
diff --git a/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist b/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist
@@ -3,3 +3,4 @@ https://www.youtube.com/watch?v=guestvid0002
https://www.youtube.com/watch?v=plainvid0001
https://www.youtube.com/watch?v=plainvid0002
https://www.youtube.com/watch?v=needsauthvid01
+https://www.youtube.com/watch?v=livevid000001
diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts
@@ -22,6 +22,10 @@ const ROOT = `test-transcripts/channels/${CHANNEL}`;
const GUEST = ["guestvid0001", "guestvid0002"];
const PLAIN = ["plainvid0001", "plainvid0002"];
const NEEDS_AUTH = "needsauthvid01";
+// A FINISHED livestream VOD. Its title contains neither "guest" nor "plain", so
+// the only thing that can let it through is `includeLivestreams`.
+const LIVE = "livevid000001";
+const SETTLED_BY_GUEST = [LIVE, ...PLAIN].sort();
type MetadataScan = {
version: number;
@@ -123,19 +127,21 @@ test("the scan reads titles without creating a single video directory", async ({
await runScan(page);
const scan = await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`);
- expect(Object.keys(scan.entries).sort()).toEqual([...GUEST, ...PLAIN].sort());
+ expect(Object.keys(scan.entries).sort()).toEqual(
+ [...GUEST, ...PLAIN, LIVE].sort(),
+ );
expect(scan.entries.guestvid0001.title).toBe("Synthetic guestvid0001");
// The one video the fake refuses without cookies. No cookie spec is
// configured, so there is no auth retry to make and it stays an error.
expect(Object.keys(scan.errors)).toEqual([NEEDS_AUTH]);
expect(scan.errors[NEEDS_AUTH].class).toBe("needs_auth");
- expect(scan.lastRun?.scanned).toBe(4);
+ expect(scan.lastRun?.scanned).toBe(5);
// THE LOAD-BEARING ASSERTION. A video directory holding a metadata.info.json
// is admitted to the LMDB index and the exported site by buildIndex, and its
// dir name is what deriveChannelSets reads as "ever fetched". The scan must
// create none of them.
- for (const id of [...GUEST, ...PLAIN, NEEDS_AUTH]) {
+ for (const id of [...GUEST, ...PLAIN, LIVE, NEEDS_AUTH]) {
expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false);
}
expect(await readArchive()).toBe("");
@@ -143,11 +149,11 @@ test("the scan reads titles without creating a single video directory", async ({
// The report then settles the non-matches, with nothing written per video.
const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
const snapshot = await refreshReport(page, before.generatedAt);
- expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN);
+ expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(SETTLED_BY_GUEST);
expect([...snapshot.undownloadedIds].sort()).toEqual(
[...GUEST, NEEDS_AUTH].sort(),
);
- expect(snapshot.metadataScan?.scanned).toBe(4);
+ expect(snapshot.metadataScan?.scanned).toBe(5);
expect(snapshot.metadataScan?.errors).toBe(1);
// The unreadable video is not re-queued for a scan for a day — otherwise a
// members-only video would keep the backlog above zero forever. It is still
@@ -172,13 +178,13 @@ test("a settled video is never downloaded", async ({ page }) => {
// Settled: no directory, no archive line, and — the part that matters on a
// channel with ~1,800 of them — yt-dlp was never invoked for them at all.
const invocations = await readInvocations();
- for (const id of PLAIN) {
+ for (const id of SETTLED_BY_GUEST) {
expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false);
expect(await readArchive()).not.toContain(id);
expect(invocations).not.toContain(`prefetch:https://www.youtube.com/watch?v=${id}`);
}
await expect(page.getByLabel("Download videos output")).toContainText(
- "Settled by this channel's download filter: 2 skipped",
+ "Settled by this channel's download filter: 3 skipped",
);
});
@@ -191,7 +197,7 @@ test("changing the filter re-decides the channel with no rescan", async ({
await runScan(page);
const afterScan = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
const settled = await refreshReport(page, afterScan.generatedAt);
- expect(settled.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN);
+ expect(settled.buckets.skippedByTitleFilter ?? []).toEqual(SETTLED_BY_GUEST);
const invocationsBefore = await readInvocations();
@@ -200,7 +206,9 @@ test("changing the filter re-decides the channel with no rescan", async ({
await writeFilterConfig({ include: "plain" });
const flipped = await refreshReport(page, settled.generatedAt);
- expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(GUEST);
+ expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(
+ [...GUEST, LIVE].sort(),
+ );
for (const id of PLAIN) expect(flipped.undownloadedIds).toContain(id);
for (const id of GUEST) expect(flipped.undownloadedIds).not.toContain(id);
// No rescan: re-deciding is pure computation over what the scan already read.
@@ -225,7 +233,7 @@ test("the Playlist stage reports the scan and what the filter makes of it", asyn
await page.goto(channelStage(CHANNEL, "playlist"));
const summary = page.getByLabel("metadata scan summary");
- await expect(summary).toContainText("Scanned 0 of 5 listed");
+ await expect(summary).toContainText("Scanned 0 of 6 listed");
await page.getByRole("button", { name: "Scan metadata" }).click();
await expect(page.getByLabel("Scan metadata output")).toContainText(
@@ -234,11 +242,14 @@ test("the Playlist stage reports the scan and what the filter makes of it", asyn
);
await page.goto(channelStage(CHANNEL, "playlist"));
- await expect(summary).toContainText("Scanned 4 of 5 listed");
+ await expect(summary).toContainText("Scanned 5 of 6 listed");
await expect(summary).toContainText("1 could not be read");
const verdict = page.getByLabel("download filter verdict");
await expect(verdict).toContainText("2 match");
- await expect(verdict).toContainText("2 filtered out");
+ await expect(verdict).toContainText("3 filtered out");
+ await expect(page.getByLabel("download filter match split")).toContainText(
+ "2 by title/description · 0 livestreams",
+ );
await page.getByText("Matched titles").click();
const matched = page.getByLabel("matched titles");
await expect(matched).toContainText("Synthetic guestvid0001");
@@ -329,7 +340,7 @@ test("/operations/metadata-scan lists the channel and can run it", async ({
// unscanned playlist before the operation page can offer the work.
const first = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
const seeded = await refreshReport(page, first.generatedAt);
- expect(seeded.metadataScan?.unscanned).toBe(5);
+ expect(seeded.metadataScan?.unscanned).toBe(6);
await page.goto("/operations/metadata-scan");
await expect(
@@ -350,5 +361,52 @@ test("/operations/metadata-scan lists the channel and can run it", async ({
).length,
{ timeout: 60_000, intervals: [500, 1000] },
)
- .toBe(4);
+ .toBe(5);
+});
+
+test("Include every livestream lets the VOD through, with no rescan", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await runScan(page);
+ const afterScan = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const settled = await refreshReport(page, afterScan.generatedAt);
+ // The livestream is settled by /guest/ like any other non-matching title.
+ expect(settled.buckets.skippedByTitleFilter ?? []).toContain(LIVE);
+
+ const invocationsBefore = await readInvocations();
+
+ await page.goto(channelStage(CHANNEL, "configure"));
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await page.getByLabel(/^include every livestream/i).check();
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () =>
+ (
+ await readJson<{ downloadFilter?: Record<string, unknown> }>(
+ `${ROOT}/config.json`,
+ ).catch(() => ({}) as { downloadFilter?: Record<string, unknown> })
+ ).downloadFilter ?? null,
+ { timeout: 30_000, intervals: [250, 500] },
+ )
+ .toEqual({ include: "guest", includeLivestreams: true });
+
+ const flipped = await refreshReport(page, settled.generatedAt);
+ expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN);
+ expect(flipped.undownloadedIds).toContain(LIVE);
+ // A second positive selector, evaluated over what the scan already read.
+ expect(await readInvocations()).toBe(invocationsBefore);
+
+ // The stage now says WHY it is matched, which is the whole reason the count
+ // is split: turning this on would otherwise look like widening the regex.
+ await page.goto(channelStage(CHANNEL, "playlist"));
+ await expect(page.getByLabel("download filter match split")).toContainText(
+ "2 by title/description · 1 livestream",
+ );
+
+ await download(page);
+ expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.en.vtt`)).toBe(true);
});
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4070,6 +4070,32 @@ never pulls `reader-fs.ts` into a client chunk.
rescan and nothing to migrate — asserted in e2e by a byte-identical
`fake-ytdlp.invocations` across the flip. `DownloadOutcomeRecord.filter` is
`{name, reason}` and deliberately carries no verdict.
+- **`includeLivestreams` is a SECOND positive selector, not a modifier.** The
+ rule, in `classifyAgainstFilter` and nowhere else: `exclude` always wins; with
+ no positive selector (exclude-only) everything else passes; otherwise a video
+ must satisfy `include` over title+description OR be a livestream when
+ `includeLivestreams` is on. The corner that surprises: `{ includeLivestreams:
+ true }` with no `include` means "livestreams only" — plain uploads are
+ rejected. Live statuses that count are `was_live`, `post_live`, `is_live`;
+ `is_live` is included because the FILTER accepting a live stream and skip-live
+ declining to download it until it ends are two different questions, answered
+ by two registry entries. The store persists `liveStatus` for exactly this, and
+ the matcher reads both `liveStatus` (store) and `live_status` (raw yt-dlp) so
+ the scan's prediction and the download's decision cannot diverge. The
+ Playlist-stage count is SPLIT (by title/description vs livestreams) because
+ otherwise switching the flag on is indistinguishable from widening the regex.
+- **A run of "Video unavailable" is a soft block, not availability.** Measured
+ 2026-09-20 on a raw yt-dlp pass over one channel's playlist: after ~100 good
+ records YouTube answered `Video unavailable` for 545 CONSECUTIVE videos, all of
+ which fetched fine when probed individually minutes later. `runMetadataScan`
+ therefore HOLDS those errors instead of recording them, releasing them only
+ when the streak breaks (a readable record, or a different error); at
+ `SOFT_BLOCK_STREAK` = 10 it stops the run, records the shared per-platform
+ cooldown and DISCARDS the streak, so the next run re-reads those ids. Recording
+ them would mark public videos deleted AND suppress re-reading each for a day.
+ Members-only ("Join this channel…") breaks the streak and is recorded as
+ `members_only` — it is genuinely not fetchable and must not be discarded with
+ a block's ids.
- **It fails closed in both directions.** No filter, or an unparseable one
(which is inert at download time too), settles nothing. An id the scan only has
an ERROR for is never settled — we do not know what it is called, so we must