commit ecdb2bee84ca8e82aba5d35cfd8ce958e1212c3b
parent 3b4c089bcc36af444084991943b299e65b80ede4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 00:19:56 -0400
download filter: settled is DERIVED from a channel-level metadata scan
Replaces the stored-signature settlement. Two facts made a per-video store the
wrong place and neither is negotiable: buildIndex.ts:309-315 admits any
`data/<id>/` holding a metadata.info.json, transcript or not — so a
metadata-only pass that created directories would publish ~1,800 text-less
videos to the exported site — and deriveChannelSets keys "ever fetched" off
those same dir names, so a scanned id would stop being reported as never
fetched.
So the scan's store is `channels/<slug>/metadata-scan.json`, one file per
channel, modelled on rosterStore: versioned, normalize-on-read, additive,
write skipped when nothing changed. An entry clears that id's error; a later
error never un-scans an id.
`settledByTitleFilterIds(paths, slug, config)` computes the verdict from those
entries and the channel's CURRENT patterns, every time it is asked. That is the
whole design: editing a regex re-decides the channel on the next snapshot with
no rescan, no migration and nothing stored per video to rewrite. It fails
closed both ways — no filter (or an unparseable one) settles nothing, and an id
we only have an ERROR for is never settled, because we do not know what it is
called.
`DownloadOutcomeRecord.filter` goes back to {name, reason}: the outcome records
what happened, not a verdict. The four consumers — snapshot bucket,
undownloadedIds, the sync page walk, the batch exclusion pass — each load the
set ONCE per run instead of reading a sidecar per video. A download-time
rejection upserts the prefetched metadata into the store, so it settles without
a second fetch.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
16 files changed, 1125 insertions(+), 310 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -46,7 +46,7 @@ import {
} from "../jobs/autoQueuePolicy";
import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
-import { isSettledByFilter } from "../lib/downloadFilters";
+import { settledByTitleFilterIds } from "./metadataScanStore";
import type { Paths } from "../lib/paths";
import { extractVideoId } from "../ytdlp/runYtdlp";
import { reconcileVideoDirs } from "./reconcileVideoDirs";
@@ -868,6 +868,12 @@ export async function generateChannelSnapshot(
if (v.excludedFromTruncatedCheck) excludedTruncatedIds.add(v.id);
}
+ // SETTLED, derived — the ids this channel's CURRENT download filter rejects
+ // out of everything the metadata scan has read. One file read per snapshot,
+ // and nothing about it is stored per video: editing a pattern re-decides the
+ // whole channel on the next report, with no rescan and nothing to migrate.
+ const settledIds = await settledByTitleFilterIds(paths, slug, config);
+
const noTranscript: string[] = [];
const downloadedNoTranscript: string[] = [];
const wrongFormatAudio: string[] = [];
@@ -880,7 +886,10 @@ export async function generateChannelSnapshot(
const corruptFullSource: string[] = [];
const nonStandardVtt: string[] = [];
const skippedByFilter: string[] = [];
- const skippedByTitleFilter: string[] = [];
+ // Settled ids that DO have a directory (a download-time rejection prefetched
+ // their metadata before deciding). Tracked separately from the full settled
+ // set only so totals.videos can subtract exactly the dirs it counted.
+ const settledOnDisk: string[] = [];
const incompleteTranscript: string[] = [];
const shortAudio: string[] = [];
const autoSubsOnly: string[] = [];
@@ -998,12 +1007,15 @@ export async function generateChannelSnapshot(
}
// SETTLED BY THE DOWNLOAD FILTER — the operator's own "not this one".
// Short-circuited here, beside the other two terminal outcomes, and for the
- // same reason: everything below classifies a video by what is missing from
- // its directory, and a settled video is missing everything. Without this it
- // would land in noTranscript AND noMetadata AND the retryable
- // skippedByFilter, three times over, ~1,800 times on a filtered channel.
- if (isSettledByFilter(outcome, config)) {
- skippedByTitleFilter.push(id);
+ // same reason: everything below classifies a video by what is MISSING from
+ // its directory, and a settled video is missing everything. Most settled ids
+ // have no directory at all (the metadata scan never makes one) and are added
+ // to the bucket below; this branch catches the ones a download-time
+ // rejection left a prefetched metadata.info.json behind for, which would
+ // otherwise land in noTranscript AND noMetadata AND the retryable
+ // skippedByFilter at once.
+ if (settledIds.has(id)) {
+ settledOnDisk.push(id);
continue;
}
if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id);
@@ -1203,7 +1215,7 @@ export async function generateChannelSnapshot(
const undownloadedIds: string[] = [];
const needsCookies: string[] = [];
- const settledByFilterSet = new Set(skippedByTitleFilter);
+
// Walked in playlist order, not sorted: this is the auto-download runner's
// work queue, and the listing is newest-first.
for (const url of urls) {
@@ -1217,7 +1229,7 @@ export async function generateChannelSnapshot(
// This is the line that makes the settlement STICK: undownloadedIds is the
// auto-download runner's work queue, and it is derived from artifacts, so an
// archive line alone would not have kept the id out of it.
- if (settledByFilterSet.has(dirId)) continue;
+ if (settledIds.has(dirId)) continue;
const effective = effectiveById.get(dirId);
if (effective && AUTH_RETRY_CLASSES.has(effective)) {
needsCookies.push(dirId);
@@ -1279,7 +1291,7 @@ export async function generateChannelSnapshot(
corruptFullSource: corruptFullSource.sort(),
nonStandardVtt: nonStandardVtt.sort(),
skippedByFilter: skippedByFilter.sort(),
- skippedByTitleFilter: skippedByTitleFilter.sort(),
+ skippedByTitleFilter: [...settledIds].sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
autoSubsOnly: autoSubsOnly.sort(),
@@ -1306,10 +1318,11 @@ export async function generateChannelSnapshot(
const snapshot: ChannelSnapshot = {
generatedAt: new Date().toISOString(),
totals: {
- // Settled videos are excluded: they are metadata stubs the operator asked
- // us not to fetch, not videos this channel has. Counting them would make
- // "downloaded 40 of 1,840" the permanent state of a filtered channel.
- videos: videoDirNames.length - skippedByTitleFilter.length,
+ // Settled videos are excluded: they are stubs the operator asked us not to
+ // fetch, not videos this channel has. Only the ones with a directory are
+ // subtracted — a scan-settled id was never counted in the first place,
+ // because the scan deliberately creates no directory.
+ videos: videoDirNames.length - settledOnDisk.length,
transcribed,
downloaded,
},
diff --git a/common/controller/metadataScanStore.test.ts b/common/controller/metadataScanStore.test.ts
@@ -0,0 +1,249 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import {
+ loadMetadataScan,
+ metadataScanPath,
+ settledByTitleFilterIds,
+ upsertMetadataScan,
+ type MetadataScanEntry,
+} from "./metadataScanStore";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/metadataScanStore.test.ts
+
+const SLUG = "chan";
+const T1 = "2026-09-01T00:00:00.000Z";
+const T2 = "2026-09-02T00:00:00.000Z";
+
+async function withPaths(fn: (paths: Paths) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-mdscan-"));
+ const paths = { channelsDir: path.join(dir, "channels") } as Paths;
+ await mkdir(path.join(paths.channelsDir, SLUG), { recursive: true });
+ try {
+ await fn(paths);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function entry(over: Partial<MetadataScanEntry> = {}): MetadataScanEntry {
+ return {
+ title: "Synthetic vid",
+ description: "Synthetic video vid",
+ uploadDate: "20240101",
+ scannedAt: T1,
+ ...over,
+ };
+}
+
+function config(include?: string, exclude?: string): ChannelConfig {
+ return {
+ handling: "youtube",
+ name: "Chan",
+ ...(include || exclude
+ ? {
+ downloadFilter: {
+ ...(include ? { include } : {}),
+ ...(exclude ? { exclude } : {}),
+ },
+ }
+ : {}),
+ };
+}
+
+test("a missing file loads as an empty store, not an error", async () => {
+ await withPaths(async (paths) => {
+ const scan = await loadMetadataScan(paths, SLUG);
+ assert.deepEqual(scan.entries, {});
+ assert.deepEqual(scan.errors, {});
+ assert.equal(scan.lastRun, null);
+ });
+});
+
+test("upsert is additive and persists across loads", async () => {
+ await withPaths(async (paths) => {
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ { entries: { a: entry({ title: "A" }) } },
+ T1,
+ );
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ { entries: { b: entry({ title: "B" }) } },
+ T2,
+ );
+ const scan = await loadMetadataScan(paths, SLUG);
+ assert.deepEqual(Object.keys(scan.entries).sort(), ["a", "b"]);
+ assert.equal(scan.entries.a.title, "A");
+ assert.equal(scan.updatedAt, T2);
+ });
+});
+
+test("an entry clears that id's error; a later error never un-scans it", async () => {
+ await withPaths(async (paths) => {
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ { errors: { a: { class: "needs_auth", message: "age", at: T1 } } },
+ T1,
+ );
+ assert.ok((await loadMetadataScan(paths, SLUG)).errors.a);
+
+ await upsertMetadataScan(paths, SLUG, { entries: { a: entry() } }, T2);
+ const after = await loadMetadataScan(paths, SLUG);
+ assert.equal(after.errors.a, undefined);
+ assert.ok(after.entries.a);
+
+ // A transient failure on a later pass must not throw away what we know.
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ { errors: { a: { class: "error", message: "flaky", at: T2 } } },
+ T2,
+ );
+ const later = await loadMetadataScan(paths, SLUG);
+ assert.equal(later.errors.a, undefined);
+ assert.ok(later.entries.a);
+ });
+});
+
+test("an unchanged upsert skips the write entirely", async () => {
+ await withPaths(async (paths) => {
+ await upsertMetadataScan(paths, SLUG, { entries: { a: entry() } }, T1);
+ const file = metadataScanPath(paths, SLUG);
+ const before = (await stat(file)).mtimeMs;
+ // Same content, different scannedAt — scannedAt alone is not a change,
+ // otherwise every re-run rewrites a multi-megabyte file for nothing.
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ { entries: { a: entry({ scannedAt: T2 }) } },
+ T2,
+ );
+ assert.equal((await stat(file)).mtimeMs, before);
+ assert.equal((await loadMetadataScan(paths, SLUG)).updatedAt, T1);
+ });
+});
+
+test("normalize-on-read drops unknown fields and malformed records", async () => {
+ await withPaths(async (paths) => {
+ await writeFile(
+ metadataScanPath(paths, SLUG),
+ JSON.stringify({
+ version: 1,
+ updatedAt: T1,
+ lastRun: { startedAt: T1, finishedAt: T1, scanned: 2, errors: 0, junk: 1 },
+ entries: {
+ good: { title: "T", description: "D", uploadDate: "20240101", scannedAt: T1, junk: true },
+ noStamp: { title: "T" },
+ notAnObject: 7,
+ },
+ errors: { bad: { class: "deleted", message: "gone", at: T1, junk: 2 } },
+ }),
+ );
+ const scan = await loadMetadataScan(paths, SLUG);
+ assert.deepEqual(Object.keys(scan.entries), ["good"]);
+ assert.deepEqual(Object.keys(scan.entries.good).sort(), [
+ "description",
+ "scannedAt",
+ "title",
+ "uploadDate",
+ ]);
+ assert.deepEqual(Object.keys(scan.errors), ["bad"]);
+ assert.deepEqual(Object.keys(scan.errors.bad).sort(), ["at", "class", "message"]);
+ assert.equal(scan.lastRun?.scanned, 2);
+ assert.equal("junk" in (scan.lastRun ?? {}), false);
+ });
+});
+
+test("a corrupt file loads as empty rather than throwing", async () => {
+ await withPaths(async (paths) => {
+ await writeFile(metadataScanPath(paths, SLUG), "{not json");
+ assert.deepEqual((await loadMetadataScan(paths, SLUG)).entries, {});
+ });
+});
+
+// --- settled, derived --------------------------------------------------------
+
+async function seed(paths: Paths) {
+ await upsertMetadataScan(
+ paths,
+ SLUG,
+ {
+ entries: {
+ guest1: entry({ title: "Synthetic guestvid0001", description: "" }),
+ guest2: entry({ title: "Episode 9", description: "with a guest" }),
+ plain1: entry({ title: "Synthetic plainvid0001", description: "" }),
+ plain2: entry({ title: "Synthetic plainvid0002", description: "" }),
+ },
+ errors: { unreadable: { class: "needs_auth", message: "age", at: T1 } },
+ },
+ T1,
+ );
+}
+
+test("settled = the ids the CURRENT filter rejects", async () => {
+ await withPaths(async (paths) => {
+ await seed(paths);
+ assert.deepEqual(
+ [...(await settledByTitleFilterIds(paths, SLUG, config("guest")))].sort(),
+ ["plain1", "plain2"],
+ );
+ // Edit the pattern: the whole channel is re-decided, with no rescan and
+ // nothing stored per video to migrate.
+ assert.deepEqual(
+ [...(await settledByTitleFilterIds(paths, SLUG, config("plain")))].sort(),
+ ["guest1", "guest2"],
+ );
+ // Exclude wins.
+ assert.deepEqual(
+ [
+ ...(await settledByTitleFilterIds(paths, SLUG, config(undefined, "guest"))),
+ ].sort(),
+ ["guest1", "guest2"],
+ );
+ });
+});
+
+test("no filter, and an inert filter, settle nothing", async () => {
+ await withPaths(async (paths) => {
+ await seed(paths);
+ assert.equal((await settledByTitleFilterIds(paths, SLUG, config())).size, 0);
+ assert.equal((await settledByTitleFilterIds(paths, SLUG, null)).size, 0);
+ // Unparseable = inert at download time, so it must settle nothing either.
+ assert.equal(
+ (await settledByTitleFilterIds(paths, SLUG, config("elf("))).size,
+ 0,
+ );
+ });
+});
+
+test("an id we only have an ERROR for is never settled", async () => {
+ await withPaths(async (paths) => {
+ await seed(paths);
+ const settled = await settledByTitleFilterIds(paths, SLUG, config("guest"));
+ // We do not know what `unreadable` is called, so we must not decide it —
+ // it stays ordinary undownloaded work.
+ assert.equal(settled.has("unreadable"), false);
+ });
+});
+
+test("an unscanned channel settles nothing, whatever the filter says", async () => {
+ await withPaths(async (paths) => {
+ assert.equal(
+ (await settledByTitleFilterIds(paths, SLUG, config("guest"))).size,
+ 0,
+ );
+ assert.equal(
+ await readFile(metadataScanPath(paths, SLUG), "utf8").catch(() => null),
+ null,
+ );
+ });
+});
diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts
@@ -0,0 +1,273 @@
+// The METADATA SCAN store: what a channel's listed-but-not-downloaded videos
+// are CALLED, without downloading them. `channels/<slug>/metadata-scan.json`.
+//
+// WHY THIS IS A CHANNEL-LEVEL FILE AND NOT data/<id>/metadata.info.json.
+// Two things downstream read the presence of a video directory as a fact about
+// the corpus, and both would be wrong if a metadata-only scan created one:
+//
+// 1. common/controller/buildIndex.ts:309-315 admits ANY `data/<id>/` that has
+// a metadata.info.json, transcript or not. A scan over a filtered channel
+// would put ~1,800 transcript-less videos into the LMDB index and into the
+// exported site — a published archive full of entries with no text.
+// 2. deriveChannelSets (./channelSets.ts:58-67, fed at
+// channelSnapshot.ts:1167-1174) keys "ever fetched" off the DIR NAMES under
+// data/. A scanned id would read as previously fetched and drop out of
+// `missingNeverFetched`, which is the one place a video that vanished
+// before we ever got it is recorded.
+//
+// So the scan never creates a video directory. One file per channel, additive,
+// and nothing derived is stored in it: whether a video is SETTLED by the
+// download filter is computed from these entries plus the channel's CURRENT
+// config every time it is asked (see settledByTitleFilterIds below). That is
+// what makes editing the regex free — no rescan, no migration, no stored
+// verdict to invalidate.
+//
+// Modelled on ./rosterStore.ts: versioned, normalize-on-read (unknown fields
+// dropped), additive merge that returns the same object when nothing changed,
+// tmp+rename write.
+
+import path from "node:path";
+import { readFile, rename, writeFile } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters";
+
+export const METADATA_SCAN_FILENAME = "metadata-scan.json";
+export const METADATA_SCAN_VERSION = 1;
+
+// One scanned video. `description` is kept because it is half of the text the
+// download filter matches against — storing only the title would make a
+// re-match against an edited pattern disagree with what the downloader does.
+export type MetadataScanEntry = {
+ title: string;
+ description: string;
+ // yt-dlp's `upload_date` (YYYYMMDD), or "" when the extractor had none.
+ uploadDate: string;
+ liveStatus?: string;
+ duration?: number;
+ scannedAt: string;
+};
+
+// A video the scan could not read. Kept so a re-run knows what to retry and the
+// UI can say why the numbers don't add up. `class` is an Availability class
+// where one could be parsed, else "unknown".
+export type MetadataScanError = {
+ class: string;
+ message: string;
+ at: string;
+};
+
+// Why the last run ended. `stopped` absent = it finished its target list.
+export type MetadataScanRun = {
+ startedAt: string;
+ finishedAt: string;
+ scanned: number;
+ errors: number;
+ stopped?: "rate_limit" | "aborted" | "error";
+ message?: string;
+};
+
+export type MetadataScan = {
+ version: number;
+ updatedAt: string;
+ lastRun: MetadataScanRun | null;
+ entries: Record<string, MetadataScanEntry>;
+ errors: Record<string, MetadataScanError>;
+};
+
+export function emptyMetadataScan(): MetadataScan {
+ return {
+ version: METADATA_SCAN_VERSION,
+ updatedAt: "",
+ lastRun: null,
+ entries: {},
+ errors: {},
+ };
+}
+
+export function metadataScanPath(paths: Paths, slug: string): string {
+ return path.join(paths.channelsDir, slug, METADATA_SCAN_FILENAME);
+}
+
+function normalizeEntry(raw: unknown): MetadataScanEntry | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ if (typeof r.scannedAt !== "string") return null;
+ const entry: MetadataScanEntry = {
+ title: typeof r.title === "string" ? r.title : "",
+ description: typeof r.description === "string" ? r.description : "",
+ uploadDate: typeof r.uploadDate === "string" ? r.uploadDate : "",
+ scannedAt: r.scannedAt,
+ };
+ if (typeof r.liveStatus === "string") entry.liveStatus = r.liveStatus;
+ if (typeof r.duration === "number" && Number.isFinite(r.duration)) {
+ entry.duration = r.duration;
+ }
+ return entry;
+}
+
+function normalizeError(raw: unknown): MetadataScanError | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ if (typeof r.at !== "string") return null;
+ return {
+ class: typeof r.class === "string" ? r.class : "unknown",
+ message: typeof r.message === "string" ? r.message : "",
+ at: r.at,
+ };
+}
+
+function normalizeRun(raw: unknown): MetadataScanRun | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ if (typeof r.startedAt !== "string" || typeof r.finishedAt !== "string") {
+ return null;
+ }
+ const run: MetadataScanRun = {
+ startedAt: r.startedAt,
+ finishedAt: r.finishedAt,
+ scanned: typeof r.scanned === "number" ? r.scanned : 0,
+ errors: typeof r.errors === "number" ? r.errors : 0,
+ };
+ if (
+ r.stopped === "rate_limit" ||
+ r.stopped === "aborted" ||
+ r.stopped === "error"
+ ) {
+ run.stopped = r.stopped;
+ }
+ if (typeof r.message === "string") run.message = r.message;
+ return run;
+}
+
+export async function loadMetadataScan(
+ paths: Paths,
+ slug: string,
+): Promise<MetadataScan> {
+ try {
+ const raw = await readFile(metadataScanPath(paths, slug), "utf8");
+ const parsed = JSON.parse(raw) as Record<string, unknown>;
+ if (!parsed || typeof parsed !== "object") return emptyMetadataScan();
+ const entries: Record<string, MetadataScanEntry> = {};
+ const rawEntries = (parsed.entries ?? {}) as Record<string, unknown>;
+ for (const [id, value] of Object.entries(rawEntries)) {
+ const entry = normalizeEntry(value);
+ if (entry) entries[id] = entry;
+ }
+ const errors: Record<string, MetadataScanError> = {};
+ const rawErrors = (parsed.errors ?? {}) as Record<string, unknown>;
+ for (const [id, value] of Object.entries(rawErrors)) {
+ const err = normalizeError(value);
+ if (err) errors[id] = err;
+ }
+ return {
+ version: METADATA_SCAN_VERSION,
+ updatedAt: typeof parsed.updatedAt === "string" ? parsed.updatedAt : "",
+ lastRun: normalizeRun(parsed.lastRun),
+ entries,
+ errors,
+ };
+ } catch {
+ return emptyMetadataScan();
+ }
+}
+
+async function writeMetadataScan(
+ paths: Paths,
+ slug: string,
+ scan: MetadataScan,
+): Promise<void> {
+ const file = metadataScanPath(paths, slug);
+ const tmp = `${file}.tmp-${process.pid}`;
+ await writeFile(tmp, JSON.stringify(scan, null, 2) + "\n");
+ await rename(tmp, file);
+}
+
+export type MetadataScanUpsert = {
+ entries?: Record<string, MetadataScanEntry>;
+ errors?: Record<string, MetadataScanError>;
+ // Replaces `lastRun` wholesale. Omitted leaves the previous run record alone,
+ // which is what an incremental mid-run flush wants.
+ lastRun?: MetadataScanRun;
+};
+
+// Fold new records into the store and persist. ADDITIVE: an id never disappears
+// here, and a successful entry CLEARS that id's error (the video was readable
+// after all). Skips the write entirely when nothing changed, so a re-run over an
+// already-scanned channel costs one read.
+export async function upsertMetadataScan(
+ paths: Paths,
+ slug: string,
+ upsert: MetadataScanUpsert,
+ now: string,
+): Promise<MetadataScan> {
+ const scan = await loadMetadataScan(paths, slug);
+ let changed = false;
+
+ for (const [id, entry] of Object.entries(upsert.entries ?? {})) {
+ const prev = scan.entries[id];
+ if (
+ !prev ||
+ prev.title !== entry.title ||
+ prev.description !== entry.description ||
+ prev.uploadDate !== entry.uploadDate ||
+ prev.liveStatus !== entry.liveStatus ||
+ prev.duration !== entry.duration
+ ) {
+ scan.entries[id] = entry;
+ changed = true;
+ }
+ // A readable video is not an error, whatever it was last time.
+ if (scan.errors[id]) {
+ delete scan.errors[id];
+ changed = true;
+ }
+ }
+
+ for (const [id, err] of Object.entries(upsert.errors ?? {})) {
+ // An id we already have metadata for stays scanned — a transient failure on
+ // a later pass must not un-scan it.
+ if (scan.entries[id]) continue;
+ const prev = scan.errors[id];
+ if (!prev || prev.class !== err.class || prev.message !== err.message) {
+ scan.errors[id] = err;
+ changed = true;
+ }
+ }
+
+ if (upsert.lastRun) {
+ scan.lastRun = upsert.lastRun;
+ changed = true;
+ }
+
+ if (!changed) return scan;
+ scan.updatedAt = now;
+ await writeMetadataScan(paths, slug, scan);
+ return scan;
+}
+
+// SETTLED, derived. The ids this channel's CURRENT download filter rejects,
+// out of everything the scan has read. The one place any of the four consumers
+// (snapshot bucket, undownloadedIds, sync page walk, batch exclusion) get the
+// answer, and it is recomputed from the config every time — which is why
+// editing a pattern re-decides the whole channel on the next snapshot with no
+// rescan and nothing to migrate.
+//
+// FAILS CLOSED in both directions: no filter configured (or an unparseable one,
+// which is inert at download time too) settles nothing, and an id we only have
+// an ERROR for is never settled — we do not know what it is called, so we must
+// not decide it.
+export async function settledByTitleFilterIds(
+ paths: Paths,
+ slug: string,
+ config: Pick<ChannelConfig, "downloadFilter"> | null | undefined,
+): Promise<Set<string>> {
+ const settled = new Set<string>();
+ const compiled = compileDownloadFilter(config?.downloadFilter);
+ if (!compiled) return settled;
+ const scan = await loadMetadataScan(paths, slug);
+ for (const [id, entry] of Object.entries(scan.entries)) {
+ if (titleFilterRejects(compiled, entry)) settled.add(id);
+ }
+ return settled;
+}
diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts
@@ -1,12 +1,11 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import type { ChannelConfig } from "./channelConfig";
-import type { DownloadOutcomeRecord } from "./downloadOutcome";
import {
compileDownloadFilter,
- downloadFilterSignature,
+ downloadFilterText,
evaluateDownloadFilters,
- isSettledByFilter,
+ titleFilterRejects,
type DownloadFilterContext,
} from "./downloadFilters";
import type { RawMetadata } from "./transcripts-server";
@@ -27,9 +26,7 @@ function meta(over: Partial<RawMetadata> = {}): RawMetadata {
};
}
-function ctx(
- over: Partial<DownloadFilterContext> = {},
-): DownloadFilterContext {
+function ctx(over: Partial<DownloadFilterContext> = {}): DownloadFilterContext {
return {
metadata: meta(),
channelConfig: BASE_CONFIG,
@@ -39,10 +36,7 @@ function ctx(
};
}
-function withFilter(
- include?: string,
- exclude?: string,
-): ChannelConfig {
+function withFilter(include?: string, exclude?: string): ChannelConfig {
return {
...BASE_CONFIG,
downloadFilter: {
@@ -52,86 +46,118 @@ function withFilter(
};
}
-test("no filter configured -> no decision, and no signature", () => {
- assert.equal(downloadFilterSignature(undefined), null);
- assert.equal(downloadFilterSignature({}), null);
- assert.equal(downloadFilterSignature({ include: " " }), null);
+// The compiled filter for a pair of patterns, asserted non-null.
+function compiled(include?: string, exclude?: string) {
+ const c = compileDownloadFilter(withFilter(include, exclude).downloadFilter);
+ assert.ok(c);
+ return c;
+}
+
+test("no filter configured -> nothing compiles, nothing is decided", () => {
assert.equal(compileDownloadFilter(undefined), null);
+ assert.equal(compileDownloadFilter({}), null);
+ assert.equal(compileDownloadFilter({ include: " " }), null);
assert.equal(evaluateDownloadFilters(ctx()), null);
});
-test("include matches the title -> downloads", () => {
- const d = evaluateDownloadFilters(
- ctx({
- channelConfig: withFilter("guest"),
- metadata: meta({ title: "Synthetic guestvid0001", description: "x" }),
+test("invalid regex compiles to null (inert), it never throws", () => {
+ assert.equal(compileDownloadFilter({ include: "elf(" }), null);
+ assert.equal(compileDownloadFilter({ exclude: "a[" }), null);
+});
+
+test("titleFilterRejects: include hit, include miss, description-only hit", () => {
+ const c = compiled("guest");
+ assert.equal(
+ titleFilterRejects(c, { title: "Synthetic guestvid0001", description: "" }),
+ false,
+ );
+ assert.equal(
+ titleFilterRejects(c, { title: "Synthetic plainvid0001", description: "" }),
+ true,
+ );
+ // The description alone can satisfy include — it is half the matched text.
+ assert.equal(
+ titleFilterRejects(c, {
+ title: "Episode 12",
+ description: "With a special guest this week",
}),
+ false,
);
- assert.equal(d, null);
});
-test("include misses -> permanent skip carrying the signature", () => {
- const config = withFilter("guest");
- const d = evaluateDownloadFilters(
- ctx({
- channelConfig: config,
- metadata: meta({ title: "Synthetic plainvid0001", description: "x" }),
+test("titleFilterRejects: exclude wins over include", () => {
+ const c = compiled("guest", "rerun");
+ assert.equal(
+ titleFilterRejects(c, { title: "guest episode (rerun)", description: "" }),
+ true,
+ );
+ assert.equal(
+ titleFilterRejects(c, { title: "guest episode", description: "" }),
+ false,
+ );
+});
+
+test("matching is case-insensitive in both directions", () => {
+ assert.equal(
+ titleFilterRejects(compiled("GUEST"), {
+ title: "a guest appears",
+ description: "",
}),
+ false,
+ );
+ assert.equal(
+ titleFilterRejects(compiled(undefined, "rerun"), {
+ title: "A RERUN",
+ description: "",
+ }),
+ true,
);
- assert.ok(d);
- assert.equal(d.skip, true);
- assert.equal(d.filter, "titleFilter");
- assert.equal(d.permanent, true);
- assert.equal(d.signature, downloadFilterSignature(config.downloadFilter));
- assert.match(d.reason, /does not match include/);
});
-test("the description alone can satisfy include", () => {
- const d = evaluateDownloadFilters(
- ctx({
- channelConfig: withFilter("special guest"),
- metadata: meta({
- title: "Episode 12",
- description: "With a special guest this week",
- }),
+test("the matched text is title + newline + description, missing fields empty", () => {
+ assert.equal(downloadFilterText({ title: "a", description: "b" }), "a\nb");
+ assert.equal(downloadFilterText({ title: "a" }), "a\n");
+ assert.equal(downloadFilterText(null), "\n");
+ // A pattern must not straddle the boundary by accident — the separator is a
+ // newline, so /title description/ does NOT match.
+ assert.equal(
+ titleFilterRejects(compiled("title description"), {
+ title: "title",
+ description: "description",
}),
+ true,
);
- assert.equal(d, null);
});
-test("exclude wins over include", () => {
+test("registry: include miss -> a skip naming the reason, with no verdict", () => {
const d = evaluateDownloadFilters(
ctx({
- channelConfig: withFilter("guest", "rerun"),
- metadata: meta({ title: "guest episode (rerun)", description: "" }),
+ channelConfig: withFilter("guest"),
+ metadata: meta({ title: "Synthetic plainvid0001", description: "x" }),
}),
);
assert.ok(d);
- assert.equal(d.permanent, true);
- assert.match(d.reason, /matches exclude/);
+ assert.equal(d.skip, true);
+ assert.equal(d.filter, "titleFilter");
+ assert.match(d.reason, /does not match include \/guest\/i/);
+ // Whether this video is SETTLED is derived from the metadata-scan store and
+ // the channel's current patterns — never carried on the decision.
+ assert.deepEqual(Object.keys(d).sort(), ["filter", "reason", "skip"]);
});
-test("matching is case-insensitive in both directions", () => {
+test("registry: include hit downloads", () => {
assert.equal(
evaluateDownloadFilters(
ctx({
- channelConfig: withFilter("GUEST"),
- metadata: meta({ title: "a guest appears", description: "" }),
+ channelConfig: withFilter("guest"),
+ metadata: meta({ title: "Synthetic guestvid0001", description: "x" }),
}),
),
null,
);
- const d = evaluateDownloadFilters(
- ctx({
- channelConfig: withFilter(undefined, "rerun"),
- metadata: meta({ title: "A RERUN", description: "" }),
- }),
- );
- assert.ok(d);
- assert.equal(d.permanent, true);
});
-test("invalid regex -> inert filter, one log line, nothing skipped", () => {
+test("registry: invalid regex -> inert filter, exactly one log line", () => {
const lines: string[] = [];
const d = evaluateDownloadFilters(
ctx({
@@ -143,91 +169,22 @@ test("invalid regex -> inert filter, one log line, nothing skipped", () => {
assert.equal(d, null);
assert.equal(lines.length, 1);
assert.match(lines[0], /INERT/);
- assert.equal(compileDownloadFilter({ include: "elf(" }), null);
- // An inert filter still HAS a signature — the config is what it is; only the
- // compile fails. Nothing gets settled because nothing is skipped.
- assert.notEqual(downloadFilterSignature({ include: "elf(" }), null);
});
-test("null metadata with a filter configured -> NON-permanent skip", () => {
+test("registry: null metadata with a filter configured -> retryable skip", () => {
const d = evaluateDownloadFilters(
ctx({ channelConfig: withFilter("guest"), metadata: null }),
);
assert.ok(d);
assert.equal(d.skip, true);
assert.equal(d.filter, "titleFilter");
- assert.equal(d.permanent, false);
assert.match(d.reason, /failing closed/);
});
-test("null metadata with NO filter configured -> no skip (fail-open)", () => {
+test("registry: null metadata with NO filter -> no skip (fail-open)", () => {
assert.equal(evaluateDownloadFilters(ctx({ metadata: null })), null);
});
-function settledOutcome(signature: string): DownloadOutcomeRecord {
- return {
- videoId: "plainvid0001",
- status: "skipped-filtered",
- startedAt: "2026-01-01T00:00:00.000Z",
- finishedAt: "2026-01-01T00:00:01.000Z",
- attempts: [],
- filter: {
- name: "titleFilter",
- reason: "does not match include",
- permanent: true,
- signature,
- },
- };
-}
-
-test("a signature change flips isSettledByFilter back to false", () => {
- const config = withFilter("guest");
- const sig = downloadFilterSignature(config.downloadFilter);
- assert.ok(sig);
- const outcome = settledOutcome(sig);
-
- assert.equal(isSettledByFilter(outcome, config), true);
- // Same include, different exclude -> different signature -> unsettled.
- assert.equal(isSettledByFilter(outcome, withFilter("guest", "rerun")), false);
- // A different include -> unsettled.
- assert.equal(isSettledByFilter(outcome, withFilter("plain")), false);
- // Filter removed entirely -> unsettled.
- assert.equal(isSettledByFilter(outcome, BASE_CONFIG), false);
- assert.equal(isSettledByFilter(outcome, null), false);
-});
-
-test("only a permanent, signed filter skip is settled", () => {
- const config = withFilter("guest");
- const sig = downloadFilterSignature(config.downloadFilter)!;
- assert.equal(isSettledByFilter(null, config), false);
- assert.equal(
- isSettledByFilter(
- { status: "ok", filter: { name: "titleFilter", reason: "", permanent: true, signature: sig } },
- config,
- ),
- false,
- );
- // skip-live's skip carries neither flag: never settled, always retried.
- assert.equal(
- isSettledByFilter(
- { status: "skipped-filtered", filter: { name: "skipLive", reason: "live" } },
- config,
- ),
- false,
- );
- // Permanent but unsigned can never match a signature.
- assert.equal(
- isSettledByFilter(
- {
- status: "skipped-filtered",
- filter: { name: "titleFilter", reason: "", permanent: true },
- },
- config,
- ),
- false,
- );
-});
-
test("skipLive still fires, and titleFilter runs first", () => {
// A live video with no title filter: skip-live decides, as before.
const live = evaluateDownloadFilters(
@@ -235,10 +192,10 @@ test("skipLive still fires, and titleFilter runs first", () => {
);
assert.ok(live);
assert.equal(live.filter, "skipLive");
- assert.equal(live.permanent, undefined);
// A live video that ALSO misses the include: the title filter answers,
- // because it is first in the registry and its verdict is the terminal one.
+ // because it is first in the registry — the operator's "not this one" is a
+ // better description of the video than "it is live".
const both = evaluateDownloadFilters(
ctx({
channelConfig: withFilter("guest"),
diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts
@@ -11,7 +11,6 @@
// context (metadata + channel config + resolved settings).
import type { ChannelConfig, DownloadFilterConfig } from "./channelConfig";
-import type { DownloadOutcomeRecord } from "./downloadOutcome";
import type { RawMetadata } from "./transcripts-server";
export type DownloadFilterSettings = {
@@ -33,16 +32,6 @@ export type DownloadFilterDecision = {
skip: boolean;
filter: string;
reason: string;
- // TERMINAL, not "tried and failed". A permanent skip settles the video: the
- // snapshot keeps it out of undownloadedIds and out of totals.videos, and sync
- // counts it as a hit so a filtered channel's paged walk still stops on its
- // first settled page instead of re-walking every non-match every day.
- // Undefined/false = the historical retryable skip (skip-live).
- permanent?: boolean;
- // The filter identity a permanent skip is settled AGAINST. Re-evaluated on
- // every read: a video is settled only while this equals the channel's current
- // signature, so editing a pattern un-settles everything it decided.
- signature?: string;
};
type DownloadFilter = {
@@ -87,75 +76,83 @@ const skipLive: DownloadFilter = {
// The per-channel title/description filter.
// ---------------------------------------------------------------------------
-// The filter's identity, and the whole settlement mechanism in one string.
-// Returns null when the channel has no filter configured (both patterns blank),
-// which is also the "nothing is settled here" answer — an outcome recorded under
-// a signature can never match null.
-//
-// Versioned ("v1") so a future change to what the patterns are matched against
-// invalidates every stored settlement rather than silently keeping decisions
-// that were made against different text. NUL-separated because a regex may
-// contain any other character.
-export function downloadFilterSignature(
- filter: DownloadFilterConfig | undefined | null,
-): string | null {
- const include = filter?.include?.trim() ?? "";
- const exclude = filter?.exclude?.trim() ?? "";
- if (!include && !exclude) return null;
- return `v1\0${include}\0${exclude}`;
-}
-
export type CompiledDownloadFilter = {
include: RegExp | null;
exclude: RegExp | null;
- signature: string;
};
// Compile a channel's patterns. Returns null when there is no filter AND when a
-// pattern doesn't parse — an invalid regex makes the filter inert rather than
-// failing every download on the channel. The editor form refuses to save one, so
-// this only fires for a hand-edited config.json.
+// pattern doesn't parse — an invalid regex makes the filter INERT rather than
+// failing every download on the channel. The editor form refuses to save one,
+// so this only fires for a hand-edited config.json.
export function compileDownloadFilter(
filter: DownloadFilterConfig | undefined | null,
): CompiledDownloadFilter | null {
- const signature = downloadFilterSignature(filter);
- if (!signature) return null;
const include = filter?.include?.trim() ?? "";
const exclude = filter?.exclude?.trim() ?? "";
+ if (!include && !exclude) return null;
try {
return {
include: include ? new RegExp(include, "i") : null,
exclude: exclude ? new RegExp(exclude, "i") : null,
- signature,
};
} catch {
return null;
}
}
-// The only text a download filter ever sees. Kept here so the preview job and
-// the filter itself can never match against different strings.
+// The only text a download filter ever sees. Kept here so the metadata scan's
+// stored entries and the download-time prefetch can never be matched against
+// different strings.
export function downloadFilterText(
- meta: Pick<RawMetadata, "title" | "description"> | null | undefined,
+ meta: { title?: string | null; description?: string | null } | null | undefined,
): string {
return `${meta?.title ?? ""}\n${meta?.description ?? ""}`;
}
+// PURE, and the single matcher. `true` = this video is rejected (the filter says
+// don't download it). Exclude wins over include.
+//
+// Both the download-time registry entry below and the derived settled set
+// (controller/metadataScanStore.ts:settledByTitleFilterIds) call exactly this,
+// which is what makes "what the scan predicted" and "what the download did"
+// the same computation rather than two that look alike.
+export function titleFilterRejects(
+ compiled: CompiledDownloadFilter,
+ meta: { title?: string | null; description?: string | null },
+): boolean {
+ const text = downloadFilterText(meta);
+ if (compiled.exclude && compiled.exclude.test(text)) return true;
+ if (compiled.include && !compiled.include.test(text)) return true;
+ return false;
+}
+
+// Why a rejection happened, for the log and the outcome record.
+export function titleFilterReason(compiled: CompiledDownloadFilter, text: string): string {
+ if (compiled.exclude && compiled.exclude.test(text)) {
+ return `title/description matches exclude /${compiled.exclude.source}/i`;
+ }
+ return `title/description does not match include /${compiled.include?.source ?? ""}/i`;
+}
+
// Per-channel include/exclude over title + description. Runs BEFORE skipLive so
-// a video the operator doesn't want is settled without also being classified as
-// a live-stream retry.
+// a video the operator doesn't want is declined on its own terms rather than
+// being classified as a live-stream retry.
//
// FAILS CLOSED on missing metadata: a prefetch that produced nothing cannot be
// matched, and downloading it anyway would defeat the filter on exactly the
-// videos a flaky extractor hides. The skip is recorded NON-permanent, so it
-// lands in the existing retryable skippedByFilter bucket and the next run tries
-// again — the opposite of a settled video.
+// videos a flaky extractor hides. The skip is an ORDINARY retryable one (it
+// lands in the existing skippedByFilter bucket and the next run tries again) —
+// nothing here is terminal, because nothing here decides what is settled. The
+// settled set is derived from the metadata-scan store.
const titleFilter: DownloadFilter = {
name: "titleFilter",
evaluate(ctx) {
const raw = ctx.channelConfig.downloadFilter;
- const signature = downloadFilterSignature(raw);
- if (!signature) return null;
+ const configured = Boolean(
+ raw?.include?.trim() || raw?.exclude?.trim(),
+ );
+ if (!configured) return null;
const compiled = compileDownloadFilter(raw);
if (!compiled) {
ctx.onLog?.(
@@ -174,30 +171,14 @@ const titleFilter: DownloadFilter = {
filter: "titleFilter",
reason:
"no metadata to match the download filter against (failing closed)",
- permanent: false,
- signature,
};
}
- const text = downloadFilterText(m);
- if (compiled.exclude && compiled.exclude.test(text)) {
- return {
- skip: true,
- filter: "titleFilter",
- reason: `title/description matches exclude /${compiled.exclude.source}/i`,
- permanent: true,
- signature,
- };
- }
- if (compiled.include && !compiled.include.test(text)) {
- return {
- skip: true,
- filter: "titleFilter",
- reason: `title/description does not match include /${compiled.include.source}/i`,
- permanent: true,
- signature,
- };
- }
- return null;
+ if (!titleFilterRejects(compiled, m)) return null;
+ return {
+ skip: true,
+ filter: "titleFilter",
+ reason: titleFilterReason(compiled, downloadFilterText(m)),
+ };
},
};
@@ -214,26 +195,3 @@ export function evaluateDownloadFilters(
}
return null;
}
-
-// THE ONE DEFINITION OF "SETTLED". Every reader — the snapshot's bucket, the
-// sync page walk, the batch exclusion pass, the video page's badge — asks this
-// and nothing else, because "is this video done with?" answered two ways is how
-// a filtered channel ends up both settled and re-queued.
-//
-// Three things must hold: the outcome is a filter skip, it was recorded as
-// permanent, and its signature is still the channel's. The third is what makes
-// an edited pattern un-settle every video it decided, with no migration and no
-// sweep.
-export function isSettledByFilter(
- outcome:
- | Pick<DownloadOutcomeRecord, "status" | "filter">
- | null
- | undefined,
- config: Pick<ChannelConfig, "downloadFilter"> | null | undefined,
-): boolean {
- if (!outcome || outcome.status !== "skipped-filtered") return false;
- const recorded = outcome.filter;
- if (!recorded?.permanent || !recorded.signature) return false;
- const current = downloadFilterSignature(config?.downloadFilter);
- return current !== null && current === recorded.signature;
-}
diff --git a/common/lib/downloadOutcome-server.ts b/common/lib/downloadOutcome-server.ts
@@ -1,10 +1,5 @@
import path from "node:path";
import { readFile, rename, writeFile } from "node:fs/promises";
-import type { ChannelConfig } from "./channelConfig";
-import {
- downloadFilterSignature,
- isSettledByFilter,
-} from "./downloadFilters";
import {
DOWNLOAD_OUTCOME_FILENAME,
DOWNLOAD_OUTCOME_STATUS_VALUES,
@@ -49,18 +44,5 @@ export async function writeDownloadOutcome(
await rename(tmp, file);
}
-// Per-video-dir predicate for the download filter's settlement, shaped exactly
-// like isDeferredAuthExcluded (runYtdlp.ts) so the batch/sync exclusion passes
-// treat the two the same way. The signature check runs FIRST: a channel with no
-// filter configured pays no file read at all, which matters because this is
-// called once per URL on every page of every sync.
-export async function isVideoSettledByFilter(
- videoDir: string,
- config: Pick<ChannelConfig, "downloadFilter"> | null | undefined,
-): Promise<boolean> {
- if (downloadFilterSignature(config?.downloadFilter) === null) return false;
- return isSettledByFilter(await loadDownloadOutcome(videoDir), config);
-}
-
// Narrow re-export so callers don't have to import from both modules.
export type { DownloadOutcomeStatus };
diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts
@@ -100,21 +100,13 @@ export type DownloadOutcomeRecord = {
// Set when status is "skipped-filtered": which app-level filter declined the
// download and why. Recorded so the UI/log can explain the skip.
//
- // `permanent` + `signature` are what make a skip SETTLE. skip-live's skip is
- // transient (the stream ends and the video becomes downloadable), so it
- // carries neither and every sync retries it. The per-channel download filter's
- // skip is terminal FOR AS LONG AS THE FILTER SAYS SO: `signature` is the
- // filter's identity (see downloadFilterSignature) and a settled video is one
- // whose recorded signature still equals the channel's current one. Change
- // either pattern and the signature changes, the video stops being settled, and
- // it is re-evaluated on the next run — which is the whole reason this is a
- // signature rather than an archive line.
- filter?: {
- name: string;
- reason: string;
- permanent?: boolean;
- signature?: string;
- };
+ // DELIBERATELY CARRIES NO VERDICT. A filter skip is never itself terminal:
+ // whether a video is SETTLED by the channel's download filter is derived from
+ // the metadata-scan store plus the channel's current patterns
+ // (controller/metadataScanStore.ts:settledByTitleFilterIds), never stored per
+ // video. Writing the verdict here would mean editing a regex has to find and
+ // rewrite ~1,800 sidecars before the change takes effect.
+ filter?: { name: string; reason: string };
};
export const DOWNLOAD_OUTCOME_FILENAME = "download-outcome.json";
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -46,6 +46,7 @@ import {
isLivestreamMetadata,
} from "../lib/transcripts-server";
import { evaluateDownloadFilters } from "../lib/downloadFilters";
+import { upsertMetadataScan } from "../controller/metadataScanStore";
import { detectPlatform, type Platform } from "../lib/platform";
import { probeMediaDurationSec } from "./ffprobeDuration";
import {
@@ -608,10 +609,43 @@ async function runManagedDownload(
});
if (decision?.skip) {
opts.onLog(
- `Skipping ${canonicalId}: ${decision.reason} [filter=${decision.filter}${
- decision.permanent ? ", settled" : ""
- }]\n`,
+ `Skipping ${canonicalId}: ${decision.reason} [filter=${decision.filter}]\n`,
);
+ // A title-filter rejection feeds the metadata-scan store, so this video is
+ // settled from here on WITHOUT a second metadata fetch. The store is the
+ // one place a settled verdict is derived from; the outcome below records
+ // only what happened, exactly as skip-live's does. A metadata-less skip
+ // (the fail-closed branch) writes nothing here — there is nothing to
+ // store, and it must stay retryable.
+ if (decision.filter === "titleFilter" && metadata) {
+ try {
+ await upsertMetadataScan(
+ opts.paths,
+ opts.channelSlug,
+ {
+ entries: {
+ [canonicalId]: {
+ title: metadata.title ?? "",
+ description: metadata.description ?? "",
+ uploadDate: metadata.upload_date ?? "",
+ ...(metadata.live_status
+ ? { liveStatus: metadata.live_status }
+ : {}),
+ ...(typeof metadata.duration === "number"
+ ? { duration: metadata.duration }
+ : {}),
+ scannedAt: new Date().toISOString(),
+ },
+ },
+ },
+ new Date().toISOString(),
+ );
+ } catch (err) {
+ opts.onLog(
+ `Failed to record the metadata scan entry: ${(err as Error).message}\n`,
+ );
+ }
+ }
const finishedAt = new Date().toISOString();
const record: DownloadOutcomeRecord = {
videoId: canonicalId,
@@ -620,16 +654,7 @@ async function runManagedDownload(
startedAt,
finishedAt,
attempts,
- // permanent/signature ride along verbatim: they are the filter's own
- // verdict, and isSettledByFilter re-checks the signature on every read.
- filter: {
- name: decision.filter,
- reason: decision.reason,
- ...(decision.permanent === undefined
- ? {}
- : { permanent: decision.permanent }),
- ...(decision.signature ? { signature: decision.signature } : {}),
- },
+ filter: { name: decision.filter, reason: decision.reason },
};
try {
await mkdir(videoDir, { recursive: true });
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -28,7 +28,7 @@ import {
type ResolvedCookiePolicy,
} from "../lib/cookiePolicy";
import { resolveEffectiveAvailability } from "../lib/availability-server";
-import { isVideoSettledByFilter } from "../lib/downloadOutcome-server";
+import { settledByTitleFilterIds } from "../controller/metadataScanStore";
import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability";
import { runAvailabilityCheck } from "../controller/checkAvailability";
import { writeMaybeMissing } from "../controller/maybeMissingStore";
@@ -669,6 +669,12 @@ async function downloadPlaylistManaged(
// (and re-failing) on every batch. A forceCookies run resolves to mode
// "always" and so bypasses the defer exclusion.
const runCookiePolicy = resolveRunCookiePolicy(opts, effectiveChannelConfig);
+ // ONE read of the channel-level metadata-scan store for the whole batch.
+ const settledByFilter = await settledByTitleFilterIds(
+ opts.paths,
+ opts.channelSlug,
+ effectiveChannelConfig,
+ );
const excludedCounts = { members_only: 0, deleted: 0, private: 0 };
let deferredAuthCount = 0;
let settledByFilterCount = 0;
@@ -676,11 +682,11 @@ async function downloadPlaylistManaged(
for (const url of tofetch) {
const dirId = extractVideoId(url);
if (dirId) {
- // Settled by the channel's download filter. Checked FIRST and cheaply
- // (isVideoSettledByFilter short-circuits on a channel with no filter), so
- // a filtered channel's batch doesn't re-run the metadata prefetch for
- // every non-match on every run just to reach the same verdict.
- if (await isVideoSettledByFilter(path.join(dataDir, dirId), effectiveChannelConfig)) {
+ // Settled by the channel's download filter, from the set resolved once
+ // above. Checked FIRST, so a filtered channel's batch doesn't run the
+ // metadata prefetch for every non-match on every run just to reach the
+ // same verdict the scan already reached.
+ if (settledByFilter.has(dirId)) {
settledByFilterCount++;
continue;
}
@@ -1310,23 +1316,27 @@ function fullSweepDue(opts: RunYtdlpOpts): boolean {
// The per-page download filter, shared by both passes so they can never drift:
// entries already in the archive are counted as hits (the paged walk's stop
-// signal), videos settled by the channel's download filter are counted as hits
+// signal), videos SETTLED by the channel's download filter are counted as hits
// TOO, defer-mode needs_auth videos are held back for the Needs-cookies bucket,
// and everything else is queued for download.
//
// WHY A SETTLED VIDEO IS A HIT. `archivedHits > 0` is the paged walk's stop
// signal — "we have reached content we already have". A settled video is
-// content we have already decided about, and it never gets an archive line (an
-// archive line would mean "downloaded" to verifyTranscripts and the
-// missingFromArchive bucket). Counting it as new instead would make a filtered
-// channel's daily sync walk every page of ~1,800 non-matches to find nothing,
-// forever: the walk would never see a hit on the pages the filter emptied.
+// content we have already decided about, and it deliberately has no archive
+// line (an archive line would mean "downloaded" to verifyTranscripts and the
+// missingFromArchive bucket) and, if the metadata scan settled it, no directory
+// at all. Counting it as new instead would make a filtered channel's daily sync
+// walk every page of ~1,800 non-matches to find nothing, forever.
+//
+// `settled` is a SET, resolved once per run by the caller. It used to be a
+// per-video sidecar read; the whole point of deriving settlement from the
+// channel-level metadata-scan store is that this loop does no I/O for it.
async function selectDownloadableUrls(
pageUrls: ReadonlyArray<string>,
archive: Awaited<ReturnType<typeof readArchive>>,
dataDir: string,
runCookiePolicy: ResolvedCookiePolicy,
- channelConfig: Pick<ChannelConfig, "downloadFilter">,
+ settled: ReadonlySet<string>,
): Promise<{
newUrls: string[];
archivedHits: number;
@@ -1346,10 +1356,7 @@ async function selectDownloadableUrls(
const dirId = extractVideoId(url);
// Settled by the channel's download filter: counted as a hit (see above),
// and reported separately so the log doesn't claim they were archived.
- if (
- dirId &&
- (await isVideoSettledByFilter(path.join(dataDir, dirId), channelConfig))
- ) {
+ if (dirId && settled.has(dirId)) {
archivedHits++;
settledCount++;
continue;
@@ -1401,6 +1408,13 @@ async function syncPaged(opts: RunYtdlpOpts): Promise<void> {
// Resolved once for the whole sync: drives both the defer-mode needs_auth
// exclusion below and the managed downloads themselves.
const runCookiePolicy = resolveRunCookiePolicy(opts, opts.channelConfig);
+ // Resolved ONCE for the whole walk: the settled set is derived from one
+ // channel-level file plus the current patterns, so a page costs no I/O for it.
+ const settledIds = await settledByTitleFilterIds(
+ opts.paths,
+ opts.channelSlug,
+ opts.channelConfig,
+ );
for (
let page = 0;
@@ -1430,7 +1444,7 @@ async function syncPaged(opts: RunYtdlpOpts): Promise<void> {
archive,
dataDir,
runCookiePolicy,
- opts.channelConfig,
+ settledIds,
);
opts.onLog(
`Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` +
@@ -1568,6 +1582,13 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
let totalSkipped = 0;
let firstFailure: Error | null = null;
const runCookiePolicy = resolveRunCookiePolicy(opts, opts.channelConfig);
+ // Resolved ONCE for the whole walk: the settled set is derived from one
+ // channel-level file plus the current patterns, so a page costs no I/O for it.
+ const settledIds = await settledByTitleFilterIds(
+ opts.paths,
+ opts.channelSlug,
+ opts.channelConfig,
+ );
for (
let page = 0;
@@ -1586,7 +1607,7 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
archive,
dataDir,
runCookiePolicy,
- opts.channelConfig,
+ settledIds,
);
opts.onLog(
`Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` +
diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx
@@ -128,7 +128,7 @@ export function DiagnosticsStage({
// the configuration from the wrong page. Editing the patterns in
// Configure re-evaluates every one of them on the next report.
description:
- "Declined by this channel's download filter (Configure > Advanced) and settled, so nothing re-attempts them. Change the include/exclude patterns to bring them back — the signature they were settled under stops matching and they return as ordinary undownloaded videos.",
+ "The metadata scan read these titles and this channel's download filter (Configure > Advanced) rejects them, so nothing downloads or re-attempts them. Change the include/exclude patterns and they are re-decided on the next report — no rescan needed.",
ariaLabel: "filtered out",
},
];
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -1488,11 +1488,11 @@ function DownloadOutcomeBadge({
case "skipped-filtered": {
const f = outcome.filter;
const why = f?.reason ? `: ${f.reason}` : "";
- // "(settled)" is the whole distinction a reader needs here. A skip-live
- // skip comes back on the next sync by itself; a settled one does not,
- // and will not, until the channel's download filter changes.
- return f?.permanent
- ? `Filtered out (settled)${why}`
+ // The outcome records what happened, not a verdict — whether this video
+ // is SETTLED is derived from the channel's metadata scan and its current
+ // patterns, and the Diagnostics bucket is where that answer lives.
+ return f?.name === "titleFilter"
+ ? `Filtered out by the download filter${why}`
: `Skipped by filter${why}`;
}
default:
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -704,7 +704,7 @@ export function ChannelForm({
value={filterInclude}
onChange={setFilterInclude}
placeholder="(download everything)"
- hint="Case-insensitive regex. Only videos whose title + description match are downloaded. Non-matching videos are settled — not retried — until this filter changes; the listing and the roster stay complete either way."
+ hint="Case-insensitive regex. Only videos whose title + description match are downloaded. Run Metadata scan on the Playlist stage to read the titles first — non-matching videos are then settled (never downloaded, never retried) until you change this. The listing and the roster stay complete either way."
/>
<Field
label="Download filter: exclude (regex)"
@@ -712,7 +712,7 @@ export function ChannelForm({
value={filterExclude}
onChange={setFilterExclude}
placeholder="(exclude nothing)"
- hint="Case-insensitive regex, applied to title + description and winning over include. Matching videos are settled until this filter changes."
+ hint="Case-insensitive regex, applied to title + description and winning over include. Matching videos are settled until you change this."
/>
<Field
label="Cookies from browser"
diff --git a/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/config.json b/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/config.json
@@ -0,0 +1,8 @@
+{
+ "handling": "youtube",
+ "name": "Test Title Filter",
+ "url": "https://www.youtube.com/@example/videos",
+ "downloadFilter": {
+ "include": "guest"
+ }
+}
diff --git a/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist b/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist
@@ -0,0 +1,4 @@
+https://www.youtube.com/watch?v=guestvid0001
+https://www.youtube.com/watch?v=guestvid0002
+https://www.youtube.com/watch?v=plainvid0001
+https://www.youtube.com/watch?v=plainvid0002
diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts
@@ -0,0 +1,282 @@
+import { readFile, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import {
+ channelStage,
+ generateReport,
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+} from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+// The per-channel download filter, end to end. The fake yt-dlp titles every
+// video "Synthetic <id>" (fixtures/bin/fake-ytdlp.mjs), so an include of
+// /guest/ matches guestvid0001 and guestvid0002 and misses plainvid0001 and
+// plainvid0002 — no fixture-specific metadata needed.
+const CHANNEL = "test-filter";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+const GUEST = ["guestvid0001", "guestvid0002"];
+const PLAIN = ["plainvid0001", "plainvid0002"];
+
+// The signature the fixture's filter settles under: "v1\0<include>\0<exclude>".
+const GUEST_SIGNATURE = "v1\0guest\0";
+
+type DownloadOutcome = {
+ status: string;
+ filter?: {
+ name: string;
+ reason: string;
+ permanent?: boolean;
+ signature?: string;
+ };
+};
+
+type Snapshot = {
+ generatedAt: string;
+ totals: { videos: number; transcribed: number; downloaded: number };
+ buckets: { skippedByTitleFilter?: string[]; skippedByFilter?: string[] };
+ undownloadedIds: string[];
+};
+
+async function readArchive(): Promise<string> {
+ return readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(() => "");
+}
+
+async function writeConfig(filter: Record<string, string> | null) {
+ await writeFile(
+ resolvePath(`${ROOT}/config.json`),
+ JSON.stringify(
+ {
+ handling: "youtube",
+ name: "Test Title Filter",
+ url: "https://www.youtube.com/@example/videos",
+ ...(filter ? { downloadFilter: filter } : {}),
+ },
+ null,
+ 2,
+ ),
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+}
+
+// Regenerate the report and wait for the snapshot to actually advance. The
+// click is retried for the usual reason (a pre-hydration click fires nothing);
+// regenerating is idempotent.
+async function refreshReport(
+ page: import("@playwright/test").Page,
+ after: string,
+): Promise<Snapshot> {
+ await page.goto("/channels");
+ const refresh = page.getByRole("button", {
+ name: `refresh report ${CHANNEL}`,
+ });
+ await refresh.waitFor({ state: "visible" });
+ await expect
+ .poll(
+ async () => {
+ await refresh.click({ timeout: 5_000 }).catch(() => {});
+ for (let i = 0; i < 20; i++) {
+ const cur = await readJson<Snapshot>(
+ `${ROOT}/snapshot.json`,
+ ).catch(() => null);
+ if (cur && cur.generatedAt > after) return true;
+ await new Promise((r) => setTimeout(r, 250));
+ }
+ return false;
+ },
+ { timeout: 60_000, intervals: [1000] },
+ )
+ .toBe(true);
+ return readJson<Snapshot>(`${ROOT}/snapshot.json`);
+}
+
+async function download(page: import("@playwright/test").Page) {
+ await page.goto(channelStage(CHANNEL, "download"));
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+ return log;
+}
+
+test("a non-matching video is settled, not re-queued", async ({ page }) => {
+ test.setTimeout(150_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await page.goto(channelStage(CHANNEL, "download"));
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+
+ await expect(log).toContainText(
+ "Skipping plainvid0001: title/description does not match include /guest/i [filter=titleFilter, settled]",
+ { timeout: 60_000 },
+ );
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+
+ // The matching videos downloaded and were archived, exactly as with no filter.
+ for (const id of GUEST) {
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true);
+ expect(await readArchive()).toContain(id);
+ }
+
+ // The non-matching ones are SETTLED: a terminal outcome carrying the filter's
+ // signature, and deliberately NO archive line (an archive line means
+ // "downloaded" to verifyTranscripts and the missingFromArchive bucket).
+ for (const id of PLAIN) {
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/${id}/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("skipped-filtered");
+ expect(outcome.filter?.name).toBe("titleFilter");
+ expect(outcome.filter?.permanent).toBe(true);
+ expect(outcome.filter?.signature).toBe(GUEST_SIGNATURE);
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(
+ false,
+ );
+ expect(await readArchive()).not.toContain(id);
+ }
+
+ // The report is regenerated by the debounced scheduler after the job, so poll
+ // for the bucket rather than reading the instant the job ends.
+ await expect
+ .poll(
+ async () => {
+ const snap = await readJson<Snapshot>(
+ `${ROOT}/snapshot.json`,
+ ).catch(() => null);
+ return snap?.buckets.skippedByTitleFilter ?? null;
+ },
+ { timeout: 20_000, intervals: [200, 300, 500] },
+ )
+ .toEqual(PLAIN);
+
+ const snapshot = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ // Out of the runner's work queue — the line that makes the settlement stick,
+ // since undownloadedIds is artifact-derived and an archive line would not
+ // have kept them out of it.
+ for (const id of PLAIN) expect(snapshot.undownloadedIds).not.toContain(id);
+ // And out of the RETRYABLE skip bucket: a settled video is not "we'll try
+ // again", and it must not be double-counted.
+ expect(snapshot.buckets.skippedByFilter ?? []).toEqual([]);
+ // Two videos, not four: the settled pair are metadata stubs the operator
+ // asked us not to fetch.
+ expect(snapshot.totals.videos).toBe(2);
+ expect(snapshot.totals.downloaded).toBe(2);
+});
+
+test("changing the filter un-settles every video it decided", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await download(page);
+
+ await expect
+ .poll(
+ async () => {
+ const snap = await readJson<Snapshot>(
+ `${ROOT}/snapshot.json`,
+ ).catch(() => null);
+ return snap?.buckets.skippedByTitleFilter ?? null;
+ },
+ { timeout: 20_000, intervals: [200, 300, 500] },
+ )
+ .toEqual(PLAIN);
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+
+ // Flip the filter on disk. The stored outcomes still carry the OLD signature,
+ // so nothing about them matches the new one — no migration, no sweep.
+ await writeConfig({ include: "plain" });
+ const snapshot = await refreshReport(page, before.generatedAt);
+
+ expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual([]);
+ for (const id of PLAIN) expect(snapshot.undownloadedIds).toContain(id);
+ // All four are videos again.
+ expect(snapshot.totals.videos).toBe(4);
+
+ // And the next download actually fetches them — the settlement is gone, not
+ // merely hidden from the report.
+ await download(page);
+ for (const id of PLAIN) {
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true);
+ expect(await readArchive()).toContain(id);
+ }
+});
+
+test("the form round-trips both patterns and refuses an invalid regex", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await page.goto(channelStage(CHANNEL, "configure"));
+
+ // The two fields live under Advanced.
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ const include = page.getByLabel(/^download filter: include/i);
+ const exclude = page.getByLabel(/^download filter: exclude/i);
+ await expect(include).toHaveValue("guest");
+ await expect(exclude).toHaveValue("");
+
+ // An unparseable pattern is refused by the form, because a bad pattern on
+ // disk makes the filter silently inert at download time.
+ await exclude.fill("elf(");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect(
+ page.getByText(/Download filter exclude is not a valid regular expression/),
+ ).toBeVisible({ timeout: 30_000 });
+ // Nothing was written.
+ const unchanged = await readJson<{
+ downloadFilter?: { include?: string; exclude?: string };
+ }>(`${ROOT}/config.json`);
+ expect(unchanged.downloadFilter).toEqual({ include: "guest" });
+
+ // A valid pair round-trips.
+ await page.goto(channelStage(CHANNEL, "configure"));
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await page.getByLabel(/^download filter: include/i).fill("guest|special");
+ await page.getByLabel(/^download filter: exclude/i).fill("rerun");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () => {
+ const cfg = await readJson<{
+ downloadFilter?: { include?: string; exclude?: string };
+ }>(`${ROOT}/config.json`).catch(() => null);
+ return cfg?.downloadFilter ?? null;
+ },
+ { timeout: 30_000, intervals: [250, 500] },
+ )
+ .toEqual({ include: "guest|special", exclude: "rerun" });
+
+ await page.goto(channelStage(CHANNEL, "configure"));
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await expect(page.getByLabel(/^download filter: include/i)).toHaveValue(
+ "guest|special",
+ );
+ await expect(page.getByLabel(/^download filter: exclude/i)).toHaveValue(
+ "rerun",
+ );
+
+ // Clearing both clears the stored key (CHANNEL_FORM_FIELDS), which is what
+ // makes "no filter" reachable from the form at all.
+ await page.getByLabel(/^download filter: include/i).fill("");
+ await page.getByLabel(/^download filter: exclude/i).fill("");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () => {
+ const cfg = await readJson<{ downloadFilter?: unknown }>(
+ `${ROOT}/config.json`,
+ ).catch(() => null);
+ return cfg ? "downloadFilter" in cfg : null;
+ },
+ { timeout: 30_000, intervals: [250, 500] },
+ )
+ .toBe(false);
+});
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4047,3 +4047,54 @@ never pulls `reader-fs.ts` into a client chunk.
- **The full editor suite took 38.8 min on `838da4a`** (533/533, nothing flaked) against the
23-min idle-machine baseline; two other Opus implementers were building in sibling
worktrees at the time. Wall time is not a signal here; the pass count is.
+
+## Per-channel download filter (verified 2026-09-20, branch `feat/download-title-filter`)
+
+- **A declined video is settled by an OUTCOME, never by an archive line.** Three separate
+ mechanisms make an archive line the wrong tool, and each was checked:
+ `common/controller/channelSnapshot.ts:1178-1196` derives `undownloadedIds` from
+ ARTIFACTS on disk (`videoHasAnyArtifact`), so an archived id with no files is still
+ handed to the auto-download runner; `common/controller/verifyTranscripts.ts:47` and the
+ `missingFromArchive` bucket both read an archive line as "this was downloaded", so the
+ channel would report ~1,800 broken videos; and an archive line carries no record of WHY,
+ so changing the filter could never undo it. The outcome is
+ `status: "skipped-filtered"` (no enum change — `autoRunner.ts:1980` and
+ `runYtdlp.ts:916` already count that status as skipped) plus
+ `filter.permanent` + `filter.signature` on `DownloadOutcomeRecord`
+ (`common/lib/downloadOutcome.ts`).
+- **The signature is the expiry mechanism.** `downloadFilterSignature()` is
+ `"v1\0" + include + "\0" + exclude`, null when both are blank
+ (`common/lib/downloadFilters.ts`). `isSettledByFilter(outcome, config)` is the ONE
+ definition of settled — a permanent filter skip whose recorded signature still equals the
+ channel's current one. Editing either pattern changes the signature, so every video the
+ old filter decided is re-evaluated on the next report with no migration, no sweep and no
+ stored "generation" counter. `isVideoSettledByFilter(videoDir, config)`
+ (`common/lib/downloadOutcome-server.ts`) is its per-video-dir form, shaped like
+ `isDeferredAuthExcluded` (`common/ytdlp/runYtdlp.ts:247`) and short-circuiting on the
+ signature BEFORE the file read, so a channel with no filter pays nothing.
+- **`buckets.skippedByTitleFilter` is a short-circuit, beside the other two terminal
+ outcomes.** It sits immediately after the `failed-short-audio` `continue` in the per-video
+ loop (`channelSnapshot.ts`), and it has to: everything below classifies a video by what is
+ MISSING from its directory, and a settled video (one metadata.info.json, nothing else)
+ would otherwise land in `noTranscript` AND `noMetadata` AND the retryable
+ `skippedByFilter` simultaneously, once per non-match. It is also subtracted from
+ `totals.videos` and excluded from `undownloadedIds`.
+- **`skippedByFilter` and `skippedByTitleFilter` are different things and must stay
+ separate.** The first is RETRYABLE (skip-live: the stream ends and the video becomes
+ downloadable, so every sync tries again — the decision carries no `permanent`/`signature`).
+ The second is the operator's own "not this one". A configured filter that meets
+ `metadata === null` deliberately produces the FIRST kind: it fails closed (does not
+ download) but records a non-permanent skip, so a flaky extractor cannot settle anything.
+- **Sync counts a settled id as an ARCHIVED HIT** (`selectDownloadableUrls`,
+ `common/ytdlp/runYtdlp.ts`). `archivedHits > 0` is the paged walk's stop signal, and a
+ settled video never gets an archive line; without this a filtered channel's daily sync
+ walks every page of non-matches forever and never sees a hit on the pages the filter
+ emptied. The count is reported separately in the log so it doesn't claim they were
+ archived. `downloadPlaylistManaged`'s exclusion pass skips them too, first and cheapest.
+- **The registry order is `[titleFilter, skipLive]`** and the order is load-bearing: a
+ non-matching video that is also currently live must be settled, not classified as a
+ live-stream retry. A video that PASSES the include still falls through to skip-live.
+- **An invalid regex makes the filter inert, plus one log line** — it never fails a
+ download. The editor form is what refuses to save one
+ (`editor/app/channels/components/parseChannelForm.ts`, `new RegExp(v, "i")`), because a bad
+ pattern on disk would otherwise present as a channel that silently downloads everything.