commit 4924b96b8a4307c1da00ea9513904d0a8e1902c2
parent a495e1a2fcd423ae1bdbeb8fc1940031a442f1c1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 00:00:52 -0400
download filter: the snapshot has a settled bucket, and sync stops on settled pages
`buckets.skippedByTitleFilter` is the settled half of the filter skip, kept
apart from the retryable `skippedByFilter` for the same reason
corrupt-full-source and failed-short-audio are kept apart from failure: it is
a terminal outcome, and everything below the short-circuit classifies a video
by what is MISSING from its directory. A settled video is missing everything,
so without the short-circuit each of ~1,800 metadata stubs would land in
noTranscript AND noMetadata AND skippedByFilter at once.
They are also out of undownloadedIds — the auto-download runner's work queue,
and artifact-derived, which is precisely why an archive line could never have
settled anything — and out of totals.videos, so a filtered channel does not
read "downloaded 40 of 1,840" forever.
selectDownloadableUrls now counts a settled id as an ARCHIVED HIT. That is the
load-bearing line for a daily sync: archivedHits > 0 is the paged walk's stop
signal, and a settled video never gets an archive line, so without this the
walk would page through every non-match every day and never see a hit on the
pages the filter emptied. The batch exclusion pass skips them too and says how
many, so the reason the run did nothing is in the log.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
4 files changed, 111 insertions(+), 11 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -46,6 +46,7 @@ import {
} from "../jobs/autoQueuePolicy";
import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
+import { isSettledByFilter } from "../lib/downloadFilters";
import type { Paths } from "../lib/paths";
import { extractVideoId } from "../ytdlp/runYtdlp";
import { reconcileVideoDirs } from "./reconcileVideoDirs";
@@ -177,6 +178,20 @@ export type ChannelSnapshot = {
// declined as currently-live/upcoming and will be retried on a later sync.
// Optional: older snapshots lack it; readers must default to [].
skippedByFilter: string[];
+ // Videos SETTLED by the per-channel download filter: a terminal
+ // "skipped-filtered" outcome whose recorded signature still equals the
+ // channel's current include/exclude (isSettledByFilter). Distinct from
+ // skippedByFilter above, which is the RETRYABLE skip — that bucket says
+ // "we'll try again", this one says "the operator asked us not to".
+ //
+ // These ids are short-circuited out of every other bucket, out of
+ // undownloadedIds, and out of totals.videos: a filtered channel accumulates
+ // one metadata stub per non-match (~1,800 on the channel this was built
+ // for), and counting them as videos would make every ratio on every page
+ // wrong. Change either pattern and the signature stops matching, so they
+ // re-appear as ordinary undownloaded videos on the next report.
+ // Optional: older snapshots lack it; readers must default to [].
+ skippedByTitleFilter: string[];
// Videos whose transcript covers only a small fraction of the video's
// duration — the audio download silently truncated (yt-dlp exited "ok") so
// whisper transcribed just the first few minutes. The detection threshold
@@ -865,6 +880,7 @@ export async function generateChannelSnapshot(
const corruptFullSource: string[] = [];
const nonStandardVtt: string[] = [];
const skippedByFilter: string[] = [];
+ const skippedByTitleFilter: string[] = [];
const incompleteTranscript: string[] = [];
const shortAudio: string[] = [];
const autoSubsOnly: string[] = [];
@@ -980,6 +996,16 @@ export async function generateChannelSnapshot(
if (!excludedTruncatedIds.has(id)) shortAudio.push(id);
continue;
}
+ // SETTLED BY THE DOWNLOAD FILTER — the operator's own "not this one".
+ // Short-circuited here, beside the other two terminal outcomes, and for the
+ // same reason: everything below classifies a video by what is missing from
+ // its directory, and a settled video is missing everything. Without this it
+ // would land in noTranscript AND noMetadata AND the retryable
+ // skippedByFilter, three times over, ~1,800 times on a filtered channel.
+ if (isSettledByFilter(outcome, config)) {
+ skippedByTitleFilter.push(id);
+ continue;
+ }
if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id);
// A transcribed video whose cues stop far short of its duration — the audio
// download truncated silently. Threshold lives in transcriptCoverage.
@@ -1177,6 +1203,7 @@ export async function generateChannelSnapshot(
const undownloadedIds: string[] = [];
const needsCookies: string[] = [];
+ const settledByFilterSet = new Set(skippedByTitleFilter);
// Walked in playlist order, not sorted: this is the auto-download runner's
// work queue, and the listing is newest-first.
for (const url of urls) {
@@ -1186,6 +1213,11 @@ export async function generateChannelSnapshot(
if (!dirId || !listedIdSet.has(dirId)) continue;
const f = filesById.get(dirId);
if (f && videoHasAnyArtifact(f)) continue;
+ // A settled video has no artifact and never will while the filter stands.
+ // This is the line that makes the settlement STICK: undownloadedIds is the
+ // auto-download runner's work queue, and it is derived from artifacts, so an
+ // archive line alone would not have kept the id out of it.
+ if (settledByFilterSet.has(dirId)) continue;
const effective = effectiveById.get(dirId);
if (effective && AUTH_RETRY_CLASSES.has(effective)) {
needsCookies.push(dirId);
@@ -1247,6 +1279,7 @@ export async function generateChannelSnapshot(
corruptFullSource: corruptFullSource.sort(),
nonStandardVtt: nonStandardVtt.sort(),
skippedByFilter: skippedByFilter.sort(),
+ skippedByTitleFilter: skippedByTitleFilter.sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
autoSubsOnly: autoSubsOnly.sort(),
@@ -1273,7 +1306,10 @@ export async function generateChannelSnapshot(
const snapshot: ChannelSnapshot = {
generatedAt: new Date().toISOString(),
totals: {
- videos: videoDirNames.length,
+ // Settled videos are excluded: they are metadata stubs the operator asked
+ // us not to fetch, not videos this channel has. Counting them would make
+ // "downloaded 40 of 1,840" the permanent state of a filtered channel.
+ videos: videoDirNames.length - skippedByTitleFilter.length,
transcribed,
downloaded,
},
diff --git a/common/views/pipeline/channelFlow.ts b/common/views/pipeline/channelFlow.ts
@@ -390,6 +390,12 @@ export function computeChannelFlow(
"Declined as currently live or upcoming; retried on a later sync.",
),
...siding(
+ "filtered out",
+ buckets.skippedByTitleFilter.length,
+ "diagnostics",
+ "Declined by this channel's download filter and settled. Change the filter to bring them back.",
+ ),
+ ...siding(
"need cookies",
buckets.needsCookies.length,
"download",
diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts
@@ -38,6 +38,7 @@ export function normalizeBuckets(
corruptFullSource: raw?.corruptFullSource ?? [],
nonStandardVtt: raw?.nonStandardVtt ?? [],
skippedByFilter: raw?.skippedByFilter ?? [],
+ skippedByTitleFilter: raw?.skippedByTitleFilter ?? [],
incompleteTranscript: raw?.incompleteTranscript ?? [],
shortAudio: raw?.shortAudio ?? [],
autoSubsOnly: raw?.autoSubsOnly ?? [],
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -28,6 +28,7 @@ import {
type ResolvedCookiePolicy,
} from "../lib/cookiePolicy";
import { resolveEffectiveAvailability } from "../lib/availability-server";
+import { isVideoSettledByFilter } from "../lib/downloadOutcome-server";
import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability";
import { runAvailabilityCheck } from "../controller/checkAvailability";
import { writeMaybeMissing } from "../controller/maybeMissingStore";
@@ -670,10 +671,19 @@ async function downloadPlaylistManaged(
const runCookiePolicy = resolveRunCookiePolicy(opts, effectiveChannelConfig);
const excludedCounts = { members_only: 0, deleted: 0, private: 0 };
let deferredAuthCount = 0;
+ let settledByFilterCount = 0;
const filteredTofetch: string[] = [];
for (const url of tofetch) {
const dirId = extractVideoId(url);
if (dirId) {
+ // Settled by the channel's download filter. Checked FIRST and cheaply
+ // (isVideoSettledByFilter short-circuits on a channel with no filter), so
+ // a filtered channel's batch doesn't re-run the metadata prefetch for
+ // every non-match on every run just to reach the same verdict.
+ if (await isVideoSettledByFilter(path.join(dataDir, dirId), effectiveChannelConfig)) {
+ settledByFilterCount++;
+ continue;
+ }
const cls = await resolveEffectiveAvailability(
path.join(dataDir, dirId),
);
@@ -700,6 +710,11 @@ async function downloadPlaylistManaged(
`Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`,
);
}
+ if (settledByFilterCount > 0) {
+ opts.onLog(
+ `Settled by this channel's download filter: ${settledByFilterCount} skipped (already decided). Change the include/exclude patterns in Configure to re-evaluate them.\n`,
+ );
+ }
if (deferredAuthCount > 0) {
opts.onLog(
`Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`,
@@ -1295,30 +1310,52 @@ function fullSweepDue(opts: RunYtdlpOpts): boolean {
// The per-page download filter, shared by both passes so they can never drift:
// entries already in the archive are counted as hits (the paged walk's stop
-// signal), defer-mode needs_auth videos are held back for the Needs-cookies
-// bucket, and everything else is queued for download.
+// signal), videos settled by the channel's download filter are counted as hits
+// TOO, defer-mode needs_auth videos are held back for the Needs-cookies bucket,
+// and everything else is queued for download.
+//
+// WHY A SETTLED VIDEO IS A HIT. `archivedHits > 0` is the paged walk's stop
+// signal — "we have reached content we already have". A settled video is
+// content we have already decided about, and it never gets an archive line (an
+// archive line would mean "downloaded" to verifyTranscripts and the
+// missingFromArchive bucket). Counting it as new instead would make a filtered
+// channel's daily sync walk every page of ~1,800 non-matches to find nothing,
+// forever: the walk would never see a hit on the pages the filter emptied.
async function selectDownloadableUrls(
pageUrls: ReadonlyArray<string>,
archive: Awaited<ReturnType<typeof readArchive>>,
dataDir: string,
runCookiePolicy: ResolvedCookiePolicy,
+ channelConfig: Pick<ChannelConfig, "downloadFilter">,
): Promise<{
newUrls: string[];
archivedHits: number;
deferredAuthCount: number;
+ settledCount: number;
}> {
const newUrls: string[] = [];
let archivedHits = 0;
let deferredAuthCount = 0;
+ let settledCount = 0;
for (const url of pageUrls) {
const archiveId = await archiveIdForUrl(url, dataDir);
if (archiveId && archive.ids.has(archiveId)) {
archivedHits++;
continue;
}
+ const dirId = extractVideoId(url);
+ // Settled by the channel's download filter: counted as a hit (see above),
+ // and reported separately so the log doesn't claim they were archived.
+ if (
+ dirId &&
+ (await isVideoSettledByFilter(path.join(dataDir, dirId), channelConfig))
+ ) {
+ archivedHits++;
+ settledCount++;
+ continue;
+ }
// Cookie mode "defer": don't re-attempt known auth-gated videos on
// every sync — they wait in the Needs-cookies bucket instead.
- const dirId = extractVideoId(url);
if (
dirId &&
(await isDeferredAuthExcluded(path.join(dataDir, dirId), runCookiePolicy))
@@ -1328,7 +1365,7 @@ async function selectDownloadableUrls(
}
newUrls.push(url);
}
- return { newUrls, archivedHits, deferredAuthCount };
+ return { newUrls, archivedHits, deferredAuthCount, settledCount };
}
async function syncPaged(opts: RunYtdlpOpts): Promise<void> {
@@ -1387,10 +1424,20 @@ async function syncPaged(opts: RunYtdlpOpts): Promise<void> {
"listing",
);
- const { newUrls, archivedHits, deferredAuthCount } =
- await selectDownloadableUrls(pageUrls, archive, dataDir, runCookiePolicy);
+ const { newUrls, archivedHits, deferredAuthCount, settledCount } =
+ await selectDownloadableUrls(
+ pageUrls,
+ archive,
+ dataDir,
+ runCookiePolicy,
+ opts.channelConfig,
+ );
opts.onLog(
- `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`,
+ `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` +
+ (settledCount
+ ? ` (${settledCount} settled by the download filter)`
+ : "") +
+ `.\n`,
);
if (deferredAuthCount > 0) {
opts.onLog(
@@ -1533,10 +1580,20 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
);
if (pageUrls.length === 0) break;
- const { newUrls, archivedHits, deferredAuthCount } =
- await selectDownloadableUrls(pageUrls, archive, dataDir, runCookiePolicy);
+ const { newUrls, archivedHits, deferredAuthCount, settledCount } =
+ await selectDownloadableUrls(
+ pageUrls,
+ archive,
+ dataDir,
+ runCookiePolicy,
+ opts.channelConfig,
+ );
opts.onLog(
- `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`,
+ `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` +
+ (settledCount
+ ? ` (${settledCount} settled by the download filter)`
+ : "") +
+ `.\n`,
);
if (deferredAuthCount > 0) {
opts.onLog(