Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 4924b96b8a4307c1da00ea9513904d0a8e1902c2
parent a495e1a2fcd423ae1bdbeb8fc1940031a442f1c1
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 00:00:52 -0400

download filter: the snapshot has a settled bucket, and sync stops on settled pages

`buckets.skippedByTitleFilter` is the settled half of the filter skip, kept
apart from the retryable `skippedByFilter` for the same reason
corrupt-full-source and failed-short-audio are kept apart from failure: it is
a terminal outcome, and everything below the short-circuit classifies a video
by what is MISSING from its directory. A settled video is missing everything,
so without the short-circuit each of ~1,800 metadata stubs would land in
noTranscript AND noMetadata AND skippedByFilter at once.

They are also out of undownloadedIds — the auto-download runner's work queue,
and artifact-derived, which is precisely why an archive line could never have
settled anything — and out of totals.videos, so a filtered channel does not
read "downloaded 40 of 1,840" forever.

selectDownloadableUrls now counts a settled id as an ARCHIVED HIT. That is the
load-bearing line for a daily sync: archivedHits > 0 is the paged walk's stop
signal, and a settled video never gets an archive line, so without this the
walk would page through every non-match every day and never see a hit on the
pages the filter emptied. The batch exclusion pass skips them too and says how
many, so the reason the run did nothing is in the log.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 38+++++++++++++++++++++++++++++++++++++-
Mcommon/views/pipeline/channelFlow.ts | 6++++++
Mcommon/views/pipeline/stageStatus.ts | 1+
Mcommon/ytdlp/runYtdlp.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----------
4 files changed, 111 insertions(+), 11 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -46,6 +46,7 @@ import { } from "../jobs/autoQueuePolicy"; import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; +import { isSettledByFilter } from "../lib/downloadFilters"; import type { Paths } from "../lib/paths"; import { extractVideoId } from "../ytdlp/runYtdlp"; import { reconcileVideoDirs } from "./reconcileVideoDirs"; @@ -177,6 +178,20 @@ export type ChannelSnapshot = { // declined as currently-live/upcoming and will be retried on a later sync. // Optional: older snapshots lack it; readers must default to []. skippedByFilter: string[]; + // Videos SETTLED by the per-channel download filter: a terminal + // "skipped-filtered" outcome whose recorded signature still equals the + // channel's current include/exclude (isSettledByFilter). Distinct from + // skippedByFilter above, which is the RETRYABLE skip — that bucket says + // "we'll try again", this one says "the operator asked us not to". + // + // These ids are short-circuited out of every other bucket, out of + // undownloadedIds, and out of totals.videos: a filtered channel accumulates + // one metadata stub per non-match (~1,800 on the channel this was built + // for), and counting them as videos would make every ratio on every page + // wrong. Change either pattern and the signature stops matching, so they + // re-appear as ordinary undownloaded videos on the next report. + // Optional: older snapshots lack it; readers must default to []. + skippedByTitleFilter: string[]; // Videos whose transcript covers only a small fraction of the video's // duration — the audio download silently truncated (yt-dlp exited "ok") so // whisper transcribed just the first few minutes. The detection threshold @@ -865,6 +880,7 @@ export async function generateChannelSnapshot( const corruptFullSource: string[] = []; const nonStandardVtt: string[] = []; const skippedByFilter: string[] = []; + const skippedByTitleFilter: string[] = []; const incompleteTranscript: string[] = []; const shortAudio: string[] = []; const autoSubsOnly: string[] = []; @@ -980,6 +996,16 @@ export async function generateChannelSnapshot( if (!excludedTruncatedIds.has(id)) shortAudio.push(id); continue; } + // SETTLED BY THE DOWNLOAD FILTER — the operator's own "not this one". + // Short-circuited here, beside the other two terminal outcomes, and for the + // same reason: everything below classifies a video by what is missing from + // its directory, and a settled video is missing everything. Without this it + // would land in noTranscript AND noMetadata AND the retryable + // skippedByFilter, three times over, ~1,800 times on a filtered channel. + if (isSettledByFilter(outcome, config)) { + skippedByTitleFilter.push(id); + continue; + } if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); // A transcribed video whose cues stop far short of its duration — the audio // download truncated silently. Threshold lives in transcriptCoverage. @@ -1177,6 +1203,7 @@ export async function generateChannelSnapshot( const undownloadedIds: string[] = []; const needsCookies: string[] = []; + const settledByFilterSet = new Set(skippedByTitleFilter); // Walked in playlist order, not sorted: this is the auto-download runner's // work queue, and the listing is newest-first. for (const url of urls) { @@ -1186,6 +1213,11 @@ export async function generateChannelSnapshot( if (!dirId || !listedIdSet.has(dirId)) continue; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; + // A settled video has no artifact and never will while the filter stands. + // This is the line that makes the settlement STICK: undownloadedIds is the + // auto-download runner's work queue, and it is derived from artifacts, so an + // archive line alone would not have kept the id out of it. + if (settledByFilterSet.has(dirId)) continue; const effective = effectiveById.get(dirId); if (effective && AUTH_RETRY_CLASSES.has(effective)) { needsCookies.push(dirId); @@ -1247,6 +1279,7 @@ export async function generateChannelSnapshot( corruptFullSource: corruptFullSource.sort(), nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), + skippedByTitleFilter: skippedByTitleFilter.sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), autoSubsOnly: autoSubsOnly.sort(), @@ -1273,7 +1306,10 @@ export async function generateChannelSnapshot( const snapshot: ChannelSnapshot = { generatedAt: new Date().toISOString(), totals: { - videos: videoDirNames.length, + // Settled videos are excluded: they are metadata stubs the operator asked + // us not to fetch, not videos this channel has. Counting them would make + // "downloaded 40 of 1,840" the permanent state of a filtered channel. + videos: videoDirNames.length - skippedByTitleFilter.length, transcribed, downloaded, }, diff --git a/common/views/pipeline/channelFlow.ts b/common/views/pipeline/channelFlow.ts @@ -390,6 +390,12 @@ export function computeChannelFlow( "Declined as currently live or upcoming; retried on a later sync.", ), ...siding( + "filtered out", + buckets.skippedByTitleFilter.length, + "diagnostics", + "Declined by this channel's download filter and settled. Change the filter to bring them back.", + ), + ...siding( "need cookies", buckets.needsCookies.length, "download", diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts @@ -38,6 +38,7 @@ export function normalizeBuckets( corruptFullSource: raw?.corruptFullSource ?? [], nonStandardVtt: raw?.nonStandardVtt ?? [], skippedByFilter: raw?.skippedByFilter ?? [], + skippedByTitleFilter: raw?.skippedByTitleFilter ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], shortAudio: raw?.shortAudio ?? [], autoSubsOnly: raw?.autoSubsOnly ?? [], diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -28,6 +28,7 @@ import { type ResolvedCookiePolicy, } from "../lib/cookiePolicy"; import { resolveEffectiveAvailability } from "../lib/availability-server"; +import { isVideoSettledByFilter } from "../lib/downloadOutcome-server"; import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability"; import { runAvailabilityCheck } from "../controller/checkAvailability"; import { writeMaybeMissing } from "../controller/maybeMissingStore"; @@ -670,10 +671,19 @@ async function downloadPlaylistManaged( const runCookiePolicy = resolveRunCookiePolicy(opts, effectiveChannelConfig); const excludedCounts = { members_only: 0, deleted: 0, private: 0 }; let deferredAuthCount = 0; + let settledByFilterCount = 0; const filteredTofetch: string[] = []; for (const url of tofetch) { const dirId = extractVideoId(url); if (dirId) { + // Settled by the channel's download filter. Checked FIRST and cheaply + // (isVideoSettledByFilter short-circuits on a channel with no filter), so + // a filtered channel's batch doesn't re-run the metadata prefetch for + // every non-match on every run just to reach the same verdict. + if (await isVideoSettledByFilter(path.join(dataDir, dirId), effectiveChannelConfig)) { + settledByFilterCount++; + continue; + } const cls = await resolveEffectiveAvailability( path.join(dataDir, dirId), ); @@ -700,6 +710,11 @@ async function downloadPlaylistManaged( `Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`, ); } + if (settledByFilterCount > 0) { + opts.onLog( + `Settled by this channel's download filter: ${settledByFilterCount} skipped (already decided). Change the include/exclude patterns in Configure to re-evaluate them.\n`, + ); + } if (deferredAuthCount > 0) { opts.onLog( `Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`, @@ -1295,30 +1310,52 @@ function fullSweepDue(opts: RunYtdlpOpts): boolean { // The per-page download filter, shared by both passes so they can never drift: // entries already in the archive are counted as hits (the paged walk's stop -// signal), defer-mode needs_auth videos are held back for the Needs-cookies -// bucket, and everything else is queued for download. +// signal), videos settled by the channel's download filter are counted as hits +// TOO, defer-mode needs_auth videos are held back for the Needs-cookies bucket, +// and everything else is queued for download. +// +// WHY A SETTLED VIDEO IS A HIT. `archivedHits > 0` is the paged walk's stop +// signal — "we have reached content we already have". A settled video is +// content we have already decided about, and it never gets an archive line (an +// archive line would mean "downloaded" to verifyTranscripts and the +// missingFromArchive bucket). Counting it as new instead would make a filtered +// channel's daily sync walk every page of ~1,800 non-matches to find nothing, +// forever: the walk would never see a hit on the pages the filter emptied. async function selectDownloadableUrls( pageUrls: ReadonlyArray<string>, archive: Awaited<ReturnType<typeof readArchive>>, dataDir: string, runCookiePolicy: ResolvedCookiePolicy, + channelConfig: Pick<ChannelConfig, "downloadFilter">, ): Promise<{ newUrls: string[]; archivedHits: number; deferredAuthCount: number; + settledCount: number; }> { const newUrls: string[] = []; let archivedHits = 0; let deferredAuthCount = 0; + let settledCount = 0; for (const url of pageUrls) { const archiveId = await archiveIdForUrl(url, dataDir); if (archiveId && archive.ids.has(archiveId)) { archivedHits++; continue; } + const dirId = extractVideoId(url); + // Settled by the channel's download filter: counted as a hit (see above), + // and reported separately so the log doesn't claim they were archived. + if ( + dirId && + (await isVideoSettledByFilter(path.join(dataDir, dirId), channelConfig)) + ) { + archivedHits++; + settledCount++; + continue; + } // Cookie mode "defer": don't re-attempt known auth-gated videos on // every sync — they wait in the Needs-cookies bucket instead. - const dirId = extractVideoId(url); if ( dirId && (await isDeferredAuthExcluded(path.join(dataDir, dirId), runCookiePolicy)) @@ -1328,7 +1365,7 @@ async function selectDownloadableUrls( } newUrls.push(url); } - return { newUrls, archivedHits, deferredAuthCount }; + return { newUrls, archivedHits, deferredAuthCount, settledCount }; } async function syncPaged(opts: RunYtdlpOpts): Promise<void> { @@ -1387,10 +1424,20 @@ async function syncPaged(opts: RunYtdlpOpts): Promise<void> { "listing", ); - const { newUrls, archivedHits, deferredAuthCount } = - await selectDownloadableUrls(pageUrls, archive, dataDir, runCookiePolicy); + const { newUrls, archivedHits, deferredAuthCount, settledCount } = + await selectDownloadableUrls( + pageUrls, + archive, + dataDir, + runCookiePolicy, + opts.channelConfig, + ); opts.onLog( - `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`, + `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` + + (settledCount + ? ` (${settledCount} settled by the download filter)` + : "") + + `.\n`, ); if (deferredAuthCount > 0) { opts.onLog( @@ -1533,10 +1580,20 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> { ); if (pageUrls.length === 0) break; - const { newUrls, archivedHits, deferredAuthCount } = - await selectDownloadableUrls(pageUrls, archive, dataDir, runCookiePolicy); + const { newUrls, archivedHits, deferredAuthCount, settledCount } = + await selectDownloadableUrls( + pageUrls, + archive, + dataDir, + runCookiePolicy, + opts.channelConfig, + ); opts.onLog( - `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`, + `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived` + + (settledCount + ? ` (${settledCount} settled by the download filter)` + : "") + + `.\n`, ); if (deferredAuthCount > 0) { opts.onLog(