Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit e45b36b6deedca7a81b08928fbf7f91d90b676a8
parent 0086f0411b524e09652af94bbc2336ddfc2c6901
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 03:14:47 -0400

metadata scan: the backlog it advertises is the work it does, and it needs media

Two ways the scan and the report disagreed about the same channel.

metadataScanTargets filtered on `!scan.entries[id]` alone while the snapshot's
`metadataScan.unscanned` applied the 24 h error cooldown through
metadataScanWanted — so /operations/metadata-scan offered a Run for work the
run would then decline, and a members-only video was re-requested on every
scan forever. Both now ask metadataScanWanted, and both ask ONE "already
fetched" predicate (isVideoFetched, beside its siblings in videoStatus)
instead of two hand-written spellings of it.

And `needsMedia` is now TRUE. The scan creates no video directory, which is why
it was false — but the test on JobKindMeta is "does it open data/", and this
does: its entire target set is "listed, minus what is on disk". Against an
unmounted drive it read every downloaded video as unfetched and would
re-request the whole channel. The readdir no longer swallows the failure
either: only a genuinely absent data/ means "nothing fetched".

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 5++++-
Mcommon/jobs/jobKinds.ts | 13+++++++++----
Mcommon/lib/videoStatus.ts | 8++++++++
Mcommon/ytdlp/metadataScan.ts | 33++++++++++++++++++++++++++-------
4 files changed, 47 insertions(+), 12 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -5,6 +5,7 @@ import pLimit from "p-limit"; import { readArchive } from "../lib/archive"; import { isVideoDownloaded, + isVideoFetched, isVideoTranscribed, readVideoFiles, VTT_FILENAME, @@ -1262,8 +1263,10 @@ export async function generateChannelSnapshot( // Never fetched and never read: work for the metadata scan. Counted before // the settlement check, because a settled video HAS been read — it is the // scan's output, not its input. + // The SAME "already fetched" predicate metadataScanTargets uses, so the + // number this advertises is the number Run will actually fetch. if ( - !(f?.hasMeta ?? false) && + !(f ? isVideoFetched(f) : false) && metadataScanWanted(metadataScanStore, dirId, scanNow) ) { metadataScanUnscanned++; diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -449,16 +449,21 @@ const JOB_KINDS: Record<string, JobKindMeta> = { }, // The metadata scan (ytdlp/metadataScan.ts). On the platform download queue, // because it contends for the same thing a download does — the source's - // patience — but `needsMedia` is FALSE and that is the whole point: it reads - // titles, writes one channel-level file, and never opens or creates a video - // directory. An unmounted media drive is no reason to refuse it. + // patience. + // + // `needsMedia` IS TRUE, despite the scan never creating a video directory, + // and the test in JobKindMeta is exactly why: does it open `data/`? It does — + // its whole target set is "listed, minus what is already on disk". Against an + // unmounted drive it would read every downloaded video as unfetched and + // re-request the entire channel. It writes no media; that is a different + // question from whether it READS the media dir. "metadata-scan": { kind: "metadata-scan", label: "Metadata scan", drainable: true, replayable: true, queueKeyStrategy: "platform", - needsMedia: false, + needsMedia: true, }, // Replayable kinds that never had a JOB_KIND_LABELS entry: label omitted so // jobKindLabel() keeps falling back to the raw kind (unchanged behavior). diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -217,6 +217,14 @@ export function isVideoTranscribed(files: VideoFiles): boolean { return files.hasWhisper || files.hasYtVtt; } +// HAS THIS VIDEO EVER BEEN FETCHED? Any artifact, or even just the metadata a +// prefetch left behind. The metadata scan and the snapshot must answer this the +// same way or the advertised backlog and what Run actually fetches disagree — +// which is how a scan ends up re-requesting videos the report says are done. +export function isVideoFetched(files: VideoFiles): boolean { + return isVideoDownloaded(files) || files.hasMeta; +} + export function isVideoDownloaded(files: VideoFiles): boolean { return ( files.hasWhisper || diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts @@ -34,10 +34,11 @@ import { import type { Paths } from "../lib/paths"; import { getSettings } from "../lib/settings"; import { extractVideoId } from "../lib/videoId"; -import { isVideoDownloaded, readVideoFiles } from "../lib/videoStatus"; +import { isVideoFetched, readVideoFiles } from "../lib/videoStatus"; import type { JobProgress } from "../jobs/registry"; import { loadMetadataScan, + metadataScanWanted, upsertMetadataScan, type MetadataScanEntry, type MetadataScanError, @@ -156,8 +157,17 @@ async function readPlaylistIds(channelDir: string): Promise<string[]> { return ids; } -// Ids that already have media/transcripts on disk. A downloaded video needs no -// scan — its metadata.info.json is the better record, and the snapshot reads it. +// Ids that have already been fetched. A downloaded video needs no scan — its +// metadata.info.json is the better record, and the snapshot reads it. +// +// ENOENT IS NOT "NOTHING FETCHED". This used to swallow a readdir failure and +// return an empty set, which on a relocated channel whose drive is unmounted +// meant the scan happily re-requested the whole channel, downloaded videos +// included. The kind declares `needsMedia: true` for exactly this reason — the +// target set is DERIVED from data/ — so runManagedFunction refuses the job +// before it starts, and anything that slips past that throws here rather than +// lying. A channel that has simply never downloaded anything has no data/ dir +// at all, which is the one case that legitimately means "nothing fetched". async function fetchedIds(dataDir: string): Promise<Set<string>> { const out = new Set<string>(); let names: string[]; @@ -165,12 +175,13 @@ async function fetchedIds(dataDir: string): Promise<Set<string>> { names = (await readdir(dataDir, { withFileTypes: true })) .filter((e) => e.isDirectory()) .map((e) => e.name); - } catch { - return out; + } catch (err) { + if ((err as NodeJS.ErrnoException).code === "ENOENT") return out; + throw err; } for (const name of names) { const files = await readVideoFiles(path.join(dataDir, name)); - if (isVideoDownloaded(files) || files.hasMeta) out.add(name); + if (isVideoFetched(files)) out.add(name); } return out; } @@ -181,6 +192,7 @@ async function fetchedIds(dataDir: string): Promise<Set<string>> { export async function metadataScanTargets( paths: Paths, slug: string, + now: number = Date.now(), ): Promise<string[]> { const channelDir = path.join(paths.channelsDir, slug); const [listed, fetched, scan] = await Promise.all([ @@ -188,7 +200,14 @@ export async function metadataScanTargets( fetchedIds(path.join(channelDir, "data")), loadMetadataScan(paths, slug), ]); - return listed.filter((id) => !fetched.has(id) && !scan.entries[id]); + // metadataScanWanted, not `!scan.entries[id]`: it also applies the 24 h error + // cooldown, which is what the snapshot's advertised backlog + // (`metadataScan.unscanned`) uses. Without it the two disagreed — the + // operation page would offer a Run for work the run would not do, and a + // members-only video would be re-requested on every single scan. + return listed.filter( + (id) => !fetched.has(id) && metadataScanWanted(scan, id, now), + ); } function urlForId(id: string, channelConfig: ChannelConfig): string {