commit e45b36b6deedca7a81b08928fbf7f91d90b676a8
parent 0086f0411b524e09652af94bbc2336ddfc2c6901
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 03:14:47 -0400
metadata scan: the backlog it advertises is the work it does, and it needs media
Two ways the scan and the report disagreed about the same channel.
metadataScanTargets filtered on `!scan.entries[id]` alone while the snapshot's
`metadataScan.unscanned` applied the 24 h error cooldown through
metadataScanWanted — so /operations/metadata-scan offered a Run for work the
run would then decline, and a members-only video was re-requested on every
scan forever. Both now ask metadataScanWanted, and both ask ONE "already
fetched" predicate (isVideoFetched, beside its siblings in videoStatus)
instead of two hand-written spellings of it.
And `needsMedia` is now TRUE. The scan creates no video directory, which is why
it was false — but the test on JobKindMeta is "does it open data/", and this
does: its entire target set is "listed, minus what is on disk". Against an
unmounted drive it read every downloaded video as unfetched and would
re-request the whole channel. The readdir no longer swallows the failure
either: only a genuinely absent data/ means "nothing fetched".
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
4 files changed, 47 insertions(+), 12 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -5,6 +5,7 @@ import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
import {
isVideoDownloaded,
+ isVideoFetched,
isVideoTranscribed,
readVideoFiles,
VTT_FILENAME,
@@ -1262,8 +1263,10 @@ export async function generateChannelSnapshot(
// Never fetched and never read: work for the metadata scan. Counted before
// the settlement check, because a settled video HAS been read — it is the
// scan's output, not its input.
+ // The SAME "already fetched" predicate metadataScanTargets uses, so the
+ // number this advertises is the number Run will actually fetch.
if (
- !(f?.hasMeta ?? false) &&
+ !(f ? isVideoFetched(f) : false) &&
metadataScanWanted(metadataScanStore, dirId, scanNow)
) {
metadataScanUnscanned++;
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -449,16 +449,21 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
},
// The metadata scan (ytdlp/metadataScan.ts). On the platform download queue,
// because it contends for the same thing a download does — the source's
- // patience — but `needsMedia` is FALSE and that is the whole point: it reads
- // titles, writes one channel-level file, and never opens or creates a video
- // directory. An unmounted media drive is no reason to refuse it.
+ // patience.
+ //
+ // `needsMedia` IS TRUE, despite the scan never creating a video directory,
+ // and the test in JobKindMeta is exactly why: does it open `data/`? It does —
+ // its whole target set is "listed, minus what is already on disk". Against an
+ // unmounted drive it would read every downloaded video as unfetched and
+ // re-request the entire channel. It writes no media; that is a different
+ // question from whether it READS the media dir.
"metadata-scan": {
kind: "metadata-scan",
label: "Metadata scan",
drainable: true,
replayable: true,
queueKeyStrategy: "platform",
- needsMedia: false,
+ needsMedia: true,
},
// Replayable kinds that never had a JOB_KIND_LABELS entry: label omitted so
// jobKindLabel() keeps falling back to the raw kind (unchanged behavior).
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -217,6 +217,14 @@ export function isVideoTranscribed(files: VideoFiles): boolean {
return files.hasWhisper || files.hasYtVtt;
}
+// HAS THIS VIDEO EVER BEEN FETCHED? Any artifact, or even just the metadata a
+// prefetch left behind. The metadata scan and the snapshot must answer this the
+// same way or the advertised backlog and what Run actually fetches disagree —
+// which is how a scan ends up re-requesting videos the report says are done.
+export function isVideoFetched(files: VideoFiles): boolean {
+ return isVideoDownloaded(files) || files.hasMeta;
+}
+
export function isVideoDownloaded(files: VideoFiles): boolean {
return (
files.hasWhisper ||
diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts
@@ -34,10 +34,11 @@ import {
import type { Paths } from "../lib/paths";
import { getSettings } from "../lib/settings";
import { extractVideoId } from "../lib/videoId";
-import { isVideoDownloaded, readVideoFiles } from "../lib/videoStatus";
+import { isVideoFetched, readVideoFiles } from "../lib/videoStatus";
import type { JobProgress } from "../jobs/registry";
import {
loadMetadataScan,
+ metadataScanWanted,
upsertMetadataScan,
type MetadataScanEntry,
type MetadataScanError,
@@ -156,8 +157,17 @@ async function readPlaylistIds(channelDir: string): Promise<string[]> {
return ids;
}
-// Ids that already have media/transcripts on disk. A downloaded video needs no
-// scan — its metadata.info.json is the better record, and the snapshot reads it.
+// Ids that have already been fetched. A downloaded video needs no scan — its
+// metadata.info.json is the better record, and the snapshot reads it.
+//
+// ENOENT IS NOT "NOTHING FETCHED". This used to swallow a readdir failure and
+// return an empty set, which on a relocated channel whose drive is unmounted
+// meant the scan happily re-requested the whole channel, downloaded videos
+// included. The kind declares `needsMedia: true` for exactly this reason — the
+// target set is DERIVED from data/ — so runManagedFunction refuses the job
+// before it starts, and anything that slips past that throws here rather than
+// lying. A channel that has simply never downloaded anything has no data/ dir
+// at all, which is the one case that legitimately means "nothing fetched".
async function fetchedIds(dataDir: string): Promise<Set<string>> {
const out = new Set<string>();
let names: string[];
@@ -165,12 +175,13 @@ async function fetchedIds(dataDir: string): Promise<Set<string>> {
names = (await readdir(dataDir, { withFileTypes: true }))
.filter((e) => e.isDirectory())
.map((e) => e.name);
- } catch {
- return out;
+ } catch (err) {
+ if ((err as NodeJS.ErrnoException).code === "ENOENT") return out;
+ throw err;
}
for (const name of names) {
const files = await readVideoFiles(path.join(dataDir, name));
- if (isVideoDownloaded(files) || files.hasMeta) out.add(name);
+ if (isVideoFetched(files)) out.add(name);
}
return out;
}
@@ -181,6 +192,7 @@ async function fetchedIds(dataDir: string): Promise<Set<string>> {
export async function metadataScanTargets(
paths: Paths,
slug: string,
+ now: number = Date.now(),
): Promise<string[]> {
const channelDir = path.join(paths.channelsDir, slug);
const [listed, fetched, scan] = await Promise.all([
@@ -188,7 +200,14 @@ export async function metadataScanTargets(
fetchedIds(path.join(channelDir, "data")),
loadMetadataScan(paths, slug),
]);
- return listed.filter((id) => !fetched.has(id) && !scan.entries[id]);
+ // metadataScanWanted, not `!scan.entries[id]`: it also applies the 24 h error
+ // cooldown, which is what the snapshot's advertised backlog
+ // (`metadataScan.unscanned`) uses. Without it the two disagreed — the
+ // operation page would offer a Run for work the run would not do, and a
+ // members-only video would be re-requested on every single scan.
+ return listed.filter(
+ (id) => !fetched.has(id) && metadataScanWanted(scan, id, now),
+ );
}
function urlForId(id: string, channelConfig: ChannelConfig): string {