commit 6e75b2cb001dafa18a2b20e42e2a26cee12e300f
parent dbc7fdf0e18df729041e93f6f89668a2240f4c46
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 01:45:08 -0400
download lane: the metadata scan is dispatched, not waited for
Nothing auto-dispatched the scan, so a filtered channel's backlog sat there
until an operator pressed Run — and meanwhile every unscanned non-match was
prefetched, rejected and discarded one yt-dlp invocation at a time against a
source whose patience is the scarce resource. One batch answers the whole
channel, so the scan is not another kind of work competing with downloads: it
decides what the downloads ARE, and it goes first.
METADATA_SCAN_OPERATION declares runner: "download". The runner's next() runs
a pre-pick before any video pick — pickMetadataScanChannel, pure and exported,
taking the first channel in priority order with a COMPILING filter, a non-zero
snapshot.metadataScan.unscanned, no scan already done by this runner, and a
platform that is neither busy nor cooling down. The unit is channel-scoped: a
synthetic `metadata-scan <slug>` key (a space, which no video id contains),
leaf id `metadata-scan`, and an EMPTY path so no node's maxWorkers moves for
it. It holds a platform slot exactly as a download unit does, so a scan and a
download never hit one source at once.
The job itself moves to common/controller/metadataScanJob.ts so both
dispatchers use one definition — kind, queue key, replay spec, and the shared
per-platform cooldown. The editor action keeps what a CLICK is owed: the
cooldown answered as a sentence rather than a queued job, and its three
revalidations, now passed as afterRun.
ONE SCAN PER CHANNEL PER RUNNER. A scan that left the backlog above zero was
stopped by something; re-offering it on the next three-second tick would
hammer the source that just refused us.
pauseLaneFor now answers "download" for the scan, because pauseLaneFor asks
`runner` first. The old exemption's reasoning is intact and now belongs to the
Run button, which checks the platform cooldown and no gate at all: you can
still scan while downloads are paused, which is exactly when you want to
decide what the lane should fetch when it resumes. FACTS amended in this
commit.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
11 files changed, 681 insertions(+), 44 deletions(-)
diff --git a/common/controller/autoRunner.test.ts b/common/controller/autoRunner.test.ts
@@ -7,6 +7,7 @@ import {
focusHoldLine,
laneDispatchRoot,
makeFocusHoldReporter,
+ pickMetadataScanChannel,
priorityContextFor,
resetPriorityContextForTest,
} from "./autoRunner";
@@ -21,6 +22,7 @@ import { LANES } from "../lib/autoQueueTypes";
import type {
AutoQueueGroup,
AutoQueuePolicy,
+ ChannelWork,
} from "../jobs/autoQueuePolicy";
import type { Paths } from "../lib/paths";
import type { SiteSettings } from "../lib/settings";
@@ -331,3 +333,114 @@ test("the context is cached on the document, and a changed document re-resolves"
"only-here",
]);
});
+
+// ---------------------------------------------------------------------------
+// The download lane's metadata-scan pre-pick.
+//
+// It runs before every video pick on the lane, so its rule is the one thing
+// standing between a filtered channel and 1,800 prefetch-reject-discard cycles
+// against a source that is counting them.
+
+function work(over: Partial<ChannelWork> & { slug: string }): ChannelWork {
+ return { platform: "youtube", buckets: {}, ...over };
+}
+
+const noSkip = {
+ scanned: new Set<string>(),
+ platformSkip: new Set<string>(),
+ platformOf: () => "youtube",
+};
+
+test("a filtered channel with an unscanned backlog is the pick", () => {
+ const pick = pickMetadataScanChannel(
+ [
+ work({ slug: "plain", scanUnscanned: 500, filtered: false }),
+ work({ slug: "filtered", scanUnscanned: 12, filtered: true }),
+ ],
+ noSkip,
+ );
+ assert.deepEqual(pick, { slug: "filtered", targets: 12 });
+});
+
+test("no filter, no scan — however big the backlog", () => {
+ // The scan's whole output is a verdict the filter makes. Without a filter it
+ // would fetch a title per listed video, decide nothing, and spend the
+ // source's patience doing it.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "plain", scanUnscanned: 9999, filtered: false })],
+ noSkip,
+ ),
+ null,
+ );
+});
+
+test("an empty backlog is not work", () => {
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 0, filtered: true })],
+ noSkip,
+ ),
+ null,
+ );
+ // And a projection with no scan fields at all — every lane but this one —
+ // offers nothing rather than throwing or defaulting to "yes".
+ assert.equal(
+ pickMetadataScanChannel([work({ slug: "filtered" })], noSkip),
+ null,
+ );
+});
+
+test("once per runner: a channel already scanned this run is skipped", () => {
+ // A scan that left the backlog above zero was stopped by something — a soft
+ // block, a cooldown, a batch that ended early. Re-offering it three seconds
+ // later would hammer the source that just refused us.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 40, filtered: true })],
+ { ...noSkip, scanned: new Set(["filtered"]) },
+ ),
+ null,
+ );
+});
+
+test("a busy or cooling platform offers no scan either", () => {
+ // The same gate the downloads honour: the scan is one more thing asking that
+ // source for titles, and the per-platform cap is one at a time.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 40, filtered: true })],
+ { ...noSkip, platformSkip: new Set(["youtube"]) },
+ ),
+ null,
+ );
+ // A DIFFERENT platform is unaffected — the gate is per source, not global.
+ assert.deepEqual(
+ pickMetadataScanChannel(
+ [
+ work({ slug: "yt", scanUnscanned: 40, filtered: true }),
+ work({ slug: "od", scanUnscanned: 7, filtered: true, platform: "odysee" }),
+ ],
+ {
+ scanned: new Set<string>(),
+ platformSkip: new Set(["youtube"]),
+ platformOf: (slug) => (slug === "od" ? "odysee" : "youtube"),
+ },
+ ),
+ { slug: "od", targets: 7 },
+ );
+});
+
+test("the first candidate in list order wins, because that order is the priority", () => {
+ assert.deepEqual(
+ pickMetadataScanChannel(
+ [
+ work({ slug: "first", scanUnscanned: 1, filtered: true }),
+ work({ slug: "second", scanUnscanned: 900, filtered: true }),
+ ],
+ noSkip,
+ ),
+ { slug: "first", targets: 1 },
+ );
+});
+
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -7,6 +7,8 @@ import { getSettings, type SiteSettings } from "../lib/settings";
import { diskGate } from "../lib/diskSpace";
import { formatBytes } from "../lib/format";
import { detectPlatform } from "../lib/platform";
+import { compileDownloadFilter } from "../lib/downloadFilters";
+import { runMetadataScanJob } from "./metadataScanJob";
import { getWorkerPool } from "../jobs/workerPool";
import { getRegistry } from "../jobs/registry";
import { runManagedFunction } from "../jobs/streamCommand";
@@ -160,6 +162,14 @@ const CHANNEL_LIST_TTL_MS = 30_000;
// grow it unbounded. FIFO eviction; the snapshot catches up within ~1s anyway.
const COMPLETED_CAP = 5000;
+// THE DOWNLOAD LANE'S CHANNEL-SCOPED UNIT. A metadata scan has no video id, so
+// it gets a synthetic one under a prefix nothing else can produce (a video id
+// never contains a space) and a leaf id that exists in no tree — the pick is
+// not made by `selectNextWork` and never reaches one. Both are exported so the
+// console and the tests name the same strings the runner writes.
+export const METADATA_SCAN_UNIT_PREFIX = "metadata-scan ";
+export const METADATA_SCAN_LEAF_ID = "metadata-scan";
+
// --- Live status (for /api/auto-queue/status), per kind --------------------
export type AutoRunnerInFlight = {
@@ -167,6 +177,13 @@ export type AutoRunnerInFlight = {
leafId: string;
channelSlug: string;
startedAt: number;
+ // A SENTENCE FOR A UNIT THAT IS NOT A VIDEO. `videoId` is the map key and
+ // therefore always set, but the download lane's metadata-scan unit is
+ // channel-scoped: its key is a synthetic `metadata-scan <slug>` that names no
+ // video directory. A console must not link it as one, so a unit carrying a
+ // note renders the note instead of a video link. Absent for every ordinary
+ // unit, which is how the surfaces stay unchanged.
+ note?: string;
};
// Why the runner is up but dispatching nothing. Every one of these was already
@@ -675,7 +692,21 @@ async function buildChannelWork(
for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
}
}
- channels.push({ slug, platform, buckets, operations: ops });
+ // THE CHANNEL-SCOPED HALF, off the snapshot already read. The download
+ // lane's pre-pick uses these two to decide whether to run a metadata scan
+ // for this channel before it downloads anything from it; every other lane
+ // ignores them. Both come out of `snap`/`config` with no extra I/O — the
+ // scan backlog is a number the snapshot already publishes, and the filter
+ // is the config listChannelConfigs already parsed.
+ const config = meta[i].config;
+ channels.push({
+ slug,
+ platform,
+ buckets,
+ operations: ops,
+ scanUnscanned: snap.metadataScan?.unscanned ?? 0,
+ filtered: compileDownloadFilter(config.downloadFilter) !== null,
+ });
}
return { channels, owner };
}
@@ -1013,7 +1044,59 @@ function eligibleSlots(): number {
// One unit of work selected by the policy, carried from the runPool `next`
// (selection + reservation) into `run` (execution + release).
-type Picked = { pick: WorkPick; channelSlug: string; unitPlatform: string | null };
+// WHICH CHANNEL GETS A METADATA SCAN THIS TICK, or none — the download lane's
+// pre-pick, as a pure function over the projection the runner already built.
+//
+// Pure because the rule is the whole feature and everything else about it is
+// plumbing. Three conditions, and each one is a bug if it goes:
+//
+// - A FILTER. Without one a scan settles nothing: it would fetch a title per
+// listed video, change no verdict, and spend the source's patience to do
+// it. `filtered` is "the channel's downloadFilter compiles", which is the
+// same predicate `settledByTitleFilterIds` opens the store for.
+// - A BACKLOG. `scanUnscanned` is the snapshot's own `metadataScan.unscanned`,
+// defined to mean exactly what `metadataScanTargets()` will fetch, so a
+// channel offered here is a channel the scan has work for.
+// - ONCE PER RUNNER. A channel whose scan left the backlog above zero — a
+// soft block, a cooldown, a batch that stopped early — is not retried by
+// this runner. Re-offering it on the next three-second tick would hammer
+// the source that just refused us; the operator's Run and the next restart
+// both still work.
+//
+// The platform gate is the same one the downloads honour: a busy or cooling
+// platform offers no scan either, because a scan is one more thing asking that
+// source for titles.
+//
+// FIRST IN `channels` ORDER, which is `metaCache` order, which is the priority
+// order every other pick on this lane already uses.
+export function pickMetadataScanChannel(
+ channels: ReadonlyArray<ChannelWork>,
+ opts: {
+ scanned: ReadonlySet<string>;
+ platformSkip: ReadonlySet<string>;
+ platformOf: (slug: string) => string;
+ },
+): { slug: string; targets: number } | null {
+ for (const c of channels) {
+ const targets = c.scanUnscanned ?? 0;
+ if (!c.filtered || targets <= 0) continue;
+ if (opts.scanned.has(c.slug)) continue;
+ if (opts.platformSkip.has(opts.platformOf(c.slug))) continue;
+ return { slug: c.slug, targets };
+ }
+ return null;
+}
+
+type Picked = {
+ pick: WorkPick;
+ channelSlug: string;
+ unitPlatform: string | null;
+ // THE CHANNEL-SCOPED UNIT. Set only by the download lane's metadata-scan
+ // pre-pick: `pick.videoId` is then a synthetic key naming no video and
+ // `pick.path` is empty, so no node's active count moves for it. `run()`
+ // branches on this and nothing else.
+ scan?: { slug: string; targets: number };
+};
async function runLoop(
kind: AutoQueueKind,
@@ -1111,6 +1194,13 @@ async function runLoop(
// "unknown" for unrecognized hosts). Unused for transcription.
const platformInFlight = new Map<string, number>();
const PER_PLATFORM_CAP = 1;
+ // Channels this runner has already scanned (or tried to). See the pre-pick in
+ // next() for why a scan is never retried inside one runner's lifetime.
+ const scannedThisRun = new Set<string>();
+ // Platforms this tick may not touch — busy or cooling down. Hoisted out of the
+ // download branch because the metadata-scan pre-pick honours it too: a
+ // platform in a 429 cooldown must not be asked for 1,800 titles either.
+ const platformSkip = new Set<string>();
const platformKey = (slug: string, slugToPlatform: Map<string, string>) =>
slugToPlatform.get(slug) ?? "unknown";
@@ -1489,6 +1579,7 @@ async function runLoop(
// download (busy) OR is in a rate-limit/network backoff window (cooling
// down), so a busy/throttled platform yields to the next-priority free one.
let anyCooling = false;
+ platformSkip.clear();
if (kind === "download") {
// Merge in any cooldown a manual sync/import wrote to the shared state
// (read-modify-write from outside the runner) since our last persist.
@@ -1504,29 +1595,86 @@ async function runLoop(
} catch {
// Best-effort: a transient read failure just skips this iteration's merge.
}
- const skip = new Set<string>();
for (const [pf, n] of platformInFlight) {
- if (n >= PER_PLATFORM_CAP) skip.add(pf);
+ if (n >= PER_PLATFORM_CAP) platformSkip.add(pf);
}
for (const pf of Object.keys(kindState.platformBackoff)) {
if (isCoolingDown(kindState.platformBackoff, pf, now)) {
- skip.add(pf);
+ platformSkip.add(pf);
// A platform skipped for a cooldown is a different answer to "why is
// it idle?" than one skipped for being busy, so the two are tracked
// apart rather than both reading as "capped".
anyCooling = true;
}
}
- if (skip.size > 0) {
+ if (platformSkip.size > 0) {
for (const leafId of Object.keys(pending)) {
pending[leafId] = pending[leafId].filter((id) => {
const slug = owner.get(id);
- return !(slug && skip.has(platformKey(slug, slugToPlatform)));
+ return !(slug && platformSkip.has(platformKey(slug, slugToPlatform)));
});
}
}
}
+ // ── THE SCAN COMES BEFORE THE DOWNLOADS ────────────────────────────────
+ //
+ // A filtered channel with unscanned listed videos is a channel whose
+ // download queue is WRONG, not merely incomplete: every non-matching video
+ // in it will be prefetched, rejected and discarded one at a time, at one
+ // yt-dlp invocation each, against a source whose patience is the scarce
+ // resource. The scan answers the same question for the whole channel in one
+ // batch. So it is not another kind of work competing with downloads — it is
+ // the thing that decides what the downloads ARE, and it goes first.
+ //
+ // ONE PER CHANNEL PER RUNNER, and never a second while one is in flight.
+ // `scannedThisRun` is the session's memory: a channel whose scan left the
+ // backlog above zero (a soft block, a cooldown, a batch that stopped early)
+ // is NOT retried by this runner — the operator's Run button and the next
+ // restart both still work, and re-offering it on the next three-second tick
+ // would hammer the very source that just refused us.
+ //
+ // It takes a platform slot exactly as a download unit does, so a scan and a
+ // download never hit one source at once, and it inherits the platform
+ // cooldown for free: a platform in `platformSkip` offers no scan either.
+ if (kind === "download") {
+ const scanCandidate = pickMetadataScanChannel(channels, {
+ scanned: scannedThisRun,
+ platformSkip,
+ platformOf: (slug) => platformKey(slug, slugToPlatform),
+ });
+ if (scanCandidate) {
+ const { slug, targets } = scanCandidate;
+ const unitPlatform = platformKey(slug, slugToPlatform);
+ scannedThisRun.add(slug);
+ const videoId = `${METADATA_SCAN_UNIT_PREFIX}${slug}`;
+ const note = `scanning ${targets} listed video${
+ targets === 1 ? "" : "s"
+ } for ${slug}`;
+ live.idleReason = null;
+ // No `path`, so no node's active count moves: the scan is not a leaf's
+ // work and must not consume a leaf's or a group's maxWorkers.
+ live.inFlight.set(videoId, {
+ videoId,
+ leafId: METADATA_SCAN_LEAF_ID,
+ channelSlug: slug,
+ startedAt: Date.now(),
+ note,
+ });
+ platformInFlight.set(
+ unitPlatform,
+ (platformInFlight.get(unitPlatform) ?? 0) + 1,
+ );
+ onLog(`Auto-download: ${note}.`);
+ return {
+ pick: { leafId: METADATA_SCAN_LEAF_ID, videoId, path: [] },
+ channelSlug: slug,
+ unitPlatform,
+ scan: { slug, targets },
+ };
+ }
+ }
+
const pick = selectNextWork(root, pending, runtime, live.active);
if (!pick) {
// Attribute the idleness. Nothing pending at all is a different situation
@@ -1682,6 +1830,17 @@ async function runLoop(
result = { outcome: "skipped" };
return;
}
+ if (picked.scan) {
+ result = await launchMetadataScan({
+ paths,
+ slug: picked.scan.slug,
+ targets: picked.scan.targets,
+ tracker,
+ onLog,
+ onChildJob: (jid) => childJobIds.set(pick.videoId, jid),
+ });
+ return;
+ }
result = isOperationLane(kind)
? await runOperationPick(picked, runSignal)
: await launchUnit({
@@ -1776,6 +1935,55 @@ async function runLoop(
);
}
+// ONE METADATA SCAN, as a child job on the channel's platform download queue.
+//
+// The SAME job the operator's Run button dispatches — kind, queue key, replay
+// spec and the shared per-platform cooldown all come from
+// controller/metadataScanJob.ts — so the registry serializes it against a
+// manual sync on that platform exactly as it serializes a download unit, and
+// there is one definition of what a scan IS.
+//
+// `background: true` for the same reason a download unit is: a clicked action
+// preempts queued units without interrupting a running one.
+//
+// A SCAN IS NEVER A FAILURE FOR BACKOFF PURPOSES. runMetadataScan does not
+// throw on a rate limit — it stops, records the shared cooldown through its own
+// onPlatformBackoff, and keeps what it flushed. So this returns "skipped" or
+// "transcribed" and never a `failureClass`: the cooldown is already recorded by
+// the time we get here, and a second backoff entry from the runner would
+// double-count one refusal.
+async function launchMetadataScan(args: {
+ paths: Paths;
+ slug: string;
+ targets: number;
+ tracker: ReturnType<typeof makeTaskTracker>;
+ onLog: (line: string) => void;
+ onChildJob?: (jobId: string) => void;
+}): Promise<UnitResult> {
+ const config = await readChannelConfig(args.paths, args.slug);
+ if (!config) return { outcome: "skipped" };
+ const res = await runMetadataScanJob({
+ paths: args.paths,
+ slug: args.slug,
+ channelConfig: config,
+ background: true,
+ });
+ if (!res.ok) {
+ args.onLog(
+ `Auto-download: could not enqueue the metadata scan for ${args.slug}: ${res.error}`,
+ );
+ return { outcome: "skipped" };
+ }
+ args.onChildJob?.(res.jobId);
+ const term = await res.done;
+ if (term.status === "cancelled") return { outcome: "skipped" };
+ if (term.status === "failed") {
+ args.onLog(`Auto-download: the metadata scan for ${args.slug} failed.`);
+ return { outcome: "skipped" };
+ }
+ return { outcome: "transcribed" };
+}
+
type LaunchArgs = {
kind: AutoQueueKind;
paths: Paths;
diff --git a/common/controller/metadataScanJob.ts b/common/controller/metadataScanJob.ts
@@ -0,0 +1,80 @@
+// THE METADATA SCAN AS A JOB — one definition, two dispatchers.
+//
+// The scan itself is `ytdlp/metadataScan.ts`, which is pure work: it takes a
+// channel and a config and reads titles. What it is NOT is a job — the kind,
+// the queue key, the replay spec and the platform cooldown that surrounds it
+// all lived in the editor's server action, which is the one place the auto
+// runner cannot reach.
+//
+// So the JOB moves here and the editor action keeps what is genuinely its
+// own: the cooldown pre-check it answers to a click with (a returned
+// `{ ok: false, info: true }` sentence, not a queued job nobody watches) and
+// its three `revalidatePath` calls. The runner passes neither. What both get
+// from this module is identical: the same `kind`, the same per-platform queue
+// key, the same replay spec, and the same `onPlatformBackoff` wiring into the
+// SHARED cooldown the download lane already honours — which is what stops the
+// runner and a manual sync from arguing about whether YouTube is angry.
+//
+// `needsMedia` comes from the kind (jobs/jobKinds.ts declares it TRUE for this
+// one, because the scan's target set is "listed, minus what is already on
+// disk"), so `runManagedFunction` refuses the job against an unmounted drive
+// for either caller with nothing said here.
+
+import { detectPlatform } from "../lib/platform";
+import { downloadQueueKey, resolveQueueKey } from "../lib/queueKeys";
+import { recordDownloadBackoff } from "../jobs/downloadBackoff";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "../jobs/streamCommand";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import { runMetadataScan } from "../ytdlp/metadataScan";
+
+export const METADATA_SCAN_JOB_KIND = "metadata-scan";
+
+export type MetadataScanJobOpts = {
+ paths: Paths;
+ slug: string;
+ channelConfig: ChannelConfig;
+ // Per-run override of the platform queue key (the editor threads one through
+ // from the pipeline action's own queue plumbing).
+ queueKey?: string;
+ // Run in the background so a clicked action preempts it in the queue. The
+ // runner passes true for the same reason its download units do.
+ background?: boolean;
+ // Called after a successful scan, inside the job. The editor revalidates its
+ // three paths here; the runner asks for a snapshot regen instead.
+ afterRun?: () => void | Promise<void>;
+};
+
+export async function runMetadataScanJob(
+ opts: MetadataScanJobOpts,
+): Promise<StreamActionResult> {
+ const { paths, slug, channelConfig } = opts;
+ const platform = detectPlatform(channelConfig.url) ?? "unknown";
+ return runManagedFunction({
+ kind: METADATA_SCAN_JOB_KIND,
+ queueKey: resolveQueueKey(downloadQueueKey(channelConfig), opts.queueKey),
+ paths,
+ channelSlug: slug,
+ ...(opts.background ? { background: true } : {}),
+ spec: {
+ kind: METADATA_SCAN_JOB_KIND,
+ slug,
+ params: { queueKey: opts.queueKey },
+ },
+ fn: async (onLog, signal, setProgress) => {
+ await runMetadataScan({
+ paths,
+ channelSlug: slug,
+ channelConfig,
+ onLog,
+ signal,
+ setProgress,
+ onPlatformBackoff: () => recordDownloadBackoff(platform, paths),
+ });
+ await opts.afterRun?.();
+ },
+ });
+}
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -303,6 +303,19 @@ export type ChannelWork = {
// Optional: every projection written before operations existed omits it, and
// a leaf naming an operation simply finds nothing.
operations?: Record<string, string[]>;
+ // THE CHANNEL-SCOPED WORK A LEAF CANNOT NAME. The metadata scan is a fact
+ // about the CHANNEL, not about a video — there is no id to put in a bucket,
+ // because the whole point of the scan is that these videos have no directory
+ // — so the download lane's pre-pick reads it off the projection here rather
+ // than through `pick()`. Both optional: a projection that omits them offers
+ // no scan, which is what every caller but the download runner wants.
+ //
+ // `scanUnscanned` is the snapshot's own `metadataScan.unscanned`, which is
+ // defined to mean exactly what metadataScanTargets() will fetch.
+ // `filtered` is "this channel has a download filter that compiles" — without
+ // one a scan settles nothing and is pure cost against the source.
+ scanUnscanned?: number;
+ filtered?: boolean;
};
// `retainLeaves(pending, root, "buckets" | "operations")` USED TO LIVE HERE, and
diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts
@@ -1245,10 +1245,16 @@ test("`runner` names the auto-queue runner, and only for the two that have one",
const runners = new Map(operationCatalog().map((o) => [o.id, o.runner]));
assert.equal(runners.get("download"), "download");
assert.equal(runners.get("transcription"), "transcription");
+ // THE THIRD ONE IS CHANNEL-SCOPED, and that is the point of it being here
+ // rather than inferred: the download runner dispatches the metadata scan as a
+ // channel-scoped unit before it picks any video, so the scan's console — and
+ // its pause — are the download lane's.
+ assert.equal(runners.get("metadata-scan"), "download");
// Everything the sweep dispatches must leave it unset — a backfill kind with
// a runner would render a runner console over a lane no runner feeds.
+ const dispatched = new Set(["download", "transcription", "metadata-scan"]);
for (const op of operationCatalog()) {
- if (op.id === "download" || op.id === "transcription") continue;
+ if (dispatched.has(op.id)) continue;
assert.equal(op.runner, undefined, `${op.id} declares a runner`);
}
// Sync included, and it is the interesting one: the sync scheduler's
diff --git a/common/lib/operations.ts b/common/lib/operations.ts
@@ -1360,10 +1360,23 @@ export const SYNC_OPERATION: OperationDescriptor = {
// download: it is one metadata request per listed video, and the source counts
// them the same way it counts a download's.
//
-// NO `runner`, deliberately: nothing auto-dispatches this yet. The operator
-// presses Run. The obvious follow-up is for the download lane to run it for a
-// channel whose filter has unscanned listed videos, which is exactly the
-// backlog this declares — see plans/FACTS.md.
+// THE DOWNLOAD LANE DISPATCHES IT, and that is what `runner` says. The runner
+// checks, before every pick, whether a filtered channel has unscanned listed
+// videos — exactly the backlog this entry declares — and runs the scan for it
+// as one channel-scoped unit on the platform download queue, holding the same
+// per-platform slot a download unit would.
+//
+// It goes FIRST because it decides what the downloads are. Every unscanned
+// non-match on a filtered channel is otherwise prefetched, rejected and
+// discarded one yt-dlp invocation at a time, against a source whose patience is
+// the scarce resource; one batch answers the whole channel.
+//
+// WHAT NAMING THE RUNNER ALSO CHANGES: `pauseLaneFor` asks `runner` first, so
+// this operation's console is the download lane's and the download pause now
+// holds the AUTO-dispatch. The operator's own Run button is unaffected — it
+// checks the platform cooldown and nothing else — which keeps the original
+// point of the exemption intact: you can still scan, while downloads are
+// paused, precisely to decide what the lane should fetch when it resumes.
export const METADATA_SCAN_OPERATION: OperationDescriptor = {
id: "metadata-scan",
label: "Metadata scan",
@@ -1375,6 +1388,7 @@ export const METADATA_SCAN_OPERATION: OperationDescriptor = {
dispatch: "external",
scope: "channel",
trigger: "backlog",
+ runner: "download",
};
export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [
diff --git a/common/lib/pauseGates.test.ts b/common/lib/pauseGates.test.ts
@@ -167,11 +167,14 @@ test("pauseLaneFor answers for every catalog id", () => {
// Catalogued, and deliberately gateless: the sync scheduler's own `enabled`
// is its switch, and its queue key is neither sweep's.
sync: null,
- // Catalogued and gateless for a different reason: the metadata scan rides
- // the platform download queue but is NOT held by the download gate. It
- // fetches no media, and the operator runs it precisely to decide what a
- // paused download lane should fetch when it resumes.
- "metadata-scan": null,
+ // THE DOWNLOAD LANE'S, because the download runner is what dispatches it.
+ // It used to be null, on the reasoning that a scan fetches no media and the
+ // operator runs it precisely to decide what a paused lane should fetch when
+ // it resumes. That reasoning is intact and now belongs to the RUN BUTTON,
+ // which checks the platform cooldown and no gate at all: pausing downloads
+ // stops the lane from auto-dispatching a scan, and stops nothing the
+ // operator does by hand.
+ "metadata-scan": "download",
};
const ids = operationCatalog().map((o) => o.id);
assert.equal(ids.length, 8);
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -25,7 +25,7 @@ import {
import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
-import { runMetadataScan } from "yt-dlp-transcript-common/ytdlp/metadataScan";
+import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates";
import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy";
@@ -351,22 +351,17 @@ export async function runMetadataScanAction(
`The metadata scan will run once the cooldown lapses.`,
};
}
- return runManagedFunction({
- kind: "metadata-scan",
- queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey),
+ // THE JOB ITSELF LIVES IN common/controller/metadataScanJob.ts, because the
+ // download lane's runner dispatches the same one and cannot reach a server
+ // action. What stays here is what a CLICK is owed and a runner is not: the
+ // cooldown answered as a sentence rather than a queued job, and the three
+ // revalidations.
+ return runMetadataScanJob({
paths,
- channelSlug: slug,
- spec: { kind: "metadata-scan", slug, params: { queueKey } },
- fn: async (onLog, signal, setProgress) => {
- await runMetadataScan({
- paths,
- channelSlug: slug,
- channelConfig,
- onLog,
- signal,
- setProgress,
- onPlatformBackoff: () => recordDownloadBackoff(platform, paths),
- });
+ slug,
+ channelConfig,
+ queueKey,
+ afterRun: () => {
revalidatePath(`/channels/${slug}`);
revalidatePath("/channels");
revalidatePath("/operations/[id]", "page");
diff --git a/editor/app/operations/components/InFlightList.tsx b/editor/app/operations/components/InFlightList.tsx
@@ -41,16 +41,38 @@ export function InFlightList({
aria-hidden="true"
className="size-1.5 shrink-0 self-center rounded-full bg-info animate-pulse motion-reduce:animate-none"
/>
- <Link
- href={`/channels/${item.channelSlug}/videos/${item.videoId}`}
- className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
- >
- {item.channelSlug}/{item.videoId}
- </Link>
- <span className="text-xs text-muted-foreground">
- {at >= 0 ? `rule ${at + 1}` : item.leafId}
- {leaf ? ` · ${leafSentence(leaf, channels)}` : ""}
- </span>
+ {/* A UNIT THAT IS NOT A VIDEO SAYS SO INSTEAD OF LINKING.
+ The download lane's metadata-scan unit is channel-scoped:
+ its `videoId` is a synthetic key naming no directory, so the
+ usual link would be a 404 and the usual "rule N" would name
+ a leaf that exists in no tree. It carries a `note`, and a
+ unit with one renders the note — see AutoRunnerInFlight. */}
+ {item.note ? (
+ <>
+ <Link
+ href={`/channels/${item.channelSlug}`}
+ className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
+ >
+ {item.channelSlug}
+ </Link>
+ <span className="text-xs text-muted-foreground">
+ {item.note}
+ </span>
+ </>
+ ) : (
+ <>
+ <Link
+ href={`/channels/${item.channelSlug}/videos/${item.videoId}`}
+ className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
+ >
+ {item.channelSlug}/{item.videoId}
+ </Link>
+ <span className="text-xs text-muted-foreground">
+ {at >= 0 ? `rule ${at + 1}` : item.leafId}
+ {leaf ? ` · ${leafSentence(leaf, channels)}` : ""}
+ </span>
+ </>
+ )}
<span className="ml-auto tabular-nums text-xs text-muted-foreground">
{now !== null ? formatElapsed(now - item.startedAt) : "—"}
</span>
diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts
@@ -1,4 +1,4 @@
-import { mkdir, writeFile } from "node:fs/promises";
+import { mkdir, readFile, writeFile } from "node:fs/promises";
import { test, expect } from "@playwright/test";
import {
channelStage,
@@ -795,6 +795,125 @@ test("download: prioritizes channels across the per-platform queue", async ({
]);
});
+// A FILTERED download channel whose snapshot advertises a metadata-scan
+// backlog. The fake yt-dlp titles every video "Synthetic <id>", so an include
+// of /guest/ keeps guestvid* and rejects plainvid* with no fixture metadata.
+// The channel's yt-dlp invocation log, or "" before the first one.
+async function readText(relPath: string): Promise<string> {
+ return readFile(resolvePath(relPath), "utf8").catch(() => "");
+}
+
+async function makeFilteredDownloadChannel(
+ slug: string,
+ ids: string[],
+ unscanned: number,
+) {
+ const root = resolvePath(`test-transcripts/channels/${slug}`);
+ await mkdir(`${root}/data`, { recursive: true });
+ await writeFile(
+ `${root}/config.json`,
+ JSON.stringify({
+ handling: "youtube",
+ name: slug,
+ url: `https://www.youtube.com/@${slug}/videos`,
+ downloadFilter: { include: "guest" },
+ }),
+ );
+ await writeFile(
+ `${root}/playlist`,
+ ids.map((id) => `https://www.youtube.com/watch?v=${id}`).join("\n") + "\n",
+ );
+ await writeFile(
+ `${root}/snapshot.json`,
+ JSON.stringify({
+ generatedAt: "2026-06-01T00:00:00.000Z",
+ totals: { videos: 0, transcribed: 0, downloaded: 0 },
+ buckets: {},
+ undownloadedIds: ids,
+ metadataScan: { scanned: 0, errors: 0, unscanned },
+ }),
+ );
+}
+
+test("the download lane scans a filtered channel before it downloads from it", async ({
+ request,
+}) => {
+ // THE SCAN DECIDES WHAT THE DOWNLOADS ARE, so it is not another kind of work
+ // competing with them — it goes first. Without it every non-matching video on
+ // a filtered channel is prefetched, rejected and discarded one yt-dlp
+ // invocation at a time, against a source that is counting them; one batch
+ // answers the whole channel.
+ await resetData(null);
+ await makeFilteredDownloadChannel(
+ "filtered",
+ ["guestvid0001", "plainvid0001", "plainvid0002"],
+ 3,
+ );
+
+ const root: Group = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "leaf-all", match: { type: "all" } }],
+ };
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ workers: ONE_WORKER,
+ autoQueue: downloadAutoQueue(root),
+ });
+
+ await startRunner(request, "download");
+
+ const invocationsPath =
+ "test-transcripts/channels/filtered/fake-ytdlp.invocations";
+ await expect
+ .poll(
+ async () =>
+ (await readText(invocationsPath)).includes("prefetch:") ||
+ (await pathExists(
+ "test-transcripts/channels/filtered/data/guestvid0001/transcript.en.vtt",
+ )),
+ { timeout: 60_000 },
+ )
+ .toBe(true);
+
+ const invocations = await readText(invocationsPath);
+ const scanAt = invocations.indexOf("metadata-scan:");
+ const firstPrefetchAt = invocations.indexOf("prefetch:");
+ // The scan ran, and it ran BEFORE a single video was prefetched.
+ expect(scanAt).toBeGreaterThanOrEqual(0);
+ expect(firstPrefetchAt).toBeGreaterThan(scanAt);
+
+ // And it settled the two non-matches, so they were never prefetched at all.
+ const scan = await readJson<{ entries: Record<string, unknown> }>(
+ "test-transcripts/channels/filtered/metadata-scan.json",
+ );
+ expect(Object.keys(scan.entries).sort()).toEqual([
+ "guestvid0001",
+ "plainvid0001",
+ "plainvid0002",
+ ]);
+ for (const id of ["plainvid0001", "plainvid0002"]) {
+ expect(invocations).not.toContain(
+ `prefetch:https://www.youtube.com/watch?v=${id}`,
+ );
+ }
+
+ // ONE SCAN PER CHANNEL PER RUNNER. The loop keeps ticking every three
+ // seconds; a second scan would mean the source is re-asked for the whole
+ // listing forever.
+ await expect
+ .poll(async () => (await getStatus(request)).download.picks.length, {
+ timeout: 30_000,
+ })
+ .toBeGreaterThan(0);
+ expect(
+ (await readText(invocationsPath)).split("metadata-scan:").length - 1,
+ ).toBe(1);
+});
+
// A React state update that lands before hydration is discarded silently, so a
// select or a click made too early leaves the page looking changed while the
// component still holds the old value — and the next save writes the old value.
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4148,9 +4148,73 @@ never pulls `reader-fs.ts` into a client chunk.
has no `runner`, so the operator presses Run. The obvious next step is for the
download lane to run it for a channel whose filter has unscanned listed videos
— which is exactly the backlog the snapshot already carries.
+ **DONE 2026-09-21 — see the section below, which also amends the pause note
+ above.**
---
+## Filtered-channel follow-ups (verified 2026-09-21, branch `feat/filtered-channel-followups`)
+
+Three things the filter slice left: a rejection's leftover directory, nothing
+auto-dispatching the scan, and no answer but "skip" for a livestream the filter
+rejects.
+
+- **A TITLE-FILTER REJECTION NOW DELETES ITS OWN PREFETCH DIRECTORY**
+ (`ytdlp/downloadOneManaged.ts`, `discardPrefetchDir`). The prefetch writes
+ `data/<id>/metadata.info.json` BEFORE the filters look at it, so by the time
+ the operator's "not this one" is known the directory exists — and a directory
+ holding a metadata.info.json is admitted to the LMDB index and the published
+ site by `buildIndex.ts:309-315`, transcript or not, while `deriveChannelSets`
+ reads its NAME as "ever fetched". The scan creates none of them for exactly
+ those two reasons; the downloader now creates none either. It fires ONLY for a
+ video the filter needed a live prefetch for (one the scan has not read) — a
+ settled video is never invoked at all.
+- **The count does not move, and the reason is that it never came from there.**
+ `buckets.skippedByTitleFilter` is built at `channelSnapshot.ts:1378-1383` from
+ `settledIds` ← `settledIdsFrom(metadataScanStore, config)` at `:927`. Nothing
+ in it reads `download-outcome.json`. The only on-disk reader of
+ `"skipped-filtered"` is the RETRYABLE `skippedByFilter` bucket at `:1101-1105`,
+ which is why every OTHER filter (skip-live) still writes its outcome sidecar:
+ that skip says "we will try again", and a video with neither directory nor
+ outcome would silently leave the bucket. `channelSnapshot.test.ts` runs a whole
+ filtered channel with NO `data/` dirs at all and still gets the bucket.
+- **It refuses rather than guesses.** Anything `isVideoDownloaded` calls
+ downloaded, a `clips/` window dir, or a `saved-video.json` pointer and the
+ directory stands. The returned `DownloadOutcomeRecord` is unchanged either way,
+ so `runYtdlp.ts:949` and the runner's `unitStatus` check need no edit.
+- **THE DOWNLOAD LANE AUTO-DISPATCHES THE SCAN.** `METADATA_SCAN_OPERATION` now
+ declares `runner: "download"`, and the runner's `next()` runs a pre-pick before
+ any video: `pickMetadataScanChannel` (pure, exported, unit-tested) takes the
+ first channel in `metaCache` order that has a COMPILING filter, a non-zero
+ `snapshot.metadataScan.unscanned`, no scan already done by this runner, and a
+ platform that is neither busy nor cooling. It goes first because it decides
+ what the downloads ARE.
+- **The scan unit is channel-scoped and holds a platform slot.** `Picked` gains
+ `scan`; the synthetic key is `metadata-scan <slug>` (a space, which no video id
+ contains) with leaf id `metadata-scan` and an EMPTY `path`, so no node's
+ `maxWorkers` moves for it. It takes `platformInFlight` exactly as a download
+ unit does, so a scan and a download never hit one source at once.
+ `AutoRunnerInFlight` gains `note?`, and `InFlightList` renders the note (and
+ links the CHANNEL) instead of a dead video link.
+- **ONE SCAN PER CHANNEL PER RUNNER, deliberately.** A scan that left the backlog
+ above zero was stopped by something — a soft block, a cooldown, a batch that
+ ended early. `scannedThisRun` means this runner does not re-offer it three
+ seconds later; the operator's Run and the next restart both still work.
+- **`pauseLaneFor("metadata-scan")` is `"download"` now, not null, and the old
+ note's reasoning moved to the Run button.** `pauseLaneFor` asks `runner` first,
+ so naming the runner gives the operation the download lane's console AND makes
+ the download pause hold the AUTO-dispatch. What the exemption was FOR is
+ intact: `runMetadataScanAction` checks the platform cooldown and no gate at
+ all, so the operator can still scan while downloads are paused — which is
+ precisely when they want to decide what the lane should fetch on resume.
+- **`common/controller/metadataScanJob.ts` is the one definition of the scan as a
+ JOB** — kind, per-platform queue key, replay spec and the shared
+ `onPlatformBackoff` wiring. The editor action keeps only what a CLICK is owed
+ (the cooldown answered as a sentence, and three `revalidatePath` calls, passed
+ as `afterRun`); the runner passes `background: true` and neither. `needsMedia`
+ comes from the kind (TRUE for this one), so `runManagedFunction` refuses it
+ against an unmounted drive for both callers with nothing said in the module.
+
## Clip windows: media sourced for another tool (2026-09-20)
umtool asks the editor for a clip window instead of running yt-dlp