commit 6a36032cd5b828433954b9bb56f839738bfc1d90
parent ecdb2bee84ca8e82aba5d35cfd8ce958e1212c3b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 00:25:23 -0400
metadata scan: a catalogued operation that reads titles without fetching media
`runMetadataScan` is one yt-dlp pass over the listed videos nobody has read the
title of — listed, minus anything already on disk, minus anything already in
the store — printing a JSON subset per entry and writing only
`metadata-scan.json`. It flushes every 25 records, so a killed or rate-limited
run keeps what it learned.
It is in `operationCatalog()` as the second channel-scoped entry, and the two
catalog guard tests were updated rather than widened: sync stays the ONE
cadence-triggered operation (that is how /operations/sync is chosen) and the
scan is backlog-triggered. `pauseLaneFor` answers null for it — it rides the
platform download queue but is not held by the download gate, because it
fetches no media and the operator runs it precisely to decide what a paused
lane should fetch when it resumes. `needsMedia: false` for the same reason: an
unmounted drive is no reason to refuse a job that never opens a video dir.
Rate limiting is a batch-level stop, not a per-video error: the run kills the
child, records the shared per-platform cooldown the auto-download runner
honors, and keeps every record it already flushed. YouTube's bot check
("Sign in to confirm you're not a bot") now classifies as rate_limit —
it is not a 429 and not per-video, and once it fires every remaining entry in
the batch fails the same way. needs_auth ids get ONE cookie retry, and only
when the resolved mode offers cookies; defer mode offering none is the point
of defer.
`JobProgressMetric` gains "scans". Its progress cannot be re-counted from disk
— the scan writes no video directory — so the runner always sets `current`
itself.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
10 files changed, 522 insertions(+), 7 deletions(-)
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -447,6 +447,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "platform",
needsMedia: true,
},
+ // The metadata scan (ytdlp/metadataScan.ts). On the platform download queue,
+ // because it contends for the same thing a download does — the source's
+ // patience — but `needsMedia` is FALSE and that is the whole point: it reads
+ // titles, writes one channel-level file, and never opens or creates a video
+ // directory. An unmounted media drive is no reason to refuse it.
+ "metadata-scan": {
+ kind: "metadata-scan",
+ label: "Metadata scan",
+ drainable: true,
+ replayable: true,
+ queueKeyStrategy: "platform",
+ needsMedia: false,
+ },
// Replayable kinds that never had a JOB_KIND_LABELS entry: label omitted so
// jobKindLabel() keeps falling back to the raw kind (unchanged behavior).
"store-playlist": {
diff --git a/common/jobs/registry.ts b/common/jobs/registry.ts
@@ -18,7 +18,11 @@ export type JobProgressMetric =
| "downloads"
| "transcripts"
| "digests"
- | "backfills";
+ | "backfills"
+ // The metadata scan. Its progress CANNOT be re-counted from disk — the scan
+ // deliberately writes no video directory — so its runner always sets
+ // `current` itself. See ytdlp/metadataScan.ts.
+ | "scans";
export type JobProgress = {
metric: JobProgressMetric;
diff --git a/common/lib/availability.ts b/common/lib/availability.ts
@@ -227,7 +227,14 @@ export function classifyDownloadFailure(
/http error 429/.test(s) ||
/too many requests/.test(s) ||
/rate[- ]?limit/.test(s) ||
- /throttl/.test(s)
+ /throttl/.test(s) ||
+ // YouTube's bot check. Not a 429 and not per-video: once it fires, every
+ // subsequent request in the same batch fails the same way, so it is a
+ // batch-level stop signal exactly like a rate limit — and it is what a
+ // metadata scan over a whole channel listing trips first. The apostrophe is
+ // a right single quote in yt-dlp's output, so match either.
+ /sign in to confirm you['\u2019]?re not a bot/.test(s) ||
+ /confirm you['\u2019]?re not a bot/.test(s)
) {
return "rate_limit";
}
diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts
@@ -1320,12 +1320,23 @@ test("the three speaker operations share one group; digest does not", () => {
assert.equal(operationGroup("no-such-operation"), null);
});
-test("`scope` and `trigger`: sync is the one channel-scoped, cadence-triggered operation", () => {
- // The two fields exist so a per-video surface can EXCLUDE sync by a declared
- // fact rather than by its id. Every other entry is per-video and backlog-fed.
+test("`scope` and `trigger`: sync is the one cadence-triggered operation", () => {
+ // The two fields exist so a per-video surface can EXCLUDE a channel-scoped
+ // entry by a declared fact rather than by its id.
+ //
+ // TWO entries are channel-scoped now: sync and the metadata scan. They differ
+ // on `trigger`, and that difference is the point — sync runs on a cadence
+ // (the scheduler's heartbeat decides), the scan runs off a BACKLOG (listed
+ // videos nobody has read the title of). /operations/sync is chosen off
+ // `trigger`, which is why only one entry may claim "cadence".
+ const channelScoped = operationCatalog()
+ .filter((op) => op.scope === "channel")
+ .map((op) => op.id)
+ .sort();
+ assert.deepEqual(channelScoped, ["metadata-scan", "sync"]);
for (const op of operationCatalog()) {
- assert.equal(op.scope === "channel", op.id === "sync", `${op.id} scope ${op.scope}`);
assert.equal(op.trigger === "cadence", op.id === "sync", `${op.id} trigger ${op.trigger}`);
+ if (op.scope !== "channel") assert.equal(op.scope, "video", op.id);
}
assert.equal(operationCatalog()[0].id, "sync"); // upstream first
});
diff --git a/common/lib/operations.ts b/common/lib/operations.ts
@@ -1351,6 +1351,32 @@ export const SYNC_OPERATION: OperationDescriptor = {
settingsBlock: "syncScheduler",
};
+// THE METADATA SCAN. A channel-scoped, backlog-fed operation that reads what
+// this channel's listed-but-unfetched videos are CALLED, without fetching any of
+// them — so the per-channel download filter can settle the ones the operator
+// does not want before anything is downloaded.
+//
+// On the platform download queue and contending for the network, like sync and
+// download: it is one metadata request per listed video, and the source counts
+// them the same way it counts a download's.
+//
+// NO `runner`, deliberately: nothing auto-dispatches this yet. The operator
+// presses Run. The obvious follow-up is for the download lane to run it for a
+// channel whose filter has unscanned listed videos, which is exactly the
+// backlog this declares — see plans/FACTS.md.
+export const METADATA_SCAN_OPERATION: OperationDescriptor = {
+ id: "metadata-scan",
+ label: "Metadata scan",
+ hint: "Reading the title, description and date of videos that are in the listing but not downloaded, without fetching any media. What the per-channel download filter needs in order to decide: with no scan, the only way to learn a video's title is to start downloading it.",
+ group: "sync",
+ shortLabel: "Scan",
+ costBasis: "one metadata fetch per listed video, no media",
+ lane: { queueKey: "download:<platform>", contendsFor: "network" },
+ dispatch: "external",
+ scope: "channel",
+ trigger: "backlog",
+};
+
export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [
{
id: "download",
@@ -1425,6 +1451,7 @@ export type OperationDescriptor = {
export function operationCatalog(): OperationDescriptor[] {
return [
SYNC_OPERATION,
+ METADATA_SCAN_OPERATION,
...EXTERNAL_OPERATIONS,
...OPERATIONS.map((k) => ({
id: k.id,
diff --git a/common/lib/pauseGates.test.ts b/common/lib/pauseGates.test.ts
@@ -167,9 +167,14 @@ test("pauseLaneFor answers for every catalog id", () => {
// Catalogued, and deliberately gateless: the sync scheduler's own `enabled`
// is its switch, and its queue key is neither sweep's.
sync: null,
+ // Catalogued and gateless for a different reason: the metadata scan rides
+ // the platform download queue but is NOT held by the download gate. It
+ // fetches no media, and the operator runs it precisely to decide what a
+ // paused download lane should fetch when it resumes.
+ "metadata-scan": null,
};
const ids = operationCatalog().map((o) => o.id);
- assert.equal(ids.length, 7);
+ assert.equal(ids.length, 8);
for (const id of ids) {
assert.ok(id in expected, `catalog gained ${id} with no expected lane`);
assert.equal(pauseLaneFor(id), expected[id], `pauseLaneFor(${id})`);
diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts
@@ -119,6 +119,7 @@ export type StageStatus = {
const JOB_KIND_TO_STAGE: Record<string, StageId> = {
"store-playlist": "playlist",
sync: "playlist",
+ "metadata-scan": "playlist",
"download-from-playlist": "download",
"download-missing": "download",
"whisper-all": "transcribe",
diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts
@@ -0,0 +1,444 @@
+// THE METADATA SCAN — one yt-dlp metadata pass over a channel's listed but
+// never-fetched videos, reading what they are CALLED so the download filter can
+// decide about them before a single byte of media is fetched.
+//
+// It is a catalogued operation (common/lib/operations.ts, METADATA_SCAN_OPERATION)
+// on the channel's platform download queue, because it contends for exactly the
+// resource a download does: the source's patience. It writes NO video directory
+// — see controller/metadataScanStore.ts for the two downstream readers that
+// would misread one — and no media, which is why its job kind declares
+// `needsMedia: false`.
+//
+// What it costs: one metadata extraction per unscanned listed video, in a single
+// yt-dlp process reading a batch file, with `--sleep-requests 1`. What it saves:
+// on a filtered channel, every one of those videos would otherwise be prefetched
+// by the downloader on every run, forever.
+
+import path from "node:path";
+import { mkdtemp, readdir, readFile, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import { execa } from "execa";
+import {
+ classifyDownloadFailure,
+ parseUnavailableFromStderr,
+} from "../lib/availability";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters";
+import {
+ alwaysCookies,
+ authRetryCookies,
+ cookieArgs,
+ resolveCookiePolicy,
+} from "../lib/cookiePolicy";
+import type { Paths } from "../lib/paths";
+import { getSettings } from "../lib/settings";
+import { extractVideoId } from "../lib/videoId";
+import { isVideoDownloaded, readVideoFiles } from "../lib/videoStatus";
+import type { JobProgress } from "../jobs/registry";
+import {
+ loadMetadataScan,
+ upsertMetadataScan,
+ type MetadataScanEntry,
+ type MetadataScanError,
+ type MetadataScanRun,
+} from "../controller/metadataScanStore";
+
+// yt-dlp's JSON-subset print template: one JSON object per entry, so the scan
+// parses lines instead of whole info jsons — and so nothing is ever written to
+// disk by yt-dlp itself.
+const PRINT_TEMPLATE =
+ "%(.{id,title,description,upload_date,live_status,duration})j";
+
+// Flush to the store every N records. A killed or rate-limited job keeps what it
+// learned; 25 is small enough that a run stopped early has lost almost nothing
+// and large enough that a 1,800-video scan rewrites the file ~72 times, not
+// 1,800.
+const FLUSH_EVERY = 25;
+
+// `ERROR: [youtube] dQw4w9WgXcQ: Video unavailable` — yt-dlp names the id it
+// failed on, which is the only way to attribute a per-video failure in a batch.
+const ERROR_LINE = /^ERROR:\s*(?:\[[^\]]+\]\s*)?([^\s:]+)\s*:\s*(.*)$/;
+
+export type MetadataScanResult = {
+ scanned: number;
+ errors: number;
+ targets: number;
+ stopped?: MetadataScanRun["stopped"];
+};
+
+export type RunMetadataScanOpts = {
+ paths: Paths;
+ channelSlug: string;
+ channelConfig: ChannelConfig;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+ setProgress?: (progress: JobProgress) => void;
+ // Called when the source rate-limits us (HTTP 429, or YouTube's bot check),
+ // so the SHARED per-platform cooldown the auto-download runner honors is
+ // recorded. Same contract as runYtdlp's.
+ onPlatformBackoff?: (
+ failureClass: "rate_limit" | "network",
+ ) => void | Promise<void>;
+};
+
+async function readPlaylistIds(channelDir: string): Promise<string[]> {
+ const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch(
+ () => "",
+ );
+ const ids: string[] = [];
+ const seen = new Set<string>();
+ for (const line of raw.split("\n")) {
+ const url = line.trim();
+ if (!url) continue;
+ const id = extractVideoId(url);
+ if (id && !seen.has(id)) {
+ seen.add(id);
+ ids.push(id);
+ }
+ }
+ return ids;
+}
+
+// Ids that already have media/transcripts on disk. A downloaded video needs no
+// scan — its metadata.info.json is the better record, and the snapshot reads it.
+async function fetchedIds(dataDir: string): Promise<Set<string>> {
+ const out = new Set<string>();
+ let names: string[];
+ try {
+ names = (await readdir(dataDir, { withFileTypes: true }))
+ .filter((e) => e.isDirectory())
+ .map((e) => e.name);
+ } catch {
+ return out;
+ }
+ for (const name of names) {
+ const files = await readVideoFiles(path.join(dataDir, name));
+ if (isVideoDownloaded(files) || files.hasMeta) out.add(name);
+ }
+ return out;
+}
+
+// What this run will fetch: listed, not already fetched, not already scanned.
+// Deliberately NOT filtered by the download filter — the scan exists to find out
+// what the filter should say, so it cannot use the answer as its input.
+export async function metadataScanTargets(
+ paths: Paths,
+ slug: string,
+): Promise<string[]> {
+ const channelDir = path.join(paths.channelsDir, slug);
+ const [listed, fetched, scan] = await Promise.all([
+ readPlaylistIds(channelDir),
+ fetchedIds(path.join(channelDir, "data")),
+ loadMetadataScan(paths, slug),
+ ]);
+ return listed.filter((id) => !fetched.has(id) && !scan.entries[id]);
+}
+
+function urlForId(id: string, channelConfig: ChannelConfig): string {
+ // The playlist file holds real URLs; rebuilding one from an id is only safe
+ // for YouTube. So callers pass URLs through — see runMetadataScan, which reads
+ // the playlist file again for exactly this reason.
+ void channelConfig;
+ return `https://www.youtube.com/watch?v=${id}`;
+}
+
+async function readPlaylistUrlsById(
+ channelDir: string,
+): Promise<Map<string, string>> {
+ const raw = await readFile(path.join(channelDir, "playlist"), "utf8").catch(
+ () => "",
+ );
+ const byId = new Map<string, string>();
+ for (const line of raw.split("\n")) {
+ const url = line.trim();
+ if (!url) continue;
+ const id = extractVideoId(url);
+ if (id && !byId.has(id)) byId.set(id, url);
+ }
+ return byId;
+}
+
+export async function runMetadataScan(
+ opts: RunMetadataScanOpts,
+): Promise<MetadataScanResult> {
+ const { paths, channelSlug: slug, channelConfig } = opts;
+ const channelDir = path.join(paths.channelsDir, slug);
+ const startedAt = new Date().toISOString();
+
+ const targetIds = await metadataScanTargets(paths, slug);
+ const urlsById = await readPlaylistUrlsById(channelDir);
+ if (targetIds.length === 0) {
+ opts.onLog(
+ "Nothing to scan: every listed video is already downloaded or already in the metadata scan.\n",
+ );
+ await upsertMetadataScan(
+ paths,
+ slug,
+ {
+ lastRun: {
+ startedAt,
+ finishedAt: new Date().toISOString(),
+ scanned: 0,
+ errors: 0,
+ },
+ },
+ new Date().toISOString(),
+ );
+ return { scanned: 0, errors: 0, targets: 0 };
+ }
+
+ opts.onLog(
+ `Scanning metadata for ${targetIds.length} listed video(s). No media is downloaded and no video directory is created.\n`,
+ );
+ opts.setProgress?.({
+ metric: "scans",
+ initial: 0,
+ target: targetIds.length,
+ current: 0,
+ });
+
+ const policy = resolveCookiePolicy(getSettings(), channelConfig);
+ const tmpRoot = await mkdtemp(path.join(tmpdir(), "ttb-mdscan-"));
+
+ const entries: Record<string, MetadataScanEntry> = {};
+ const errors: Record<string, MetadataScanError> = {};
+ let scannedTotal = 0;
+ let pending = 0;
+ let stopped: MetadataScanRun["stopped"] | undefined;
+ let stoppedMessage: string | undefined;
+ const needsAuthIds: string[] = [];
+
+ const flush = async (force = false) => {
+ if (!force && pending < FLUSH_EVERY) return;
+ if (pending === 0 && !force) return;
+ await upsertMetadataScan(
+ paths,
+ slug,
+ { entries: { ...entries }, errors: { ...errors } },
+ new Date().toISOString(),
+ );
+ pending = 0;
+ };
+
+ // One pass over a batch file. Returns the ids it could not read as needs_auth,
+ // so the caller can decide whether a cookie retry is worth a second pass.
+ const runPass = async (
+ ids: ReadonlyArray<string>,
+ cookies: string | undefined,
+ label: string,
+ ): Promise<void> => {
+ const batchFile = path.join(tmpRoot, `${label}.batch`);
+ await writeFile(
+ batchFile,
+ ids.map((id) => urlsById.get(id) ?? urlForId(id, channelConfig)).join("\n") +
+ "\n",
+ );
+ const args = [
+ "--ignore-config",
+ "--skip-download",
+ "--no-warnings",
+ "--ignore-errors",
+ // One request per second. The scan touches every listed video of a
+ // channel in one process; without this it is the most rate-limitable
+ // thing the app does.
+ "--sleep-requests",
+ "1",
+ "--print",
+ PRINT_TEMPLATE,
+ "-a",
+ batchFile,
+ ...cookieArgs(cookies),
+ ...(channelConfig.ytdlpExtraArgs ?? []),
+ ];
+ opts.onLog(`$ ${paths.ytdlpBin} ${args.join(" ")}\n`);
+
+ const child = execa(paths.ytdlpBin, args, {
+ cwd: channelDir,
+ cancelSignal: opts.signal,
+ all: false,
+ buffer: false,
+ reject: false,
+ });
+
+ const takeRecord = (line: string) => {
+ const trimmed = line.trim();
+ if (!trimmed.startsWith("{")) return;
+ let parsed: Record<string, unknown>;
+ try {
+ parsed = JSON.parse(trimmed) as Record<string, unknown>;
+ } catch {
+ return;
+ }
+ const id = typeof parsed.id === "string" ? parsed.id : "";
+ if (!id) return;
+ entries[id] = {
+ title: typeof parsed.title === "string" ? parsed.title : "",
+ description:
+ typeof parsed.description === "string" ? parsed.description : "",
+ uploadDate:
+ typeof parsed.upload_date === "string" ? parsed.upload_date : "",
+ ...(typeof parsed.live_status === "string"
+ ? { liveStatus: parsed.live_status }
+ : {}),
+ ...(typeof parsed.duration === "number"
+ ? { duration: parsed.duration }
+ : {}),
+ scannedAt: new Date().toISOString(),
+ };
+ scannedTotal++;
+ pending++;
+ opts.setProgress?.({
+ metric: "scans",
+ initial: 0,
+ target: targetIds.length,
+ current: scannedTotal,
+ });
+ };
+
+ let stdoutBuf = "";
+ const flushes: Promise<void>[] = [];
+ child.stdout?.on("data", (c: Buffer) => {
+ stdoutBuf += c.toString("utf8");
+ let nl: number;
+ while ((nl = stdoutBuf.indexOf("\n")) !== -1) {
+ takeRecord(stdoutBuf.slice(0, nl));
+ stdoutBuf = stdoutBuf.slice(nl + 1);
+ }
+ if (pending >= FLUSH_EVERY) flushes.push(flush());
+ });
+
+ let stderrBuf = "";
+ const takeError = (line: string) => {
+ const trimmed = line.trim();
+ if (!trimmed.startsWith("ERROR")) return;
+ const cls = parseUnavailableFromStderr(trimmed);
+ const failure = classifyDownloadFailure(trimmed, cls);
+ // A batch-level signal. Continuing would just hammer the source, and on
+ // YouTube's bot check every subsequent entry fails anyway — so stop, record
+ // the shared per-platform cooldown, and keep what we have.
+ if (failure === "rate_limit") {
+ stopped = "rate_limit";
+ stoppedMessage = trimmed.slice(0, 300);
+ child.kill("SIGTERM");
+ return;
+ }
+ const m = ERROR_LINE.exec(trimmed);
+ const id = m?.[1] ?? "";
+ const message = (m?.[2] || trimmed).trim().slice(0, 300);
+ if (!id) return;
+ if (cls === "needs_auth") needsAuthIds.push(id);
+ errors[id] = { class: cls, message, at: new Date().toISOString() };
+ pending++;
+ };
+ child.stderr?.on("data", (c: Buffer) => {
+ const chunk = c.toString("utf8");
+ opts.onLog(chunk);
+ stderrBuf += chunk;
+ let nl: number;
+ while ((nl = stderrBuf.indexOf("\n")) !== -1) {
+ takeError(stderrBuf.slice(0, nl));
+ stderrBuf = stderrBuf.slice(nl + 1);
+ }
+ });
+
+ const result = await child;
+ if (stdoutBuf) takeRecord(stdoutBuf);
+ if (stderrBuf) takeError(stderrBuf);
+ await Promise.all(flushes);
+ await flush(true);
+
+ if (opts.signal.aborted) {
+ stopped = "aborted";
+ return;
+ }
+ if (stopped === "rate_limit") return;
+ // --ignore-errors means a per-video failure still exits nonzero while every
+ // readable entry was printed, so a nonzero exit is only fatal when nothing
+ // came back at all.
+ if (
+ result.exitCode !== 0 &&
+ result.exitCode !== 101 &&
+ scannedTotal === 0 &&
+ Object.keys(errors).length === 0
+ ) {
+ stopped = "error";
+ stoppedMessage = `yt-dlp exited with code ${result.exitCode}`;
+ throw new Error(stoppedMessage);
+ }
+ };
+
+ try {
+ await runPass(targetIds, alwaysCookies(policy), "main");
+
+ // ONE cookie retry for the auth-gated ids, and only when the mode actually
+ // offers cookies. Defer mode resolves to none, which is the point of defer:
+ // those videos wait for a deliberate cookie run.
+ const retryCookies = authRetryCookies(policy);
+ const retryIds = [...new Set(needsAuthIds)].filter((id) => !entries[id]);
+ if (
+ stopped === undefined &&
+ retryIds.length > 0 &&
+ retryCookies !== undefined &&
+ !opts.signal.aborted
+ ) {
+ opts.onLog(
+ `${retryIds.length} video(s) need auth; retrying once with --cookies-from-browser ${retryCookies}.\n`,
+ );
+ await runPass(retryIds, retryCookies, "auth-retry");
+ }
+ } finally {
+ await flush(true);
+ await rm(tmpRoot, { recursive: true, force: true });
+ }
+
+ if (stopped === "rate_limit") {
+ opts.onLog(
+ `STOPPED: the source is rate-limiting this scan. ${scannedTotal} video(s) were read and saved; the rest stay unscanned. A per-platform cooldown has been recorded — run the scan again once it lapses.\n`,
+ );
+ await opts.onPlatformBackoff?.("rate_limit");
+ }
+
+ const errorCount = Object.keys(errors).length;
+ await upsertMetadataScan(
+ paths,
+ slug,
+ {
+ lastRun: {
+ startedAt,
+ finishedAt: new Date().toISOString(),
+ scanned: scannedTotal,
+ errors: errorCount,
+ ...(stopped ? { stopped } : {}),
+ ...(stoppedMessage ? { message: stoppedMessage } : {}),
+ },
+ },
+ new Date().toISOString(),
+ );
+
+ // The summary the operator actually wants: not "how many did you read" but
+ // "what does my filter now say about them".
+ const compiled = compileDownloadFilter(channelConfig.downloadFilter);
+ if (compiled) {
+ const scan = await loadMetadataScan(paths, slug);
+ let matched = 0;
+ let settled = 0;
+ for (const entry of Object.values(scan.entries)) {
+ if (titleFilterRejects(compiled, entry)) settled++;
+ else matched++;
+ }
+ opts.onLog(
+ `Metadata scan complete: ${scannedTotal} scanned, ${errorCount} error(s). Download filter: ${matched} match, ${settled} filtered out.\n`,
+ );
+ } else {
+ opts.onLog(
+ `Metadata scan complete: ${scannedTotal} scanned, ${errorCount} error(s). No download filter configured on this channel.\n`,
+ );
+ }
+
+ return {
+ scanned: scannedTotal,
+ errors: errorCount,
+ targets: targetIds.length,
+ ...(stopped ? { stopped } : {}),
+ };
+}
diff --git a/editor/app/jobs/components/JobProgressBars.tsx b/editor/app/jobs/components/JobProgressBars.tsx
@@ -132,6 +132,7 @@ const METRIC_LABELS: Record<JobProgressMetric, string> = {
transcripts: "Transcripts",
digests: "Digests",
backfills: "Backfill",
+ scans: "Metadata scan",
};
const METRIC_FILL: Record<JobProgressMetric, string> = {
@@ -139,6 +140,7 @@ const METRIC_FILL: Record<JobProgressMetric, string> = {
transcripts: "bg-success",
digests: "bg-info",
backfills: "bg-warning",
+ scans: "bg-info/60",
};
export function JobProgressBar({
diff --git a/editor/app/widget/components/MonitorWidget.tsx b/editor/app/widget/components/MonitorWidget.tsx
@@ -1112,6 +1112,7 @@ const METRIC_PREFIX: Record<JobProgressMetric, string> = {
transcripts: "",
digests: "\u00b6 ",
backfills: "\u21ba ",
+ scans: "\u2315 ",
};
// One-line textual summary of a job's batch progress, e.g. "↓ 5/10 · ~2m left".