Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 3c8562a91ac4aabfa6829a6846594381a69cce2c
parent 6a36032cd5b828433954b9bb56f839738bfc1d90
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 00:31:59 -0400

metadata scan: an action, a Playlist-stage block, and a row on its operation page

`runMetadataScanAction` is deliberately NOT routed through runPipelineAction,
and each difference is a reason the scan is its own operation: it is not held
by the downloads pause (it fetches no media, and the operator runs it precisely
to decide what a paused lane should fetch when it resumes), it needs no disk
gate, and it both respects and records the per-platform rate-limit cooldown.

The Playlist stage gets the block, because that is where "what is in this
channel" already lives: scanned N of M listed, errors, last run and why it
stopped, and — when a filter is configured — matches / filtered out /
unscanned with the newest 50 matched titles behind a <details>. The point of
that list is an eyeball check that the regex caught what the operator meant
before ~1,800 videos are settled on its word.

`/operations/metadata-scan` needs no route work (the catalog is the route
table); it needs a backlog, which the snapshot now carries as
`metadataScan.unscanned` — listed, nothing on disk, neither read nor recently
failed, so it means exactly what metadataScanTargets() will fetch. The section
lists only channels that HAVE a filter: the scan is possible everywhere but is
only work that needs doing where a filter is waiting on titles.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 48++++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/controller/metadataScanStore.ts | 40+++++++++++++++++++++++++++++++++++-----
Meditor/app/channels/[slug]/components/stages/PlaylistStage.tsx | 127+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/page.tsx | 42++++++++++++++++++++++++++++++++++++++++--
Meditor/app/channels/[slug]/pipelineActions.ts | 59+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/components/actions/InlineActionButton.tsx | 10+++++++++-
Meditor/app/components/channelWork/sections.test.ts | 13++++++++++++-
Meditor/app/components/channelWork/sections.tsx | 20+++++++++++++++++++-
Meditor/app/jobs/jobReplayRegistry.ts | 8++++++++
Meditor/app/lib/actionable/loadActionable.ts | 27+++++++++++++++++++++++++++
10 files changed, 382 insertions(+), 12 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -46,7 +46,12 @@ import { } from "../jobs/autoQueuePolicy"; import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; -import { settledByTitleFilterIds } from "./metadataScanStore"; +import { + loadMetadataScan, + metadataScanWanted, + settledIdsFrom, + type MetadataScanRun, +} from "./metadataScanStore"; import type { Paths } from "../lib/paths"; import { extractVideoId } from "../ytdlp/runYtdlp"; import { reconcileVideoDirs } from "./reconcileVideoDirs"; @@ -99,6 +104,21 @@ export type ChannelSnapshot = { // local lane is actually carrying the corpus. Optional: older snapshots lack // it; readers default to {}. digestEngines?: Record<string, number>; + // The metadata scan's state for this channel (see + // controller/metadataScanStore.ts). Beside `totals` for the digestEngines + // reason: buckets is a closed literal of `string[]` id lists. + // + // `unscanned` is the operation's BACKLOG — listed videos with nothing on disk + // that the scan has neither read nor recently failed on — and it is what + // /operations/metadata-scan offers a Run for, so it must mean exactly what + // metadataScanTargets() will fetch. Optional: older snapshots lack it, and + // readers must default to zeroes rather than to "nothing to do". + metadataScan?: { + scanned: number; + errors: number; + unscanned: number; + lastRun?: MetadataScanRun; + }; // Per-OPERATION work counts, keyed by kind id (see lib/operations.ts). // Beside `totals` and NOT in `buckets`, following the digestEngines precedent // above for the same reason: buckets is a closed literal of `string[]` id @@ -872,7 +892,8 @@ export async function generateChannelSnapshot( // out of everything the metadata scan has read. One file read per snapshot, // and nothing about it is stored per video: editing a pattern re-decides the // whole channel on the next report, with no rescan and nothing to migrate. - const settledIds = await settledByTitleFilterIds(paths, slug, config); + const metadataScanStore = await loadMetadataScan(paths, slug); + const settledIds = settledIdsFrom(metadataScanStore, config); const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; @@ -1215,6 +1236,12 @@ export async function generateChannelSnapshot( const undownloadedIds: string[] = []; const needsCookies: string[] = []; + // The metadata scan's backlog, counted in the same walk: a listed video with + // nothing on disk that the scan has not read and has not recently failed on. + // This is what /operations/metadata-scan offers a Run for, so it must mean + // exactly what metadataScanTargets() will fetch. + const scanNow = Date.now(); + let metadataScanUnscanned = 0; // Walked in playlist order, not sorted: this is the auto-download runner's // work queue, and the listing is newest-first. @@ -1225,6 +1252,15 @@ export async function generateChannelSnapshot( if (!dirId || !listedIdSet.has(dirId)) continue; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; + // Never fetched and never read: work for the metadata scan. Counted before + // the settlement check, because a settled video HAS been read — it is the + // scan's output, not its input. + if ( + !(f?.hasMeta ?? false) && + metadataScanWanted(metadataScanStore, dirId, scanNow) + ) { + metadataScanUnscanned++; + } // A settled video has no artifact and never will while the filter stands. // This is the line that makes the settlement STICK: undownloadedIds is the // auto-download runner's work queue, and it is derived from artifacts, so an @@ -1327,6 +1363,14 @@ export async function generateChannelSnapshot( downloaded, }, buckets: snapshotBuckets, + metadataScan: { + scanned: Object.keys(metadataScanStore.entries).length, + errors: Object.keys(metadataScanStore.errors).length, + unscanned: metadataScanUnscanned, + ...(metadataScanStore.lastRun + ? { lastRun: metadataScanStore.lastRun } + : {}), + }, digestEngines, backfill: { ...backfillCounts, ...bucketLaneEntries }, undownloadedIds, diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts @@ -257,17 +257,47 @@ export async function upsertMetadataScan( // which is inert at download time too) settles nothing, and an id we only have // an ERROR for is never settled — we do not know what it is called, so we must // not decide it. -export async function settledByTitleFilterIds( - paths: Paths, - slug: string, +export function settledIdsFrom( + scan: MetadataScan, config: Pick<ChannelConfig, "downloadFilter"> | null | undefined, -): Promise<Set<string>> { +): Set<string> { const settled = new Set<string>(); const compiled = compileDownloadFilter(config?.downloadFilter); if (!compiled) return settled; - const scan = await loadMetadataScan(paths, slug); for (const [id, entry] of Object.entries(scan.entries)) { if (titleFilterRejects(compiled, entry)) settled.add(id); } return settled; } + +// The loading form, for callers that have no scan in hand. +export async function settledByTitleFilterIds( + paths: Paths, + slug: string, + config: Pick<ChannelConfig, "downloadFilter"> | null | undefined, +): Promise<Set<string>> { + // The store is only opened when there is a filter to apply — a channel + // without one pays nothing for this being asked on every sync page. + if (!compileDownloadFilter(config?.downloadFilter)) return new Set(); + return settledIdsFrom(await loadMetadataScan(paths, slug), config); +} + +// How long a scan error suppresses a re-scan of that id. A members-only or +// deleted video errors every time; re-queueing it on every report would make the +// backlog never reach zero and the Run button never stop being offered. +export const METADATA_SCAN_ERROR_COOLDOWN_MS = 24 * 60 * 60 * 1000; + +// Is this id worth (re-)scanning right now? Entry present = no. A recent error +// = no, for a day. +export function metadataScanWanted( + scan: MetadataScan, + id: string, + now: number, +): boolean { + if (scan.entries[id]) return false; + const err = scan.errors[id]; + if (!err) return true; + const at = Date.parse(err.at); + if (!Number.isFinite(at)) return true; + return now - at >= METADATA_SCAN_ERROR_COOLDOWN_MS; +} diff --git a/editor/app/channels/[slug]/components/stages/PlaylistStage.tsx b/editor/app/channels/[slug]/components/stages/PlaylistStage.tsx @@ -6,15 +6,41 @@ import { QueueControl } from "../../../../components/QueueControl"; import { cancelJobAction } from "../../../../jobs/actions"; import { importVideoAction, + runMetadataScanAction, storePlaylistAction, syncAction, } from "../../pipelineActions"; +import type { MetadataScanRun } from "yt-dlp-transcript-common/controller/metadataScanStore"; + +// How many matched titles the stage lists. An eyeball check that the regex +// caught what the operator meant, not an inventory — the full answer is the +// channel's download list. +export const MATCHED_PREVIEW_LIMIT = 50; + +export type MetadataScanMatch = { + id: string; + uploadDate: string; + title: string; +}; + +export type MetadataScanView = { + listed: number; + scanned: number; + errors: number; + unscanned: number; + lastRun: MetadataScanRun | null; + filterConfigured: boolean; + filteredOut: number; + matchedTotal: number; + matched: MetadataScanMatch[]; +}; type Props = { slug: string; hasUrl: boolean; defaultQueueKey: string; existingQueues: string[]; + metadataScan: MetadataScanView; }; export function PlaylistStage({ @@ -22,11 +48,13 @@ export function PlaylistStage({ hasUrl, defaultQueueKey, existingQueues, + metadataScan, }: Props) { const [storeQueue, setStoreQueue] = useState(defaultQueueKey); const [syncQueue, setSyncQueue] = useState(defaultQueueKey); const [importQueue, setImportQueue] = useState(defaultQueueKey); const [importUrl, setImportUrl] = useState(""); + const [scanQueue, setScanQueue] = useState(defaultQueueKey); if (!hasUrl) { return ( @@ -93,6 +121,28 @@ export function PlaylistStage({ </div> <div className="flex flex-col gap-2"> <Heading + title="Metadata scan" + desc="Read the title, description and date of listed videos that aren't downloaded — no media, and no video directory is created. This is what the per-channel download filter needs in order to decide: without it, the only way to learn a video's title is to start downloading it." + /> + <MetadataScanSummary view={metadataScan} /> + <StreamActionLog + trigger={() => runMetadataScanAction(slug, scanQueue)} + cancelAction={cancelJobAction} + buttonLabel="Scan metadata" + runningLabel="Scanning metadata…" + extraControls={ + <QueueControl + value={scanQueue} + onChange={setScanQueue} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel="Scan metadata" + /> + } + /> + </div> + <div className="flex flex-col gap-2"> + <Heading title="Import single video" desc="Fetch one off-playlist video by URL into this channel, using the channel's normal handling (subtitles for YouTube, audio for transcribe). It's added to the archive like any other download." /> @@ -129,6 +179,83 @@ export function PlaylistStage({ ); } +function formatUploadDate(yyyymmdd: string): string { + return /^\d{8}$/.test(yyyymmdd) + ? `${yyyymmdd.slice(0, 4)}-${yyyymmdd.slice(4, 6)}-${yyyymmdd.slice(6, 8)}` + : "—"; +} + +const STOPPED_REASON: Record<NonNullable<MetadataScanRun["stopped"]>, string> = { + rate_limit: "stopped: the source rate-limited it — run it again once the cooldown lapses", + aborted: "stopped: cancelled", + error: "stopped: yt-dlp failed", +}; + +function MetadataScanSummary({ view }: { view: MetadataScanView }) { + return ( + <div + aria-label="metadata scan summary" + className="flex flex-col gap-1 rounded border border-border p-3 text-sm" + > + <p className="text-muted-foreground"> + Scanned <strong className="text-foreground">{view.scanned}</strong> of{" "} + {view.listed} listed + {view.errors > 0 ? `, ${view.errors} could not be read` : ""} + {view.unscanned > 0 ? ` · ${view.unscanned} still to scan` : ""} + </p> + {view.lastRun && ( + <p className="text-xs text-muted-foreground"> + Last run {new Date(view.lastRun.finishedAt).toLocaleString()} —{" "} + {view.lastRun.scanned} scanned, {view.lastRun.errors} error + {view.lastRun.errors === 1 ? "" : "s"} + {view.lastRun.stopped ? ` · ${STOPPED_REASON[view.lastRun.stopped]}` : ""} + </p> + )} + {view.filterConfigured ? ( + <> + <p aria-label="download filter verdict"> + Download filter:{" "} + <strong>{view.matchedTotal}</strong> match ·{" "} + <strong>{view.filteredOut}</strong> filtered out ·{" "} + <strong>{view.unscanned}</strong> unscanned + </p> + {view.matched.length > 0 && ( + <details className="text-sm"> + <summary className="cursor-pointer text-muted-foreground"> + Matched titles + {view.matchedTotal > view.matched.length + ? ` (newest ${view.matched.length} of ${view.matchedTotal})` + : ""} + </summary> + <ul + aria-label="matched titles" + className="mt-1 flex flex-col gap-0.5" + > + {view.matched.map((m) => ( + <li key={m.id} className="flex flex-wrap items-baseline gap-2"> + <span className="font-mono text-xs text-muted-foreground tabular-nums"> + {formatUploadDate(m.uploadDate)} + </span> + <span className="font-mono text-xs text-muted-foreground"> + {m.id} + </span> + <span>{m.title}</span> + </li> + ))} + </ul> + </details> + )} + </> + ) : ( + <p className="text-xs text-muted-foreground"> + No download filter configured on this channel — set one in Configure + &rsaquo; Advanced to use what the scan reads. + </p> + )} + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -68,7 +68,16 @@ import { liveJobRows } from "../../jobs/active/buildActiveJobs"; import { CleanupStage } from "./components/stages/CleanupStage"; import { DiagnosticsStage } from "./components/stages/DiagnosticsStage"; import { DownloadStage } from "./components/stages/DownloadStage"; -import { PlaylistStage } from "./components/stages/PlaylistStage"; +import { + PlaylistStage, + MATCHED_PREVIEW_LIMIT, + type MetadataScanMatch, +} from "./components/stages/PlaylistStage"; +import { loadMetadataScan } from "yt-dlp-transcript-common/controller/metadataScanStore"; +import { + compileDownloadFilter, + titleFilterRejects, +} from "yt-dlp-transcript-common/lib/downloadFilters"; import { TranscribeStage } from "./components/stages/TranscribeStage"; import { DigestStage } from "./components/stages/DigestStage"; import { SpeakersStage } from "./components/stages/SpeakersStage"; @@ -364,15 +373,44 @@ export default async function ChannelDetailPage({ /> ); } - case "playlist": + case "playlist": { + // THE METADATA SCAN'S VIEW. Built here rather than in a view module + // because it needs the store off disk, and `views/` may not touch + // node:fs. Two reads: one small JSON file and the playlist the page has + // already counted. + const scanStore = await loadMetadataScan(paths, slug); + const compiledFilter = compileDownloadFilter(config!.downloadFilter); + const matched: MetadataScanMatch[] = []; + let filteredOut = 0; + if (compiledFilter) { + for (const [id, e] of Object.entries(scanStore.entries)) { + if (titleFilterRejects(compiledFilter, e)) filteredOut++; + else matched.push({ id, uploadDate: e.uploadDate, title: e.title }); + } + // Newest first, and capped: the list is an eyeball check that the + // regex caught what the operator meant, not an inventory. + matched.sort((a, b) => b.uploadDate.localeCompare(a.uploadDate)); + } return ( <PlaylistStage slug={slug} hasUrl={!!config!.url} defaultQueueKey={platformDefaultQueueKey} existingQueues={existingQueues} + metadataScan={{ + listed: playlistCount ?? 0, + scanned: Object.keys(scanStore.entries).length, + errors: Object.keys(scanStore.errors).length, + unscanned: snapshot.metadataScan?.unscanned ?? 0, + lastRun: scanStore.lastRun, + filterConfigured: Boolean(compiledFilter), + filteredOut, + matchedTotal: matched.length, + matched: matched.slice(0, MATCHED_PREVIEW_LIMIT), + }} /> ); + } case "download": return ( <DownloadStage diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -25,6 +25,7 @@ import { import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; +import { runMetadataScan } from "yt-dlp-transcript-common/ytdlp/metadataScan"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates"; import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; @@ -315,6 +316,64 @@ export async function syncAction( }); } +// THE METADATA SCAN. Not routed through runPipelineAction, and the differences +// are the whole reason it is its own operation: +// +// - It is NOT held by the downloads pause. It fetches no media, and the +// operator runs it precisely to decide what a paused download lane should +// fetch when it resumes. +// - It needs no disk gate, for the same reason: the only thing it writes is +// one JSON file of titles. +// - It DOES respect the per-platform rate-limit cooldown, like sync, and +// records one of its own when the source pushes back — the scan is one +// request per listed video and is the most rate-limitable thing here. +export async function runMetadataScanAction( + slug: string, + queueKey?: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const channelConfig = await readChannelConfig(paths, slug); + if (!channelConfig) { + return { ok: false, error: `Channel "${slug}" not found` }; + } + if (!channelConfig.url) { + return { ok: false, error: "Channel has no `url` configured" }; + } + const platform = detectPlatform(channelConfig.url) ?? "unknown"; + const remainingMs = await platformCooldownRemainingMs(platform, paths); + if (remainingMs > 0) { + const secs = Math.ceil(remainingMs / 1000); + return { + ok: false, + info: true, + error: + `${platform} is in a rate-limit cooldown (${secs}s remaining). ` + + `The metadata scan will run once the cooldown lapses.`, + }; + } + return runManagedFunction({ + kind: "metadata-scan", + queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey), + paths, + channelSlug: slug, + spec: { kind: "metadata-scan", slug, params: { queueKey } }, + fn: async (onLog, signal, setProgress) => { + await runMetadataScan({ + paths, + channelSlug: slug, + channelConfig, + onLog, + signal, + setProgress, + onPlatformBackoff: () => recordDownloadBackoff(platform, paths), + }); + revalidatePath(`/channels/${slug}`); + revalidatePath("/channels"); + revalidatePath("/operations/[id]", "page"); + }, + }); +} + export async function downloadMissingSubsAction( slug: string, queueKey?: string, diff --git a/editor/app/components/actions/InlineActionButton.tsx b/editor/app/components/actions/InlineActionButton.tsx @@ -3,7 +3,10 @@ import { useState, useTransition } from "react"; import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand"; import type { AudioFormat } from "yt-dlp-transcript-common/lib/channelConfig"; -import { downloadMissingAction } from "../../channels/[slug]/pipelineActions"; +import { + downloadMissingAction, + runMetadataScanAction, +} from "../../channels/[slug]/pipelineActions"; import { cleanAudioAction, cleanExtraAudioFormatsAction, @@ -20,6 +23,7 @@ import { refreshChannelSnapshotAction } from "../../channels/actions"; type Variant = | { kind: "downloadMissing"; slug: string } + | { kind: "metadataScan"; slug: string } | { kind: "transcribeMissing"; slug: string; audioFormat?: AudioFormat } | { kind: "redownloadIncomplete"; slug: string } | { kind: "clearIncomplete"; slug: string } @@ -41,6 +45,7 @@ type Status = const LABEL: Record<Variant["kind"], { idle: string; running: string }> = { downloadMissing: { idle: "Download missing", running: "Queuing…" }, + metadataScan: { idle: "Scan metadata", running: "Queuing…" }, transcribeMissing: { idle: "Transcribe pending", running: "Queuing…" }, redownloadIncomplete: { idle: "Re-download & re-transcribe", running: "Queuing…" }, clearIncomplete: { idle: "Clear & re-queue", running: "Clearing…" }, @@ -71,6 +76,9 @@ async function runAction(variant: Variant): Promise<StreamActionResult> { if (variant.kind === "downloadMissing") { return downloadMissingAction(variant.slug); } + if (variant.kind === "metadataScan") { + return runMetadataScanAction(variant.slug); + } if (variant.kind === "transcribeMissing") { return transcribeMissingAction( variant.slug, diff --git a/editor/app/components/channelWork/sections.test.ts b/editor/app/components/channelWork/sections.test.ts @@ -21,6 +21,7 @@ function distinctSummary(): ActionableSummary { return { rows: fresh(), undownloaded: fresh(), + metadataScan: fresh(), missingNeverFetched: fresh(), untranscribed: fresh(), incompleteTranscripts: fresh(), @@ -54,6 +55,10 @@ test("sectionsFor groups the sections by the page that renders them", () => { ["undownloaded", "missing-never-fetched", "short-audio"], ); assert.deepEqual( + sectionsFor("metadata-scan").map((s) => s.id), + ["metadata-scan"], + ); + assert.deepEqual( sectionsFor("transcription").map((s) => s.id), ["untranscribed", "incomplete-transcripts"], ); @@ -68,7 +73,13 @@ test("sectionsFor groups the sections by the page that renders them", () => { }); test("sectionsFor puts work before attention", () => { - for (const op of ["download", "transcription", "digest", null] as const) { + for (const op of [ + "download", + "metadata-scan", + "transcription", + "digest", + null, + ] as const) { const roles = sectionsFor(op).map((s) => s.role); assert.deepEqual(roles, [...roles].sort((a, b) => (a === b ? 0 : a === "work" ? -1 : 1))); } diff --git a/editor/app/components/channelWork/sections.tsx b/editor/app/components/channelWork/sections.tsx @@ -7,6 +7,7 @@ import { actionableCleanTranscribedCount, actionableDigestWarningsCount, actionableIncompleteTranscriptCount, + actionableMetadataScanCount, actionableMissingNeverFetchedCount, actionableShortAudioCount, actionableUndownloadedCount, @@ -27,7 +28,7 @@ export type SectionConfig = { // The page that renders this section: an operation id for sections that are // one operation's work, or null for the two cleanup sections (Storage, // /cleanup's). - operation: "download" | "transcription" | "digest" | null; + operation: "download" | "transcription" | "digest" | "metadata-scan" | null; // "work" is what the runner or sweep will do; "attention" is what a human // must look at first — a truncated download is not one the runner can retry. role: "work" | "attention"; @@ -73,6 +74,23 @@ export function channelWorkSections(): SectionConfig[] { ), }, { + id: "metadata-scan", + operation: "metadata-scan", + role: "work", + getRows: (s) => s.metadataScan, + title: "Channels whose download filter is waiting on titles", + description: + "These channels have a download filter and listed videos nobody has read the title of, so the filter cannot decide about them yet. The scan fetches title, description and date — no media, and no video directory — and every non-match is then settled without ever being downloaded.", + countLabel: "unscanned", + emptyLabel: "Every filtered channel is fully scanned.", + getCount: actionableMetadataScanCount, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "metadataScan", slug: r.channel.slug }} + /> + ), + }, + { id: "missing-never-fetched", operation: "download", role: "attention", diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -18,6 +18,7 @@ import { downloadMissingSubsAction, retryBucketAction, storePlaylistAction, + runMetadataScanAction, syncAction, } from "../channels/[slug]/pipelineActions"; import { @@ -184,6 +185,13 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { const { p, queueKey } = params(spec); return syncAction(spec.slug, queueKey, bool(p.fullSweep)); }, + // Re-derives its targets from the channel's CURRENT state, like every other + // replay here: a re-run scans whatever is still unscanned now, not the list + // the first run was given. + "metadata-scan": (spec) => { + const { queueKey } = params(spec); + return runMetadataScanAction(spec.slug, queueKey); + }, "check-post-availability": (spec) => { const { p, queueKey } = params(spec); return checkPostAvailabilityAction( diff --git a/editor/app/lib/actionable/loadActionable.ts b/editor/app/lib/actionable/loadActionable.ts @@ -18,6 +18,7 @@ export type ActionableRow = { export type ActionableSummary = { rows: ActionableRow[]; undownloaded: ActionableRow[]; + metadataScan: ActionableRow[]; missingNeverFetched: ActionableRow[]; untranscribed: ActionableRow[]; incompleteTranscripts: ActionableRow[]; @@ -79,6 +80,16 @@ export function actionableMissingNeverFetchedCount(row: ActionableRow): number { return row.snapshot?.missingNeverFetched?.length ?? 0; } +// Listed videos the metadata scan has neither read nor recently failed on — +// the scan operation's backlog, counted at snapshot-generation time so this +// costs no extra read. Deliberately NOT run through countActionable: these +// videos have no directory, so the availability exclusions (which are keyed on +// what is on disk) cannot say anything about them. Default 0 for snapshots +// written before the field existed. +export function actionableMetadataScanCount(row: ActionableRow): number { + return row.snapshot?.metadataScan?.unscanned ?? 0; +} + export function actionableUntranscribedCount(row: ActionableRow): number { return countActionable( row.snapshot, @@ -165,6 +176,21 @@ export async function loadActionableSummary( actionableMissingNeverFetchedCount(a), ); + // Only channels that HAVE a download filter: the scan is useful on any + // channel, but it is only work that needs doing on one whose filter is waiting + // on titles it has not read. Offering it everywhere would put every channel in + // the corpus on this page permanently. + const metadataScan = rows + .filter( + (r) => + actionableMetadataScanCount(r) > 0 && + Boolean( + r.channel.config.downloadFilter?.include?.trim() || + r.channel.config.downloadFilter?.exclude?.trim(), + ), + ) + .sort((a, b) => actionableMetadataScanCount(b) - actionableMetadataScanCount(a)); + const untranscribed = rows .filter((r) => actionableUntranscribedCount(r) > 0) .sort( @@ -208,6 +234,7 @@ export async function loadActionableSummary( return { rows, undownloaded, + metadataScan, missingNeverFetched, untranscribed, incompleteTranscripts,