// The METADATA SCAN store: what a channel's listed-but-not-downloaded videos // are CALLED, without downloading them. `channels//metadata-scan.json`. // // WHY THIS IS A CHANNEL-LEVEL FILE AND NOT data//metadata.info.json. // Two things downstream read the presence of a video directory as a fact about // the corpus, and both would be wrong if a metadata-only scan created one: // // 1. common/controller/buildIndex.ts:309-315 admits ANY `data//` that has // a metadata.info.json, transcript or not. A scan over a filtered channel // would put ~1,800 transcript-less videos into the LMDB index and into the // exported site — a published archive full of entries with no text. // 2. deriveChannelSets (./channelSets.ts:58-67, fed at // channelSnapshot.ts:1167-1174) keys "ever fetched" off the DIR NAMES under // data/. A scanned id would read as previously fetched and drop out of // `missingNeverFetched`, which is the one place a video that vanished // before we ever got it is recorded. // // So the scan never creates a video directory. One file per channel, additive, // and nothing derived is stored in it: whether a video is SETTLED by the // download filter is computed from these entries plus the channel's CURRENT // config every time it is asked (see settledByTitleFilterIds below). That is // what makes editing the regex free — no rescan, no migration, no stored // verdict to invalidate. // // Modelled on ./rosterStore.ts: versioned, normalize-on-read (unknown fields // dropped), additive merge that returns the same object when nothing changed, // tmp+rename write (lib/jsonFile-server.ts). import path from "node:path"; import { readFile } from "node:fs/promises"; import { writeJsonAtomic } from "../lib/jsonFile-server"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; import { compileDownloadFilter, titleFilterRejects, titleFilterWantsChat, } from "../lib/downloadFilters"; export const METADATA_SCAN_FILENAME = "metadata-scan.json"; export const METADATA_SCAN_VERSION = 1; // One scanned video. `description` is kept because it is half of the text the // download filter matches against — storing only the title would make a // re-match against an edited pattern disagree with what the downloader does. export type MetadataScanEntry = { title: string; description: string; // yt-dlp's `upload_date` (YYYYMMDD), or "" when the extractor had none. uploadDate: string; liveStatus?: string; duration?: number; // THE CHAT-ONLY DEAD END. Set when a chat-only pass ran cleanly and the // source returned no live chat — replay was off for that stream, or it has // since been dropped. Asking again tomorrow gets the same nothing, so this is // an ANSWER and not a retry: `chatOnlyIdsFrom` drops the id, it leaves // `chatOnlyPending` for good, and it settles as the ordinary filtered-out // livestream it is. Deliberately NOT set by a pass that FAILED (a non-zero // exit, a cooldown, an abort) — that one has to stay retryable. noLiveChat?: boolean; scannedAt: string; }; // A video the scan could not read. Kept so a re-run knows what to retry and the // UI can say why the numbers don't add up. `class` is an Availability class // where one could be parsed, else "unknown". export type MetadataScanError = { class: string; message: string; at: string; }; // Why the last run ended. `stopped` absent = it finished its target list. export type MetadataScanRun = { startedAt: string; finishedAt: string; scanned: number; errors: number; stopped?: "rate_limit" | "aborted" | "error"; message?: string; }; export type MetadataScan = { version: number; updatedAt: string; lastRun: MetadataScanRun | null; entries: Record; errors: Record; }; export function emptyMetadataScan(): MetadataScan { return { version: METADATA_SCAN_VERSION, updatedAt: "", lastRun: null, entries: {}, errors: {}, }; } export function metadataScanPath(paths: Paths, slug: string): string { return path.join(paths.channelsDir, slug, METADATA_SCAN_FILENAME); } function normalizeEntry(raw: unknown): MetadataScanEntry | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; if (typeof r.scannedAt !== "string") return null; const entry: MetadataScanEntry = { title: typeof r.title === "string" ? r.title : "", description: typeof r.description === "string" ? r.description : "", uploadDate: typeof r.uploadDate === "string" ? r.uploadDate : "", scannedAt: r.scannedAt, }; if (typeof r.liveStatus === "string") entry.liveStatus = r.liveStatus; if (typeof r.duration === "number" && Number.isFinite(r.duration)) { entry.duration = r.duration; } if (r.noLiveChat === true) entry.noLiveChat = true; return entry; } function normalizeError(raw: unknown): MetadataScanError | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; if (typeof r.at !== "string") return null; return { class: typeof r.class === "string" ? r.class : "unknown", message: typeof r.message === "string" ? r.message : "", at: r.at, }; } function normalizeRun(raw: unknown): MetadataScanRun | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record; if (typeof r.startedAt !== "string" || typeof r.finishedAt !== "string") { return null; } const run: MetadataScanRun = { startedAt: r.startedAt, finishedAt: r.finishedAt, scanned: typeof r.scanned === "number" ? r.scanned : 0, errors: typeof r.errors === "number" ? r.errors : 0, }; if ( r.stopped === "rate_limit" || r.stopped === "aborted" || r.stopped === "error" ) { run.stopped = r.stopped; } if (typeof r.message === "string") run.message = r.message; return run; } export async function loadMetadataScan( paths: Paths, slug: string, ): Promise { try { const raw = await readFile(metadataScanPath(paths, slug), "utf8"); const parsed = JSON.parse(raw) as Record; if (!parsed || typeof parsed !== "object") return emptyMetadataScan(); const entries: Record = {}; const rawEntries = (parsed.entries ?? {}) as Record; for (const [id, value] of Object.entries(rawEntries)) { const entry = normalizeEntry(value); if (entry) entries[id] = entry; } const errors: Record = {}; const rawErrors = (parsed.errors ?? {}) as Record; for (const [id, value] of Object.entries(rawErrors)) { const err = normalizeError(value); if (err) errors[id] = err; } return { version: METADATA_SCAN_VERSION, updatedAt: typeof parsed.updatedAt === "string" ? parsed.updatedAt : "", lastRun: normalizeRun(parsed.lastRun), entries, errors, }; } catch { return emptyMetadataScan(); } } async function writeMetadataScan( paths: Paths, slug: string, scan: MetadataScan, ): Promise { // UNIQUE PER WRITE, not per process, and chained per path: two overlapping // writers in the SAME process once shared a pid-named tmp and the second // rename threw ENOENT, failing the job. The scan serializes its own flushes, // but the download path writes here too. The shared writer's temp name and // globalThis chain replace this module's own counter, which was per module // COPY. await writeJsonAtomic(metadataScanPath(paths, slug), scan); } export type MetadataScanUpsert = { entries?: Record; errors?: Record; // Replaces `lastRun` wholesale. Omitted leaves the previous run record alone, // which is what an incremental mid-run flush wants. lastRun?: MetadataScanRun; }; // Fold new records into the store and persist. ADDITIVE: an id never disappears // here, and a successful entry CLEARS that id's error (the video was readable // after all). Skips the write entirely when nothing changed, so a re-run over an // already-scanned channel costs one read. export async function upsertMetadataScan( paths: Paths, slug: string, upsert: MetadataScanUpsert, now: string, ): Promise { const scan = await loadMetadataScan(paths, slug); let changed = false; for (const [id, entry] of Object.entries(upsert.entries ?? {})) { const prev = scan.entries[id]; if ( !prev || prev.title !== entry.title || prev.description !== entry.description || prev.uploadDate !== entry.uploadDate || prev.liveStatus !== entry.liveStatus || prev.duration !== entry.duration || prev.noLiveChat !== entry.noLiveChat ) { scan.entries[id] = entry; changed = true; } // A readable video is not an error, whatever it was last time. if (scan.errors[id]) { delete scan.errors[id]; changed = true; } } for (const [id, err] of Object.entries(upsert.errors ?? {})) { // An id we already have metadata for stays scanned — a transient failure on // a later pass must not un-scan it. if (scan.entries[id]) continue; const prev = scan.errors[id]; // An IDENTICAL error still rewrites once its `at` is a cooldown old: `at` // is what metadataScanWanted reads, so leaving it stale re-queues the id // at every runner start forever (142 members-only videos re-scanned, // cookie-authed, at each start). Refreshed, the id rests for another day. if ( !prev || prev.class !== err.class || prev.message !== err.message || Date.parse(now) - Date.parse(prev.at) >= METADATA_SCAN_ERROR_COOLDOWN_MS ) { scan.errors[id] = err; changed = true; } } if (upsert.lastRun) { scan.lastRun = upsert.lastRun; changed = true; } if (!changed) return scan; scan.updatedAt = now; await writeMetadataScan(paths, slug, scan); return scan; } // SETTLED, derived. The ids this channel's CURRENT download filter rejects, // out of everything the scan has read. The one place any of the four consumers // (snapshot bucket, undownloadedIds, sync page walk, batch exclusion) get the // answer, and it is recomputed from the config every time — which is why // editing a pattern re-decides the whole channel on the next snapshot with no // rescan and nothing to migrate. // // FAILS CLOSED in both directions: no filter configured (or an unparseable one, // which is inert at download time too) settles nothing, and an id we only have // an ERROR for is never settled — we do not know what it is called, so we must // not decide it. export function settledIdsFrom( scan: MetadataScan, config: Pick | null | undefined, ): Set { const settled = new Set(); const compiled = compileDownloadFilter(config?.downloadFilter); if (!compiled) return settled; for (const [id, entry] of Object.entries(scan.entries)) { if (titleFilterRejects(compiled, entry)) settled.add(id); } return settled; } // THE CHAT-ONLY SUBSET OF THE SETTLED SET — a strict subset, never a second // population. `titleFilterRejects` answers true for a chat-only video (see // DownloadFilterVerdict), so every id here is also in `settledIdsFrom`'s // answer; what this adds is which of them the operator wants the live chat of. // // Derived from the config every time, exactly like settledIdsFrom, so turning // `rejectedLivestreams` off again re-decides the channel with no rescan and // nothing stored per video to undo. export function chatOnlyIdsFrom( scan: MetadataScan, config: Pick | null | undefined, ): Set { const ids = new Set(); const compiled = compileDownloadFilter(config?.downloadFilter); // The mode only ever means something alongside a real filter, and a channel // that is not chat-only pays one field read for this question. if (!compiled || compiled.rejectedLivestreams !== "chat-only") return ids; for (const [id, entry] of Object.entries(scan.entries)) { // A stream we already asked and got nothing from is not chat-only work any // more — it is an ordinary settled rejection. Without this it sits in // chatOnlyPending forever and every runner restart re-prefetches it to run // a pass that will never return anything. if (entry.noLiveChat) continue; if (titleFilterWantsChat(compiled, entry)) ids.add(id); } return ids; } // The loading form, for callers that have no scan in hand. export async function settledByTitleFilterIds( paths: Paths, slug: string, config: Pick | null | undefined, ): Promise> { // The store is only opened when there is a filter to apply — a channel // without one pays nothing for this being asked on every sync page. if (!compileDownloadFilter(config?.downloadFilter)) return new Set(); return settledIdsFrom(await loadMetadataScan(paths, slug), config); } // How long a scan error suppresses a re-scan of that id. A members-only or // deleted video errors every time; re-queueing it on every report would make the // backlog never reach zero and the Run button never stop being offered. export const METADATA_SCAN_ERROR_COOLDOWN_MS = 24 * 60 * 60 * 1000; // Is this id worth (re-)scanning right now? Entry present = no. A recent error // = no, for a day. export function metadataScanWanted( scan: MetadataScan, id: string, now: number, ): boolean { if (scan.entries[id]) return false; const err = scan.errors[id]; if (!err) return true; const at = Date.parse(err.at); if (!Number.isFinite(at)) return true; return now - at >= METADATA_SCAN_ERROR_COOLDOWN_MS; }