import { writeJsonAtomic } from "../lib/jsonFile-server"; import path from "node:path"; import type { Dirent } from "node:fs"; import { lstat, readdir, readFile, stat } from "node:fs/promises"; import pLimit from "p-limit"; import { readArchive } from "../lib/archive"; import { isVideoDownloaded, isVideoFetched, isVideoTranscribed, readVideoFiles, LIVE_CHAT_FILENAME, ORIG_VTT_FILENAME, VTT_FILENAME, type VideoFiles, } from "../lib/videoStatus"; import { AUTH_RETRY_CLASSES, AVAILABILITY_VALUES, EXCLUDED_FROM_DOWNLOAD, isPermanentlyGone, type Availability, } from "../lib/availability"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { assertChannelTextReadable } from "../lib/channelMedia"; import { isTierable } from "../lib/mediaTier"; import { isDriveNotAnswering, onDrive } from "../lib/storageHealth"; import { CLIPS_DIR_NAME } from "../lib/clipWindow"; import { getSettings } from "../lib/settings"; import { loadAvailability, resolveEffectiveAvailability, } from "../lib/availability-server"; import { isDoNotClean } from "../lib/doNotClean-server"; import { loadDigest } from "../lib/digest-server"; import { addOperationState, allOperations, bucketLaneOperationId, emptyOperationCounts, presentOperationWork, reachableOperationWork, DIGEST_OPERATION_ID, DIARIZATION_OPERATION_ID, type OperationClassification, type OperationSnapshotEntry, } from "../lib/operations"; import { LANES, type AutoQueueKind } from "../lib/autoQueueTypes"; import { bucketIdsFrom, bucketLaneWorkIds, type BucketSource, } from "../jobs/autoQueuePolicy"; import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; import { chatOnlyIdsFrom, loadMetadataScan, metadataScanWanted, settledIdsFrom, type MetadataScanRun, } from "./metadataScanStore"; import type { Paths } from "../lib/paths"; import { extractVideoId } from "../ytdlp/runYtdlp"; import { reconcileVideoDirs } from "./reconcileVideoDirs"; import { loadFailedTranscriptions } from "./failedTranscriptions"; import { readChannelConfig, readChannelSnapshot, snapshotPath, SNAPSHOT_FILENAME, } from "./channels"; import { computeKeptVideoIds } from "./keptVideos"; import { loadMaybeMissing } from "./quickAvailabilityCheck"; import { loadRoster } from "./rosterStore"; import { deriveChannelSets } from "./channelSets"; import { readTranscriptCoverage } from "./normalizeTranscript"; import { resolveCaptionsProvenance } from "../lib/subtitleProvenance"; import { isIncompleteTranscript } from "../lib/transcriptCoverage"; // THE PER-OPERATION WORK LISTS, keyed by operation id. // // The name the map has wanted since it stopped being the backfill lane: it is // `operations`, and since slice 1.5 it holds one entry for EVERY lane — // `digest`, the three speaker kinds, and now `download` and `transcription`. // The DISK KEY is still `backfill`, and stays: renaming it would make every // snapshot on disk unreadable to say something a type alias says for free. export type ChannelSnapshotOperations = Record; export type AvailabilitySnapshot = { byStatus: Record; unchecked: string[]; }; export type ExcludedFromDownload = { membersOnly: string[]; deleted: string[]; private: string[]; }; export type ChannelSnapshot = { generatedAt: string; totals: { videos: number; transcribed: number; downloaded: number; }; // Digest coverage by engine: appId -> count of videos whose ai-digest.json was // produced by it. Beside `totals`, NOT in `buckets` — buckets are a closed // literal of `string[]` id lists and a Record does not belong // there. During a multi-week sweep this split is what tells you whether the // local lane is actually carrying the corpus. Optional: older snapshots lack // it; readers default to {}. digestEngines?: Record; // The metadata scan's state for this channel (see // controller/metadataScanStore.ts). Beside `totals` for the digestEngines // reason: buckets is a closed literal of `string[]` id lists. // // `unscanned` is the operation's BACKLOG — listed videos with nothing on disk // that the scan has neither read nor recently failed on — and it is what // /operations/metadata-scan offers a Run for, so it must mean exactly what // metadataScanTargets() will fetch. Optional: older snapshots lack it, and // readers must default to zeroes rather than to "nothing to do". metadataScan?: { scanned: number; errors: number; unscanned: number; lastRun?: MetadataScanRun; }; // Per-OPERATION work counts, keyed by kind id (see lib/operations.ts). // Beside `totals` and NOT in `buckets`, following the digestEngines precedent // above for the same reason: buckets is a closed literal of `string[]` id // lists, and this is per-kind counts. // // THE NUMBERS ARE NEVER SUMMED. `missing` + `stale` + `partial` is work the // lane can do today; `missingInput` needs the media re-acquired and, measured // on this corpus, is 91x larger. A single "remaining" figure here would put // every channel permanently at the top of every list — which is the documented // reason /api/widget/actionable refuses to filter on the digest work count. // // THIS MAP IS NO LONGER THE BACKFILL LANE, and anything reading it generically // must say which lane it means. It is written from allOperations — every // enabled operation in the catalog — because a work list the snapshot does not // carry is a work list nothing can select, count or schedule. The digest // operation runs on its own queue key and is the first entry here that the // backfill lane must not touch. Read it with backfillLaneEntriesOf() to get the lane, // or by id to get one operation; a bare Object.values() over this map now // means "every operation", which on this corpus is a ~75,000-video difference. // // SINCE SLICE 1.5 IT CARRIES EVERY LANE'S WORK LIST, the two bucket lanes // included: `backfill.download` and `backfill.transcription` are the fold of // those lanes' default buckets (see bucketLaneWorkIds), so "which videos does // lane X have to do" has ONE answer shape for all four lanes. Those two are // written by hand rather than by state() — they have no registry entry — and // the generator says how. // // Optional: snapshots written before this existed lack it, and readers default // to {}. The two new entries are optional in the same way and for a stronger // reason: no live snapshot is regenerated by the slice that added them, so // every reader falls back to the bucket fold until a channel's next regen. backfill?: ChannelSnapshotOperations; buckets: { noTranscript: string[]; downloadedNoTranscript: string[]; // Has audio on disk, none of it in the channel's audioFormat — the orphan // half of what "Remove wrong-format audio" deletes (the other half is // multipleAudioFormats: target present plus extras). Was `untranscoded` // until 2026-08-30, when the transcode operation that read it as a to-do // list was removed; the population itself is still the sweep's. Old // snapshots keep a stray `untranscoded` key and read this one as [] until // their next regen — no schema version, per-field optionality, as always. wrongFormatAudio: string[]; multipleAudioFormats: string[]; // Dirs with a whisper transcript AND audio still on disk — the audio is // redundant and can be cleaned. Mirrors what cleanAudioFromTranscribed // removes. Excludes "do not clean"–marked dirs. Optional: snapshots written // before this field existed lack it; readers must default to []. transcribedWithAudio: string[]; untranscribable: string[]; noMetadata: string[]; failedListed: string[]; missingFromArchive: string[]; duplicateDirs: string[]; partialDownloads: string[]; // Videos whose latest download-outcome.json is "failed-corrupt-source" (the // audio-check pipeline gave up on a malformed source) and that still have no // transcript. Surfaced as a download/source problem — NOT counted as a failed // transcription. Optional: older snapshots lack it; readers default to []. corruptSource: string[]; // Videos whose download COMPLETED (yt-dlp exit 0, all bytes) but whose final // audio integrity probe stayed malformed even after one re-download // (download-outcome.json "corrupt-full-source"). The downloaded container is // KEPT on disk for inspection; re-downloading is futile, so this is terminal // and informational — NOT counted as a failed transcription and NOT re-queued // for download. Optional: older snapshots lack it; readers default to []. corruptFullSource: string[]; // Videos whose transcript rides on a non-canonical VTT name (e.g. only // transcript.en-US.vtt, or a foreign-only transcript..vtt) instead of // the standard transcript.en.vtt — surfaced so the user can normalize/switch // the primary transcript. Excludes videos that already have a whisper // transcript or the canonical transcript.en.vtt. Optional: older snapshots // lack it; readers must default to []. nonStandardVtt: string[]; // Videos whose latest download-outcome.json is "skipped-filtered" (e.g. // skip-live) and that still have no downloaded artifact — i.e. they were // declined as currently-live/upcoming and will be retried on a later sync. // Optional: older snapshots lack it; readers must default to []. skippedByFilter: string[]; // Videos SETTLED by the per-channel download filter: a terminal // "skipped-filtered" outcome whose recorded signature still equals the // channel's current include/exclude (isSettledByFilter). Distinct from // skippedByFilter above, which is the RETRYABLE skip — that bucket says // "we'll try again", this one says "the operator asked us not to". // // These ids are short-circuited out of every other bucket, out of // undownloadedIds, and out of totals.videos: a filtered channel accumulates // one metadata stub per non-match (~1,800 on the channel this was built // for), and counting them as videos would make every ratio on every page // wrong. Change either pattern and the signature stops matching, so they // re-appear as ordinary undownloaded videos on the next report. // Optional: older snapshots lack it; readers must default to []. skippedByTitleFilter: string[]; // CHAT-ONLY videos whose chat is on disk. A rejected livestream on a channel // whose `downloadFilter.rejectedLivestreams` is "chat-only": the media was // never fetched, the live chat was, and the directory holds // metadata.info.json, transcript.live_chat.json and its normalized // live_chat.cues.json, beside the download-outcome.json and download.log // every managed download leaves. No media, and no transcript. // // THE FILE ON DISK DECIDES, not the mode: an id whose chat has landed stays // in this bucket after `rejectedLivestreams` is set back to "skip", // because buildIndex has already published it and flipping a setting does // not un-publish it. // // A MEMBER OF THIS CORPUS, not a stub. It is in totals.videos, it is in the // LMDB index, and the site publishes it as a chat track with no captions. // It is deliberately NOT in skippedByTitleFilter (that bucket means "we // decided not to have it"), NOT in noTranscript or downloadedNoTranscript // (nothing is going to transcribe a video whose audio we chose not to // fetch), and NOT in undownloadedIds. Optional: older snapshots lack it; // readers must default to [] — which normalizeBuckets does, like every // other bucket declared required here and absent from older files. chatOnly: string[]; // The same population, chat NOT yet on disk — one of the download lane's // buckets (DOWNLOAD_BUCKETS). It cannot ride in undownloadedIds: that list // means "fetch the media", and fetching the media is the one thing this // video must not have done to it. Absent from older files in the same way. chatOnlyPending: string[]; // Videos whose transcript covers only a small fraction of the video's // duration — the audio download silently truncated (yt-dlp exited "ok") so // whisper transcribed just the first few minutes. The detection threshold // lives in ../lib/transcriptCoverage (isIncompleteTranscript). Surfaced so // the user can re-download/re-transcribe. Optional: older snapshots lack it; // readers must default to []. incompleteTranscript: string[]; // Videos whose download COMPLETED (yt-dlp exit 0) but whose audio is far // shorter than the metadata duration — the source served a truncated stream // (download-outcome.json "failed-short-audio"). Caught at download time by // the duration guard BEFORE transcription. The short file is KEPT on disk // (so it isn't re-downloaded into a loop) and excluded from // downloadedNoTranscript (so it isn't auto-transcribed). Surfaced so the // user can re-download with a different format (e.g. Original). Optional: // older snapshots lack it; readers must default to []. shortAudio: string[]; // Videos whose ONLY transcript is YouTube's speech recognition (an English // VTT that sniffs as `asr` — see ../lib/subtitleProvenance) and that have no // audio on disk yet. The opt-in "replace auto-captions" lane's FIRST step: // they need an audio download before our own engine can transcribe them. // Mirrors noTranscript (no audio yet) and, like it, excludes videos that are // excluded from download, untranscribable, or terminally failed. A VTT whose // provenance can't be proven machine-generated is never listed. Optional: // older snapshots lack it; readers must default to []. autoSubsOnly: string[]; // Same candidate rule as autoSubsOnly, but audio IS on disk — the lane's // SECOND step, consumed by the opt-in auto-transcribe lane and the channel // page's "YouTube auto-captions only" section. Mirrors // downloadedNoTranscript. Optional: older snapshots lack it; readers must // default to []. downloadedAutoSubsOnly: string[]; // Videos that now have OUR transcript (transcript.json) while the superseded // English ASR VTT is still on disk as a backup. Never cleaned automatically — // this is the inventory behind the Cleanup stage's manual purge button (and // the ready-made worklist for a future AI-vs-YouTube comparison). Excludes // do-not-clean–marked dirs so the count matches what the purge would remove. // Optional: older snapshots lack it; readers must default to []. supersededAutoSubs: string[]; // Undownloaded playlist videos whose effective availability says browser // cookies could recover them (AUTH_RETRY_CLASSES: needs_auth, members_only, // private). Populated in EVERY cookie mode — members_only/private are // batch-excluded regardless, and a subscribed/owning account's cookies can // still fetch them — and drives the "Download with cookies" bucket button // (retry-bucket with forceCookies). Optional: older snapshots lack it; // readers must default to []. needsCookies: string[]; // Videos whose digest pass recorded something a human should look at: // either warnings alongside a section that WAS written, or a total failure // that wrote no section at all (DigestRecord.failures). // // Both kinds matter and the second is the one that used to be invisible — // a total failure deliberately writes no section so the video retries, and // before `failures` existed its warnings survived only in a rotating job // log. A review queue keyed on written warnings alone would have been blind // to precisely the worst outputs. // // Costs no extra I/O: the sidecar is already loaded here for // `digestEngines` and `digestWarnings`. // Optional: older snapshots lack it; readers must default to []. digestWarnings?: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; // Count of videos in the channel's keep-latest window (ChannelConfig.keepLatest). // These are protected from the Clean-audio sweep (folded into the cleanup- // exclusion set alongside do-not-clean markers) and targeted for source-video // persistence. Optional: older snapshots lack it; readers must default to 0. keptCount?: number; availability?: AvailabilitySnapshot; // Known videos absent from the channel's most recent fresh flat-playlist // fetch (the quick availability check). The maybe-missing.json sidecar is the // source of truth; this re-surfaces it so it survives snapshot regens. ids are // intersected with on-disk dirs at generation time. Optional: older snapshots // lack it; readers must default via normalizeMaybeMissing. maybeMissing?: { ids: string[]; checkedAt: string }; // Videos the channel's roster says we were told about, never downloaded, and // that the current listing no longer contains. The app could not express this // before: with no data// dir there was nothing to notice their absence // against, and `playlist` — the only place their URL lived — was overwritten // wholesale by every sweep. // // The stored URL is what makes the bucket actionable rather than merely // defensive: it is enough to attempt a direct-link download even though the // video has left the channel. Optional: older snapshots lack it; readers must // default to []. missingNeverFetched?: MissingNeverFetched[]; // Estimated bytes each cleanup operation would reclaim, mirroring the bucket // counts: transcribedWithAudio sums every audio.* file in those dirs (all are // removed); multipleAudioFormats sums the non-target audio.* files (the target // format is kept). Optional: snapshots written before this field existed lack // it; readers must default to 0. cleanupBytes?: { transcribedWithAudio: number; multipleAudioFormats: number; // Sum of every non-target audio.* file across dirs that have one (orphans // with no target file + dirs with target plus extras) — what the // "remove wrong-format audio" sweep reclaims. Optional: older snapshots // lack it; readers must default to 0. foreignAudio?: number; }; // Every audio.* byte this channel has on disk, held or not. The denominator // the four gates below partition. Optional: older snapshots lack it; readers // must default to 0 — and MUST render that as "—", not "0", because a zero // here would claim a measurement nobody took. totalAudioBytes?: number; // THE MEDIA TIER's bytes (release 17): every file `isTierable` names — the // audio and the raw live-chat replay — whether it is a link into // `channels//media/` already or still a real file a move's preflight // would tier. The figure a media move carries and a media location holds; // `totalAudioBytes` is a fraction of it and is about what a CLEANUP could // reclaim, which is a different question. (Before release 17 this was every // byte under `data//`; the text is now `totalTextBytes`.) // // Optional, and the distinction is load-bearing: a snapshot written before // this field existed lacks it, and a reader MUST render that as "size unknown // until Refresh report", never as 0 — a zero would rank a 400 GB channel // bottom of a "free up N GB" list. ABSENT TOO when the channel's media tier // could not be read (an unmounted, stalled or moving media drive): the // snapshot is still written — its text is readable — and the bytes are // unknown, never 0. totalMediaBytes?: number; // EVERYTHING ELSE UNDER `data//` BUT `clips/` — the bytes this channel // holds on the corpus disk outside the tier and the clip cache: the text // (transcripts, cues, metadata, sidecars, thumbnails) AND whatever is not // tierable — a persisted source container not yet moved to the store, a // partial download, scratch. Named for its bulk; on a channel with a few // containers or partials it is more than the text alone. // Absent on a snapshot written before release 17: "unknown", never 0. totalTextBytes?: number; // The bytes held by `data//clips/` — the clip windows umtool and the // video page fetch a few seconds at a time. A SIBLING of the two tiers since // release 17 (before it, a subset of `totalMediaBytes`): `clips/` stays on // the corpus disk and is never tiered. // // Split out because it is the one part of a channel's bytes that is a CACHE: // nothing prunes a window, so a channel walked by many reports accumulates // them, and an operator looking at a row wants to know which of those two // things the number is. `evictClipWindows` is what acts on it. // // Optional on the same terms as `totalMediaBytes`: absent means "unknown // until Refresh report", never 0. totalClipsBytes?: number; // WHY the audio that isn't reclaimable isn't reclaimable, in the sweep's own // order (cleanAudioFromTranscribed's discover loop). A video leaves at the // FIRST gate it hits, so these are an attribution and never overlapping sets: // // totalAudioBytes === noTranscript + keepLatest + doNotClean // + awaitingDiarization + cleanupBytes.transcribedWithAudio // // pinned by channelSnapshot.test.ts, because the five numbers are only // trustworthy as a partition — a double-count would read as a bigger hold. heldAudioBytes?: HeldAudio; // The same partition by VIDEO COUNT. Only videos that actually hold audio are // counted: a dir with no audio.* file holds nothing at any gate. heldAudioCounts?: HeldAudio; // Of the RECLAIMABLE set, the bytes whose CACHED availability is already // permanently-gone / needs_auth / error — what verifyBeforeClean is likely to // refuse at sweep time. An annotation on the remainder, deliberately NOT a // fifth gate: the real gate needs a live probe, and folding a guess into the // hero figure would move the sidebar badge and /cleanup's Est. reclaim too. reclaimableAtRiskBytes?: number; }; // The four holds, in sweep order. Same shape for bytes and for counts. export type HeldAudio = { noTranscript: number; keepLatest: number; doNotClean: number; awaitingDiarization: number; // The SUBSET of awaitingDiarization that the diarize lane will never produce // on its own: over maxAudioHours (`deferred`), no diarizable input, an // untranscribable transcript the kind declines, or the kind reporting itself // disabled because its models aren't configured. Running Diarize until the end // of time does not move this number, and nothing else in the app says so. diarizationNeverClears: number; }; export function emptyHeldAudio(): HeldAudio { return { noTranscript: 0, keepLatest: 0, doNotClean: 0, awaitingDiarization: 0, diarizationNeverClears: 0, }; } // Which gate a video's audio leaves the sweep at, or "reclaimable" if it // survives all four. Pure, exported, and tested — this is the one rule that // keeps the five figures from double-counting, and it has to stay in step with // cleanAudioFromTranscribed's discover loop by inspection, so it is written in // the same order with the same predicates. export type AudioHoldGate = | "noTranscript" | "keepLatest" | "doNotClean" | "awaitingDiarization" | "reclaimable"; export function attributeAudioHold(v: { // entries.includes("transcript.json") — the sweep's gate, NOT isVideoTranscribed: // an ASR-only video has an English VTT and is still uncleanable. hasWhisper: boolean; inKeepLatestWindow: boolean; doNotClean: boolean; // settings.diarization.enabled — the guard the sweep actually reads. diarizationGuardOn: boolean; hasDiarization: boolean; }): AudioHoldGate { if (!v.hasWhisper) return "noTranscript"; if (v.inKeepLatestWindow) return "keepLatest"; if (v.doNotClean) return "doNotClean"; if (v.diarizationGuardOn && !v.hasDiarization) return "awaitingDiarization"; return "reclaimable"; } // Will the diarize lane ever produce this video's sidecar? `undefined` means the // kind isn't enabled at all (allOperations dropped it — models unconfigured), // which is the case the guard cannot see: the sweep holds the audio on // settings.diarization.enabled alone, so the hold is permanent and silent. // // Only the reachable states clear: missing / stale / partial are what the lane // picks up. `deferred` (over the duration cap), `missing-input` and // `not-applicable` (an untranscribable transcript) are all holds nothing queued // will release. export function diarizationWillNeverClear( state: OperationClassification | undefined, ): boolean { if (state === undefined) return true; return !(state === "missing" || state === "stale" || state === "partial"); } export type MissingNeverFetched = { id: string; // The URL the video was last seen at. "" only when it was never observed in a // listing (a disk-seeded entry), which cannot happen for this bucket. url: string; firstSeenAt: string; }; export function emptyExcludedFromDownload(): ExcludedFromDownload { return { membersOnly: [], deleted: [], private: [] }; } export function normalizeExcludedFromDownload( raw: Partial | undefined, ): ExcludedFromDownload { return { membersOnly: raw?.membersOnly ?? [], deleted: raw?.deleted ?? [], private: raw?.private ?? [], }; } export function excludedDownloadIdSet( snapshot: Pick | null | undefined, ): Set { const e = normalizeExcludedFromDownload(snapshot?.excludedFromDownload); return new Set([...e.membersOnly, ...e.deleted, ...e.private]); } function emptyAvailability(): AvailabilitySnapshot { const byStatus = {} as Record; for (const v of AVAILABILITY_VALUES) byStatus[v] = []; return { byStatus, unchecked: [] }; } export function normalizeAvailability( raw: Partial | undefined, ): AvailabilitySnapshot { const out = emptyAvailability(); if (!raw) return out; if (raw.byStatus) { for (const v of AVAILABILITY_VALUES) { out.byStatus[v] = raw.byStatus[v] ?? []; } } if (raw.unchecked) out.unchecked = raw.unchecked; return out; } export function normalizeMaybeMissing( raw: { ids?: string[]; checkedAt?: string } | undefined, ): { ids: string[]; checkedAt: string } { return { ids: raw?.ids ?? [], checkedAt: raw?.checkedAt ?? "" }; } // Fold one operation's per-video classifications into the record a snapshot // stores for it. // // EXTRACTED SO THE INVARIANT CAN BE TESTED. It is four lines of accounting, but // they are the four that every downstream surface trusts, and inside // generateChannelSnapshot they could only be exercised by standing up lmdb, an // archive reader and a corpus on disk — which is to say, never. The rule that // matters and that nothing else checks: // // ids.length === reachableOperationWork(entry) // // A policy leaf hands `ids` out as the work to do while the stage cards render // the count, so the two drifting apart is a progress bar that stalls one short // of complete forever. That is not hypothetical — countBackfillWork re-spelled // this same rule and had already lost `blocked` from it. export function foldBackfillEntry( videos: Iterable<{ id: string; state: OperationClassification | undefined }>, ): OperationSnapshotEntry { const counts = emptyOperationCounts(); const ids: string[] = []; // The denominator: every video this kind had an opinion about. `present` is // the classification addOperationState deliberately discards, so without this // no reader can tell "0 outstanding because it is all done" from "0 // outstanding because there was nothing here". let eligible = 0; for (const { id, state } of videos) { if (!state) continue; if (state !== "not-applicable") eligible++; addOperationState(counts, state); if (state === "missing" || state === "stale" || state === "partial") { ids.push(id); } } return { ...counts, ids: ids.sort(), eligible }; } // THE SAME ENTRY FOR A BUCKET LANE, which has no per-video classifier to fold. // // download and transcription are EXTERNAL_OPERATIONS: registered for the // dependency graph, dispatched by their own runners, and carrying no // `Operation.state()`. So their four numbers are STATED rather than counted, // and only the ones that mean something are non-zero: // // ids = bucketLaneWorkIds — the lane's default buckets, in priority // order, deduped, UNSORTED. The one function the runner also // falls back to, which is what makes writing this entry // invisible: a regenerated snapshot's list and an old // snapshot's fold are the same array. // missing = ids.length. `stale` and `partial` stay 0 — a download is // fetched or it is not, and a transcript is never part done — // which keeps the invariant the test above pins // (ids.length === reachableOperationWork) true here too. // missingInput = transcription's videos with no audio yet, work the DOWNLOAD // lane has to do first. Download's input is the listing, which // is never missing, so it is 0. // eligible = present + every work count, so presentOperationWork() gives // back totals.downloaded / totals.transcribed exactly. Videos // excluded from download and untranscribable ones are in // NEITHER half: they are this operation's not-applicable, the // same exclusion foldBackfillEntry applies. // // THESE ENTRIES ARE NOT THE /channels BANDS, and must not become them. // buildBands keeps its own fold for these two lanes because its numbers are // different ones: its transcription `reachable` is downloadedNoTranscript // ALONE — the retry bucket is not in it — and its `blocked` is derived from // noTranscript (less the untranscribable and the downloaded ones), where this // entry states the whole noTranscript count as missingInput. Pointing the // bands here would move two rendered numbers. The bands' fold is the // transcription branch of common/views/pipeline/buildBands.ts. export function foldBucketLaneEntry( lane: AutoQueueKind, source: BucketSource & { present: number }, ): OperationSnapshotEntry { const ids = bucketLaneWorkIds(lane, source); const missingInput = lane === "transcription" ? bucketIdsFrom(source, "noTranscript").length : 0; return { ...emptyOperationCounts(), missing: ids.length, missingInput, ids, eligible: source.present + ids.length + missingInput, }; } // The digest work list. ONE DEFINITION: the operation registry's entry. // // The digest layer once grew its own counter (`buckets.noDigest`) beside the // registry, and the two disagreed by an entire channel — the stage card read // "All digested" while the batch reported everything stale. They could not be // reconciled: the bucket had no cues-staleness gate (so it called `deferred` // work done), no transcript gate, and no `partial`. Measured 2026-08-26, the // corpus-wide disagreement was 11,777 videos (bucket 59,159 vs entry 47,382). // The bucket is gone; `snapshot.backfill.digest` — folded from the same // `kind.state()` the digest runner dispatches from — is the definition. // // THE FALLBACK THE MIGRATION NEEDED IS RETIRED. On 2026-08-26 all 68 // `transcripts/channels/*/snapshot.json` carried `backfill.digest` (and // `eligible`), so the bucket branch was reading nothing. A snapshot with no // entry is now a channel whose FIRST snapshot has not been written yet, not an // old generation of one — and it reports unknown coverage, never zero. // // NO CALLERS IN `common/` any more — the sweep planner and the band builder read // `snapshot.backfill[id]` generically, digest included. It stays in this module // anyway: it is the editor's one adapter from an `OperationSnapshotEntry` to the // digest surfaces, and moving it would be churn without a deletion. export type DigestWork = { // Videos needing digest work that the lane can do right now. ids: string[]; reachable: number; blocked: number; deferred: number; partial: number; // Videos with a current digest, and how many were considered. Null when the // snapshot cannot say — never 0, which would render as "none digested". present: number | null; eligible: number | null; }; export function digestWorkOf( snapshot: Pick | null | undefined, ): DigestWork { const entry = snapshot?.backfill?.[DIGEST_OPERATION_ID]; if (entry) { return { ids: entry.ids ?? [], reachable: reachableOperationWork(entry), blocked: entry.blocked ?? 0, deferred: entry.deferred ?? 0, partial: entry.partial ?? 0, present: presentOperationWork(entry), eligible: entry.eligible ?? null, }; } // No entry: no work known, and coverage UNKNOWN rather than zero. Null is the // whole point — 0 would render as "none digested" on a channel nobody has // snapshotted yet. return { ids: [], reachable: 0, blocked: 0, deferred: 0, partial: 0, present: null, eligible: null, }; } // The snapshot READER lives in ./channels — reading a report shouldn't require // loading the machinery that generates one (this module pulls in lmdb, the // archive reader and the digest layer). Re-exported here so the name stays // where callers expect to find it. export { SNAPSHOT_FILENAME, snapshotPath, readChannelSnapshot }; const SNAPSHOT_VIDEO_CONCURRENCY = 16; // THE WALK YIELDS TO THE EVENT LOOP every SNAPSHOT_YIELD_EVERY videos. // // It runs in the editor's own process, on the main thread, and a video's unit // parses its cues.json and (in the reconcile pass and for a non-YouTube // archive) its metadata.info.json — ~0.6 MB each on a long VOD, a few ms of // JSON.parse apiece. Sixteen units in flight keep every turn of the loop busy // with that work; on 2026-10-01 two regenerations of 2,000–3,000-video // channels ran for over an hour while `/` and `/jobs` did not answer. So the // ids are walked in chunks, each fanned out under the same limit, and between // chunks the walk waits one `setImmediate`: the loop gets a turn with no // snapshot work queued in it, and every request whose I/O completed meanwhile // runs before the next chunk starts. // // Two full waves of the limit (32 at 16 wide): a chunk is ~0.1–0.2 s of // parsing on such a channel, and a multiple of the width keeps every wave full // — 25 left the second wave nine wide, a fifth of the walk's throughput. // Measured in the slice D0 record (plans/release-17.md). const SNAPSHOT_YIELD_EVERY = 2 * SNAPSHOT_VIDEO_CONCURRENCY; // `fn` over `items` in chunks of `size`, in order, one setImmediate between // chunks. The concurrency inside a chunk is whatever `fn` imposes. Exported for // snapshotYield.test.ts, which pins the yield. export async function mapInYieldingChunks( items: readonly T[], size: number, fn: (item: T) => Promise, ): Promise { const out: R[] = []; for (let i = 0; i < items.length; i += size) { if (i > 0) await new Promise((resolve) => setImmediate(resolve)); out.push(...(await Promise.all(items.slice(i, i + size).map(fn)))); } return out; } // ONE LEVEL of a video dir's `clips/` — the files in it, and nothing deeper. // Deliberately not recursive: `clipWindow-server.ts` writes `-.` // and `-.json` flat into it and nothing else does, so a recursion // would be machinery for a case that cannot happen. A missing dir is 0. async function dirFileBytes(dir: string): Promise { let total = 0; let entries: string[]; try { entries = await readdir(dir); } catch { return 0; } for (const name of entries) { try { const st = await stat(path.join(dir, name)); if (st.isFile()) total += st.size; } catch { // ignore — vanished mid-walk } } return total; } async function readPlaylistUrls(file: string): Promise { try { const raw = await readFile(file, "utf8"); return raw.split("\n").map((s) => s.trim()).filter(Boolean); } catch { return []; } } // WAS A PRIVATE COPY OF isVideoDownloaded, byte for byte, and this module // already imported that one (it is what increments `downloaded` in the same // walk). Two names for one predicate is how the settled short-circuit and the // totals end up disagreeing about what an artifact is, so the copy is gone and // the name stays as an alias: "does this dir hold anything we fetched?" reads // better at the five call sites than "is it downloaded?", and it is now // guaranteed to be the same question. const videoHasAnyArtifact = isVideoDownloaded; export async function generateChannelSnapshot( paths: Paths, slug: string, ): Promise { const channelDir = path.join(paths.channelsDir, slug); const dataDir = path.join(channelDir, "data"); const archivePath = path.join(channelDir, "archive"); const playlistPath = path.join(channelDir, "playlist"); // GUARD 3 OF FOUR (see plans/relocate-channel-media.md). The readdir below // swallows ENOENT as "this channel has no videos", so a relocated channel // whose drive is not mounted would generate a snapshot saying every video is // undownloaded and every transcript is missing — and everything downstream // (/cleanup, the channel bands, all four lanes' work lists) reads that // snapshot as the truth. Throw instead: the scheduler keeps the last good // snapshot.json on a failed refresh, which is exactly the right outcome. // The config is read HERE rather than in the fan-out below and passed in, so // the guard does not open config.json a second time for the one field it // needs. One extra await in front of a function that then walks the whole // channel. // // THE TEXT GUARD (release 17): the snapshot is a reading of the TEXT tier — // names, sidecars, metadata — so a moving, stalled or unmounted MEDIA drive // does not hold it (its media bytes are then unknown, below). Refused: a // `legacy` channel, whose text is on the far drive. const config = await readChannelConfig(paths, slug); const mediaLocation = await assertChannelTextReadable(paths, slug, config); // A CHANNEL ON ANOTHER DRIVE IS WALKED THROUGH THE WATCHDOG // (lib/storageHealth.ts `onDrive`): the data/ listing, the keep-latest keys and // each video directory's unit below. At most four of them wait on that drive // at once — this walk runs in the editor's own process after every download // or sync of the channel, sixteen wide, which is exactly while a long write is // stressing the drive — and one that does not answer within the budget // (`storage.health.budgetMs`, 3 s by default) throws, so the // scheduler keeps the last good snapshot.json, as on any failed refresh. // The reconcile pass just below is sequential (one read at a time) and is // not raced. // // SINCE RELEASE 17 THE TEXT IS NEVER ON ANOTHER DRIVE: the text guard above // refuses the one layout where it was (`legacy`), so `through` reads // directly. What may be on another drive is the media tier, and its stats // are the only calls that go through the watchdog (per video, below). const through = (read: () => Promise): Promise => read(); const mediaDrive = config?.mediaDir?.trim() || undefined; // Whether the media tier may be read at all this run: its links are statted // only when the media is `ok` or `in-place`. Otherwise (unmounted, stalled, // moving, inconsistent) no call is made to it and the bytes are unknown. let mediaBytesKnown = mediaLocation.status === "ok" || mediaLocation.status === "in-place"; // Heal any video dir that drifted from the canonical id layout before we read // data/* (best-effort; never fail snapshot generation on a reconcile error). try { await reconcileVideoDirs({ channelDir }); } catch { /* ignore — the migration CLI can repair stragglers */ } const [ dirEntries, urls, archive, failedListed, maybeMissingRecord, roster, ] = await Promise.all([ through(() => readdir(dataDir, { withFileTypes: true })).catch((err) => { // A drive that did not answer is not an empty channel: rethrown. if (isDriveNotAnswering(err)) throw err; return [] as Dirent[]; }), readPlaylistUrls(playlistPath), readArchive(archivePath), loadFailedTranscriptions(paths, slug), loadMaybeMissing(paths, slug), loadRoster(paths, slug), ]); const targetAudioFile = config?.audioFormat ? `audio.${config.audioFormat}` : null; // The keep-latest window: newest N videos protected from cleanup. Computed // once and folded into the cleanup-exclusion set below. const keptIds = await computeKeptVideoIds({ paths, channelSlug: slug, keepLatest: config?.keepLatest ?? 0, through, }); const videoDirNames = dirEntries .filter((d) => d.isDirectory()) .map((d) => d.name); // The archive stores yt-dlp's native extractor ids, which equal the dir name // (canonical id) only on YouTube. When the archive has any non-youtube // extractor, read each video's native id from metadata so missingFromArchive // compares like-for-like. const wantNativeIndex = [...archive.byExtractor.keys()].some( (e) => !e.toLowerCase().startsWith("youtube"), ); // Which operations are live, and what identity each would produce right now. // Resolved ONCE per channel — a settings read and some string work — so the // per-video probe below is a comparison rather than a derivation. // // This is the ONLY digest identity the snapshot resolves. It used to resolve a // second one (`resolveDigestTarget`, for the `noDigest` bucket) and the two // definitions disagreed; the property that mattered — the stage's count and // the batch runner's target cannot differ — is now preserved by construction, // because the entry is folded from `kind.state()` against this // `kind.resolveTarget(...)`, the same identity the runner dispatches from. // // allOperations, NOT backfillLaneOperations. The snapshot's job is to carry a // work list for every operation in the catalog, not for one lane: nothing can // count, select or schedule work the snapshot does not record, which is why // registering the digest kind in Phase C changed no number anywhere — its // state() was called by nothing. The lane filter belongs on the READ side, // where a surface says which lane it means (backfillLaneEntriesOf). // // COST, measured rather than assumed, because this runs per video over ~79,000 // of them: the digest classification is 0.34 ms/video on a 125-video channel // and 0.38 ms/video on a 773-video one — 0.7x to 1.2x the readVideoFiles call // directly above, which this path already pays. About 30 s across the whole // corpus, spread over 67 per-channel regenerations. The re-listing inside // isCuesJsonFresh is the obvious thing to fold into `files.entries` and it is // deliberately NOT done: at this cost it would be an optimization with no // measurement behind it. const backfillSettings = getSettings(); const liveOperations = allOperations(backfillSettings); const backfillTargets: Record = {}; for (const kind of liveOperations) { backfillTargets[kind.id] = await kind.resolveTarget({ settings: backfillSettings, paths, channelSlug: slug, }); } const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY); const perVideo = await mapInYieldingChunks( videoDirNames, SNAPSHOT_YIELD_EVERY, (id) => // One video directory's reads are one unit through the watchdog. limit(() => through(async () => { const dir = path.join(dataDir, id); const files = await readVideoFiles(dir, { checkUntranscribable: true }); // EVERY FILE IN THE DIR, `lstat`ED ONCE, feeding the byte figures. // // `audioSizes` is what it always was: the real audio files, keyed by // name, for the cleanup reclaim estimate. `mediaBytes` is the media // tier's share (what `isTierable` names) and `textBytes` everything // else; `clips/` (CLIPS_DIR_NAME, the fetched windows) is recursed ONE // LEVEL into `clipsBytes`, apart from both — it is never tiered. // // BY FILE KIND (release 17): a real file's size is its `lstat`, on the // corpus disk, no watchdog. A LINK is a tiered media file whose bytes // are on the media tier — possibly another drive — so the links are // statted together, as ONE call through the watchdog per video // (`onDrive(mediaDir)`), and only while the media is readable. A drive // that does not answer makes the channel's media bytes unknown; the // snapshot is still written. const audioSet = new Set(files.audioFiles); const audioSizes: Record = {}; let mediaBytes = 0; let textBytes = 0; let clipsBytes = 0; const linked: string[] = []; for (const name of files.entries) { try { const st = await lstat(path.join(dir, name)); if (st.isSymbolicLink()) { linked.push(name); continue; } if (!st.isFile()) { if (st.isDirectory() && name === CLIPS_DIR_NAME) { clipsBytes += await dirFileBytes(path.join(dir, name)); } continue; } if (isTierable(name)) mediaBytes += st.size; else textBytes += st.size; if (audioSet.has(name)) audioSizes[name] = st.size; } catch { // ignore — file vanished or is unreadable } } if (linked.length > 0 && mediaBytesKnown) { const statLinks = async () => { const sizes: Array<[string, number]> = []; for (const name of linked) { try { const st = await stat(path.join(dir, name)); if (st.isFile()) sizes.push([name, st.size]); } catch { // a dangling link: its bytes are not on the media tier } } return sizes; }; try { const sizes = await (mediaDrive ? onDrive(mediaDrive, statLinks) : statLinks()); for (const [name, size] of sizes) { mediaBytes += size; if (audioSet.has(name)) audioSizes[name] = size; } } catch (err) { if (!isDriveNotAnswering(err)) throw err; mediaBytesKnown = false; } } let nativeId: string | null = null; if (wantNativeIndex && files.hasMeta) { try { const raw = await readFile( path.join(dir, "metadata.info.json"), "utf8", ); const parsed = JSON.parse(raw); if (typeof parsed?.id === "string") { nativeId = parsed.id; } } catch { // ignore } } const availability = await loadAvailability(dir); const effectiveAvailability = await resolveEffectiveAvailability(dir); const doNotClean = await isDoNotClean(dir); const excludedFromTruncatedCheck = await isExcludedFromTruncatedCheck(dir); const outcome = await loadDownloadOutcome(dir); // Only transcribed (non-untranscribable) dirs can have a truncated // transcript; skip the cues.json read for everything else. const coverage = isVideoTranscribed(files) && !files.isUntranscribable ? await readTranscriptCoverage(dir) : null; // Where the English VTT came from (YouTube ASR vs a human-authored // track). Only videos that HAVE such a VTT pay the 4 KB head read — // both the auto-subs work lane (no whisper yet) and the superseded // backup inventory (whisper already won) need it. Same conditional // per-video sidecar read pattern as the cues.json coverage read above. // Over every English VTT (resolveCaptionsProvenance): the rule reads // en-orig even beside a human `en`, which is not ASR-only. const vttProvenance = files.ytVttFile ? await resolveCaptionsProvenance(dir, files.entries) : null; // Only transcribed videos can carry a digest, so everything else skips // the sidecar read entirely — the same conditional per-video // sidecar-read pattern as the two reads above. const digest = isVideoTranscribed(files) && !files.isUntranscribable ? await loadDigest(dir) : null; // Work state, per registered operation. // // This USED to cost nothing in the default configuration, because every // backfill feature ships off and `liveOperations` was therefore empty. // That is no longer true and the change is deliberate: the digest // operation is enabled whenever an app is configured (its pause lives at // dispatch, not here, so a paused lane still reports its outstanding // work), so this loop now always runs at least once per video. See the // measured per-video cost where liveOperations is resolved above — it is // roughly the readVideoFiles call this classification reuses rather than // repeats. const backfill: Record = {}; for (const kind of liveOperations) { backfill[kind.id] = await kind.state({ videoDir: dir, videoId: id, files, target: backfillTargets[kind.id], settings: backfillSettings, }); } return { id, files, backfill, audioSizes, mediaBytes, textBytes, clipsBytes, nativeId, availability, effectiveAvailability, doNotClean, excludedFromTruncatedCheck, outcome, coverage, vttProvenance, digest, }; })), ); const filesById = new Map(); // Native (yt-dlp extractor) ids present on disk, for the archive comparison. // Dir names are canonical ids and double as native ids on YouTube. const nativeIdsOnDisk = new Set(); for (const v of perVideo) { filesById.set(v.id, v.files); nativeIdsOnDisk.add(v.id); if (v.nativeId) nativeIdsOnDisk.add(v.nativeId); } // Computed up-front so the buckets below can suppress IDs that can't be // acted on (members_only / deleted / private). Surfacing them in // "Missing metadata" or "Failed transcriptions" creates noise the user // cannot resolve. const excludedById = new Map(); const effectiveById = new Map(); for (const v of perVideo) { if (v.effectiveAvailability) { effectiveById.set(v.id, v.effectiveAvailability); } if ( v.effectiveAvailability && (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray).includes( v.effectiveAvailability, ) ) { excludedById.set(v.id, v.effectiveAvailability); } } // Videos the user has opted out of cleanup for (archived media). Filtered out // of the cleanup buckets only — NOT of the download/transcribe buckets. const doNotCleanIds = new Set(); for (const v of perVideo) { if (v.doNotClean) doNotCleanIds.add(v.id); } // Everything shielded from the Clean-audio sweep: explicit do-not-clean markers // plus the rolling keep-latest window. The cleanup buckets/reclaim estimates // below exclude this whole set so they match what cleanAudioFromTranscribed // will actually remove. const protectedFromCleanup = new Set([...doNotCleanIds, ...keptIds]); // The diarization capture lane holds audio back until its sidecar exists. Read // once per snapshot rather than per video — this and the sweep's own guard key // off the same setting, and they have to agree or the reclaim estimate lies. const diarizationEnabled = getSettings().diarization.enabled; // Videos the user has opted out of the truncated/incomplete-transcript check // (e.g. legitimately have no speech for the back half). Suppressed from the // incompleteTranscript + shortAudio detection buckets below. const excludedTruncatedIds = new Set(); for (const v of perVideo) { if (v.excludedFromTruncatedCheck) excludedTruncatedIds.add(v.id); } // SETTLED, derived — the ids this channel's CURRENT download filter rejects // out of everything the metadata scan has read. One file read per snapshot, // and nothing about it is stored per video: editing a pattern re-decides the // whole channel on the next report, with no rescan and nothing to migrate. const metadataScanStore = await loadMetadataScan(paths, slug); const settledIds = settledIdsFrom(metadataScanStore, config); // A STRICT SUBSET of the settled set: the rejections the operator asked for // the live chat of (downloadFilter.rejectedLivestreams === "chat-only"). // Settled means the MEDIA is not wanted, which is true of these too — what // this adds is that the video is still a corpus member, as a chat track. const chatOnlyIds = chatOnlyIdsFrom(metadataScanStore, config); // A scan that already read an id as members-only or private is the // same answer a failed download records in availability.json — without this // the download lane re-learned it, one ~30 s cookie-authed attempt per id // (timcast-irl: 28 members-only ids the scan had classified an hour before). // Derived like settledIds: a later scan that reads the video deletes the // error, and the exclusion lifts with nothing to undo. A conclusive on-disk // availability is newer evidence and wins; `error` (a cancelled or throttled // attempt) is not conclusive. Not the scan's `deleted`: that is // inferred from a bare "Video unavailable", which is also how a soft block // reads, so it is not evidence enough to stop asking. for (const [id, err] of Object.entries(metadataScanStore.errors)) { const onDisk = effectiveById.get(id); if (onDisk && onDisk !== "error") continue; const cls = err.class as Availability; if (cls !== "members_only" && cls !== "private") continue; effectiveById.set(id, cls); excludedById.set(id, cls); } const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; const wrongFormatAudio: string[] = []; const multipleAudioFormats: string[] = []; const transcribedWithAudio: string[] = []; const untranscribable: string[] = []; const noMetadata: string[] = []; const partialDownloads: string[] = []; const corruptSource: string[] = []; const corruptFullSource: string[] = []; const nonStandardVtt: string[] = []; const skippedByFilter: string[] = []; // Settled ids that DO have a directory (a download-time rejection prefetched // their metadata before deciding). Tracked separately from the full settled // set only so totals.videos can subtract exactly the dirs it counted. const settledOnDisk: string[] = []; // Chat-only videos whose chat is on disk — corpus members, counted in // totals.videos, and out of every bucket that means "something is missing". const chatOnly: string[] = []; // Chat-only videos whose chat is NOT on disk yet: the download lane's work, // and the reason this is a bucket rather than a derived count. const chatOnlyPending: string[] = []; const incompleteTranscript: string[] = []; const shortAudio: string[] = []; const autoSubsOnly: string[] = []; const downloadedAutoSubsOnly: string[] = []; const supersededAutoSubs: string[] = []; const digestWarnings: string[] = []; const digestEngines: Record = {}; let transcribedWithAudioBytes = 0; let multipleAudioFormatsBytes = 0; let foreignAudioBytes = 0; let transcribed = 0; let downloaded = 0; // WHY the audio that isn't reclaimable isn't. The sweep reports this only as a // line in a job log after the fact ("Skipped 412 (388 protected, …)"), with no // bytes attached to it, so nothing could rank the holds or say which of them a // queued job will ever clear. Free to compute here: every input is already in // hand on the perVideo entry, so this adds ZERO I/O to a pass that runs over // ~79,000 videos. let totalAudioBytes = 0; // The media tier's bytes, the text tier's and the clip cache's — three // siblings since release 17 (see the field comments). let totalMediaBytes = 0; let totalTextBytes = 0; let totalClipsBytes = 0; const heldAudioBytes = emptyHeldAudio(); const heldAudioCounts = emptyHeldAudio(); let reclaimableAtRiskBytes = 0; // Is the lane that would release an awaiting-diarization video even running? // allOperations drops a kind whose enabled() is false, so an ABSENT entry // means "no diarize job will ever be produced" — while the sweep's guard holds // the audio on settings.diarization.enabled alone. That gap is a permanent, // silent hold, and this is the only place in the app that can see it. const diarizationKindEnabled = liveOperations.some( (k) => k.id === DIARIZATION_OPERATION_ID, ); for (const { id, files, audioSizes, mediaBytes, textBytes, clipsBytes, backfill, effectiveAvailability, outcome, coverage, vttProvenance, digest, } of perVideo) { if (isVideoTranscribed(files)) transcribed++; if (isVideoDownloaded(files)) downloaded++; totalMediaBytes += mediaBytes; totalTextBytes += textBytes; totalClipsBytes += clipsBytes; // --- Hold attribution --------------------------------------------------- // FIRST, before the short-circuits below: they `continue` past videos that // still have audio on disk, and audio dropped here would leave the partition // (held + reclaimable === total) silently short. const audioBytes = files.audioFiles.reduce( (sum, name) => sum + (audioSizes[name] ?? 0), 0, ); totalAudioBytes += audioBytes; if (files.audioFiles.length > 0) { const gate = attributeAudioHold({ // The sweep's own gate — entries.includes("transcript.json"). NOT // isVideoTranscribed, which counts any English VTT: an ASR-only video is // absent from downloadedNoTranscript and still uncleanable. // // A corrupt-full-source container is forced to this gate to stay in step // with the transcribedWithAudio bucket, which short-circuits it below: // it is an artifact, not usable audio, and the estimate must not offer // the kept file up for deletion. hasWhisper: files.hasWhisper && outcome?.status !== "corrupt-full-source", inKeepLatestWindow: keptIds.has(id), doNotClean: doNotCleanIds.has(id), diarizationGuardOn: diarizationEnabled, hasDiarization: files.hasDiarization, }); if (gate === "reclaimable") { // What verifyBeforeClean is likely to refuse at sweep time, from the // CACHED verdict only — no probe, and deliberately not a fifth gate. if ( effectiveAvailability && (isPermanentlyGone(effectiveAvailability) || effectiveAvailability === "needs_auth" || effectiveAvailability === "error") ) { reclaimableAtRiskBytes += audioBytes; } } else { heldAudioBytes[gate] += audioBytes; heldAudioCounts[gate] += 1; if ( gate === "awaitingDiarization" && diarizationWillNeverClear( diarizationKindEnabled ? backfill[DIARIZATION_OPERATION_ID] : undefined, ) ) { heldAudioBytes.diarizationNeverClears += audioBytes; heldAudioCounts.diarizationNeverClears += 1; } } } // A download that completed but stayed malformed after one re-download. The // raw container (e.g. audio.mp4) is kept on disk for inspection. It IS an // artifact (so it won't be re-queued for download), but it is NOT usable // audio — short-circuit so it doesn't land in wrongFormatAudio / // downloadedNoTranscript (which would transcribe corrupt audio, or offer // the kept file to the wrong-format sweep) or the wrong-format cleanup // estimate (which would delete it). if (outcome?.status === "corrupt-full-source") { corruptFullSource.push(id); continue; } // A download whose audio was far shorter than the video (truncated source). // Like corrupt-full-source: the stub is KEPT on disk but is NOT usable audio, // so short-circuit it out of downloadedNoTranscript (no auto-transcribe) and // the wrong-format cleanup (don't delete the kept file). If the user later // transcribes it anyway, it falls through to the incompleteTranscript path. if (outcome?.status === "failed-short-audio" && !isVideoTranscribed(files)) { if (!excludedTruncatedIds.has(id)) shortAudio.push(id); continue; } // SETTLED BY THE DOWNLOAD FILTER — the operator's own "not this one". // Short-circuited here, beside the other two terminal outcomes, and for the // same reason: everything below classifies a video by what is MISSING from // its directory, and a settled video is missing everything. Most settled ids // have no directory at all (the metadata scan never makes one) and are added // to the bucket below; this branch catches the ones a download-time // rejection left a prefetched metadata.info.json behind for, which would // otherwise land in noTranscript AND noMetadata AND the retryable // skippedByFilter at once. // // ONLY WHEN THERE IS NOTHING ON DISK. A video that was downloaded before the // filter was written (or before it was edited to reject it) is DOWNLOADED — // that is a fact, not a preference — and short-circuiting it here would drop // it from transcribedWithAudio and the cleanup estimates while it still // counted in totals.transcribed, i.e. a channel reporting more transcripts // than videos. Settlement only ever decides what NOT to fetch. // SETTLED, and the two answers it can have. // // THE CHAT ON DISK DECIDES, NOT THE MODE. A dir holding // transcript.live_chat.json is already published by buildIndex as a chat // track with no captions — that is a fact about the corpus, and flipping // `rejectedLivestreams` back to "skip" does not un-publish it. Asking // `chatOnlyIds` here instead would make the same directory count as a // settled STUB the moment the mode changed, and settledOnDisk is subtracted // from totals.videos: the site would still serve the video while the report // stopped counting it. So the file is the test, and the Diagnostics copy — // "what is already on disk stays" — is true. // // Short-circuited for the same reason the settled branch is: everything // under this line classifies a video by what is MISSING, and nothing is // missing from a chat-only video. if (settledIds.has(id) && !videoHasAnyArtifact(files)) { if (files.entries.includes(LIVE_CHAT_FILENAME)) { chatOnly.push(id); } else { settledOnDisk.push(id); } continue; } if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); // A transcribed video whose cues stop far short of its duration — the audio // download truncated silently. Threshold lives in transcriptCoverage. if ( coverage && !excludedTruncatedIds.has(id) && isIncompleteTranscript(coverage.cov, { isLivestream: coverage.isLivestream, }) ) { incompleteTranscript.push(id); } // A video the filter declined (e.g. live/upcoming) that hasn't since been // downloaded. Once it lands an artifact it drops out of this bucket. if ( outcome?.status === "skipped-filtered" && !videoHasAnyArtifact(files) ) { skippedByFilter.push(id); } // The audio-check pipeline gave up on a malformed source (left no real // audio). Surface it as a download/source problem rather than letting the // transcribe pass mislabel it as a failed transcription. if ( outcome?.status === "failed-corrupt-source" && !isVideoTranscribed(files) ) { corruptSource.push(id); } if ( targetAudioFile && files.audioFiles.length > 0 && !files.audioFiles.includes(targetAudioFile) ) { wrongFormatAudio.push(id); } // Reclaim estimate for the wrong-format sweep: every non-target audio file // in a cleanable (not do-not-clean) dir. Covers orphans (no target) and the // extras counted in multipleAudioFormats — the sweep removes them all. if (targetAudioFile && !protectedFromCleanup.has(id)) { for (const name of files.audioFiles) { if (name !== targetAudioFile) { foreignAudioBytes += audioSizes[name] ?? 0; } } } if ( targetAudioFile && files.audioFiles.includes(targetAudioFile) && files.audioFiles.length > 1 && !protectedFromCleanup.has(id) ) { multipleAudioFormats.push(id); for (const name of files.audioFiles) { if (name !== targetAudioFile) { multipleAudioFormatsBytes += audioSizes[name] ?? 0; } } } if ( files.hasWhisper && files.audioFiles.length > 0 && !protectedFromCleanup.has(id) && // Held back by the diarization guard in cleanAudioFromTranscribed. Kept // out of the bucket AND its byte total so "Est. reclaim" never promises // space the sweep is going to refuse to take. !(diarizationEnabled && !files.hasDiarization) ) { transcribedWithAudio.push(id); for (const name of files.audioFiles) { transcribedWithAudioBytes += audioSizes[name] ?? 0; } } if ( files.audioFiles.length === 0 && files.partAudioFiles.length > 0 && !excludedById.has(id) ) { partialDownloads.push(id); } // A transcript that exists only under a non-standard VTT name — either a // regional/auto English track (transcript.en-US.vtt) now picked up by the // fallback, or a foreign-only transcript..vtt that isn't recognized as // English. Whisper transcripts, the canonical transcript.en.vtt and the // original-audio transcript.en-orig.vtt (the rule's first pick) are fine. if ( !files.hasWhisper && files.hasNonCanonicalVtt && files.ytVttFile !== VTT_FILENAME && files.ytVttFile !== ORIG_VTT_FILENAME ) { nonStandardVtt.push(id); } // --- Replace-auto-captions lane ----------------------------------------- // A transcript that is ONLY YouTube's speech recognition. isVideoTranscribed // counts any English VTT, so without these buckets such a video is invisible // to every transcribe path forever. Conservative by construction: a VTT whose // provenance we can't prove ("unknown") is never a candidate, so a human // caption track is never scheduled for replacement. The corrupt-full-source // and failed-short-audio `continue`s above already excluded terminal videos. const asrVtt = files.hasYtVtt && vttProvenance === "asr"; if (asrVtt && files.hasWhisper) { // Our transcript won; the old ASR VTT lingers as a backup. do-not-clean // dirs are excluded so the count matches what the manual purge removes. if (!doNotCleanIds.has(id)) supersededAutoSubs.push(id); } else if ( asrVtt && !files.isUntranscribable && !excludedById.has(id) ) { if (files.audioFiles.length > 0) downloadedAutoSubsOnly.push(id); else autoSubsOnly.push(id); } // --- AI digest coverage ------------------------------------------------- // Counted before the untranscribable/no-transcript `continue`s below so the // accounting is unambiguous: a video is either a digest candidate or not a // transcript at all. if (isVideoTranscribed(files) && !files.isUntranscribable) { const chapters = digest?.sections.chapters; const tags = digest?.sections.tags; const engine = chapters?.provenance.appId ?? tags?.provenance.appId; const hasItems = (chapters?.items.length ?? 0) > 0 || (tags?.items.length ?? 0) > 0; // Coverage-by-engine answers "what produced what is on disk?" and stays // identity-BLIND — a section generated by an older prompt version was // still generated by that engine, and hiding it would make the coverage // split lie during exactly the config change it exists to survey. if (engine && hasItems) { digestEngines[engine] = (digestEngines[engine] ?? 0) + 1; } // Reviewable regardless of freshness: a video that failed outright is // also reachable work in `backfill.digest` (it has no section), and a // video whose section landed with warnings is fresh and would otherwise // never be surfaced again. if ( (digest?.warnings?.length ?? 0) > 0 || (digest?.failures?.length ?? 0) > 0 ) { digestWarnings.push(id); } } if (files.isUntranscribable) { untranscribable.push(id); continue; } if (!isVideoTranscribed(files)) { if (files.audioFiles.length > 0) downloadedNoTranscript.push(id); else noTranscript.push(id); } } const availability = emptyAvailability(); for (const v of perVideo) { if (!v.files.hasMeta) continue; if (v.availability) { availability.byStatus[v.availability.availability].push(v.id); } else { availability.unchecked.push(v.id); } } for (const v of AVAILABILITY_VALUES) availability.byStatus[v].sort(); availability.unchecked.sort(); const missingFromArchive: string[] = []; for (const id of archive.ids) { if (!nativeIdsOnDisk.has(id)) missingFromArchive.push(id); } const excludedFromDownload = emptyExcludedFromDownload(); for (const [id, cls] of excludedById) { if (cls === "members_only") excludedFromDownload.membersOnly.push(id); else if (cls === "deleted") excludedFromDownload.deleted.push(id); else if (cls === "private") excludedFromDownload.private.push(id); } excludedFromDownload.membersOnly.sort(); excludedFromDownload.deleted.sort(); excludedFromDownload.private.sort(); // The channel's resolved cookie mode. In "defer" mode, needs_auth videos are // ALSO dropped from undownloadedIds (which the auto-download runner // consumes), so they aren't re-attempted every run — they wait in the // needsCookies bucket for the manual cookie run instead. const cookiePolicy = resolveCookiePolicy(getSettings(), config ?? undefined); // Every set the channel is described by comes from one place, so "missing" // can't mean two things in two files. `listed` is the stored playlist's // canonical ids; `missingNeverFetched` is the category that used to be // inexpressible — in the roster, never downloaded, and no longer listed. const sets = deriveChannelSets({ roster, listedIds: new Set( urls.map((u) => extractVideoId(u)).filter((id): id is string => Boolean(id)), ), onDiskIds: new Set(videoDirNames), // See deriveChannelSets: both sets it derives are defined by the ABSENCE of // a directory, and a settled video has none since a rejection stopped // leaving its prefetch dir behind. Without this, `missingNeverFetched` — // the loudest alarm this report raises — would fire for videos the operator // asked us not to fetch. settledIds, }); const listedIdSet = new Set(sets.listed); const undownloadedIds: string[] = []; const needsCookies: string[] = []; // The metadata scan's backlog, counted in the same walk: a listed video with // nothing on disk that the scan has not read and has not recently failed on. // This is what /operations/metadata-scan offers a Run for, so it must mean // exactly what metadataScanTargets() will fetch. const scanNow = Date.now(); let metadataScanUnscanned = 0; // Walked in playlist order, not sorted: this is the auto-download runner's // work queue, and the listing is newest-first. for (const url of urls) { // Post-reconcile a video's dir is its canonical id, so the URL's canonical // id is the dir name directly. const dirId = extractVideoId(url); if (!dirId || !listedIdSet.has(dirId)) continue; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; // Never fetched and never read: work for the metadata scan. Counted before // the settlement check, because a settled video HAS been read — it is the // scan's output, not its input. // The SAME "already fetched" predicate metadataScanTargets uses, so the // number this advertises is the number Run will actually fetch. if ( !(f ? isVideoFetched(f) : false) && metadataScanWanted(metadataScanStore, dirId, scanNow) ) { metadataScanUnscanned++; } // CHAT ONLY: the media is settled (so this id never reaches // undownloadedIds) but the CHAT may still be outstanding, and that is real // download-lane work with no other home — the id has no artifact, so no // artifact-derived bucket can carry it. Once the chat lands it drops out // here and appears in `chatOnly` instead. if (chatOnlyIds.has(dirId)) { // Already a member (the chat is on disk) => nothing outstanding. The scan // store's `noLiveChat` flag is the other way out of this list: a stream // we asked and got nothing from is not in `chatOnlyIds` at all, so it // settles as an ordinary rejection instead of sitting here forever. if (!f?.entries.includes(LIVE_CHAT_FILENAME)) chatOnlyPending.push(dirId); continue; } // A settled video has no artifact and never will while the filter stands. // This is the line that makes the settlement STICK: undownloadedIds is the // auto-download runner's work queue, and it is derived from artifacts, so an // archive line alone would not have kept the id out of it. if (settledIds.has(dirId)) continue; const effective = effectiveById.get(dirId); if (effective && AUTH_RETRY_CLASSES.has(effective)) { needsCookies.push(dirId); } if (excludedById.has(dirId)) continue; if (cookiePolicy.mode === "defer" && effective === "needs_auth") continue; undownloadedIds.push(dirId); } // Re-surface the maybe-missing sidecar, intersecting with on-disk dirs so an // id whose dir was deleted drops out. const dirNameSet = new Set(videoDirNames); const maybeMissing = maybeMissingRecord ? { ids: maybeMissingRecord.ids.filter((id) => dirNameSet.has(id)), checkedAt: maybeMissingRecord.checkedAt, } : undefined; // Fold the per-video classifications into per-operation counts. Only the // reachable half carries ids — see OperationSnapshotEntry for why missing-input // does not. // // `ids` MUST stay exactly the set reachableOperationWork counts, because a // policy leaf consumes this list as the work to hand out while the stage cards // render the number: a list and a total that disagree is a progress bar that // stalls one short of complete forever. `partial` is reachable, so it is in // both. channelSnapshot.test.ts pins `ids.length === reachableOperationWork()` // for exactly this reason: the two derivations have drifted apart once already // (countBackfillWork silently dropped `blocked`), and nothing about the shape // here would have caught it. const backfillCounts: Record = {}; for (const kind of liveOperations) { backfillCounts[kind.id] = foldBackfillEntry( perVideo.map((v) => ({ id: v.id, state: v.backfill[kind.id] })), ); } const corruptSourceSet = new Set(corruptSource); const corruptFullSourceSet = new Set(corruptFullSource); const chatOnlySet = new Set(chatOnly); const snapshotBuckets: ChannelSnapshot["buckets"] = { noTranscript: noTranscript.sort(), downloadedNoTranscript: downloadedNoTranscript.sort(), wrongFormatAudio: wrongFormatAudio.sort(), multipleAudioFormats: multipleAudioFormats.sort(), transcribedWithAudio: transcribedWithAudio.sort(), untranscribable: untranscribable.sort(), noMetadata: noMetadata.sort(), failedListed: failedListed.filter( (id) => !excludedById.has(id) && !corruptSourceSet.has(id) && !corruptFullSourceSet.has(id), ), missingFromArchive: missingFromArchive.sort(), duplicateDirs: [], partialDownloads: partialDownloads.sort(), corruptSource: corruptSource.sort(), corruptFullSource: corruptFullSource.sort(), nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), // The settled set minus anything already on disk — same rule as the // short-circuit above, so the bucket and the classification cannot disagree // — and minus the chat-only subset, which has its own two buckets. A // chat-only video IS settled (its media is not wanted) but reporting it as // "skipped by the title filter" would say we decided not to have it, when // we decided to have its chat. skippedByTitleFilter: [...settledIds] .filter((id) => { // Minus the two chat buckets, whichever way an id got into them — a // chat-only video IS settled (its media is not wanted) but reporting it // as "skipped by the title filter" would say we decided not to have it, // when we decided to have its chat. `chatOnlySet` is what the loop // above actually classified (the chat is on disk); `chatOnlyIds` is the // outstanding half. if (chatOnlySet.has(id) || chatOnlyIds.has(id)) return false; const f = filesById.get(id); return !f || !videoHasAnyArtifact(f); }) .sort(), chatOnly: chatOnly.sort(), chatOnlyPending: chatOnlyPending.sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), autoSubsOnly: autoSubsOnly.sort(), downloadedAutoSubsOnly: downloadedAutoSubsOnly.sort(), supersededAutoSubs: supersededAutoSubs.sort(), needsCookies: needsCookies.sort(), digestWarnings: digestWarnings.sort(), }; // THE TWO BUCKET LANES GET A WORK LIST TOO — one entry per lane, so // `snapshot.backfill[op].ids` is where EVERY lane's dispatch list lives and // the runner has one way to ask. See foldBucketLaneEntry. const bucketLaneEntries: Record = {}; for (const lane of LANES) { const opId = bucketLaneOperationId(lane); if (!opId) continue; bucketLaneEntries[opId] = foldBucketLaneEntry(lane, { buckets: snapshotBuckets as Record, undownloadedIds, present: lane === "transcription" ? transcribed : downloaded, }); } const snapshot: ChannelSnapshot = { generatedAt: new Date().toISOString(), totals: { // Settled videos are excluded: they are stubs the operator asked us not to // fetch, not videos this channel has. Only the ones with a directory are // subtracted — a scan-settled id was never counted in the first place, // because the scan deliberately creates no directory. videos: videoDirNames.length - settledOnDisk.length, transcribed, downloaded, }, buckets: snapshotBuckets, metadataScan: { scanned: Object.keys(metadataScanStore.entries).length, errors: Object.keys(metadataScanStore.errors).length, unscanned: metadataScanUnscanned, ...(metadataScanStore.lastRun ? { lastRun: metadataScanStore.lastRun } : {}), }, digestEngines, backfill: { ...backfillCounts, ...bucketLaneEntries }, undownloadedIds, excludedFromDownload, keptCount: keptIds.size, availability, ...(maybeMissing ? { maybeMissing } : {}), // Newest-first by when we first heard of them: the most recent loss is the // one still worth chasing. missingNeverFetched: sets.missingNeverFetched .map((id) => ({ id, url: roster.entries[id]?.url ?? "", firstSeenAt: roster.entries[id]?.firstSeenAt ?? "", })) .sort((a, b) => b.firstSeenAt.localeCompare(a.firstSeenAt)), cleanupBytes: { transcribedWithAudio: transcribedWithAudioBytes, multipleAudioFormats: multipleAudioFormatsBytes, foreignAudio: foreignAudioBytes, }, // Unknown, never a partial sum, when the media tier could not be read. ...(mediaBytesKnown ? { totalAudioBytes, totalMediaBytes } : {}), totalTextBytes, totalClipsBytes, heldAudioBytes, heldAudioCounts, reclaimableAtRiskBytes, }; await writeChannelSnapshot(paths, slug, snapshot); return snapshot; } async function writeChannelSnapshot( paths: Paths, slug: string, snapshot: ChannelSnapshot, ): Promise { const file = snapshotPath(paths, slug); await writeJsonAtomic(file, snapshot); }