import type { Paths } from "yt-dlp-transcript-common/lib/paths"; import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels"; import { getChannelBriefs } from "../../lib/requestCache"; import { emptyHeldAudio, type ChannelSnapshot, type HeldAudio, } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { loadFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; // One channel's row in the cleanup ledger. The three reclaim estimates come // straight from the snapshot's precomputed `cleanupBytes` (already net of // do-not-clean + keep-latest protection). They OVERLAP — they're alternative // sweeps, not additive — so the headline total uses `transcribedBytes` (the // primary, safe "Clean audio for transcribed videos" sweep) only. export type CleanupRow = { channel: ChannelBrief; snapshot: ChannelSnapshot | null; included: boolean; transcribedBytes: number; extraFormatsBytes: number; foreignBytes: number; // Gating for the extra-format / wrong-format sweeps (need a target format). hasTargetFormat: boolean; // Housekeeping list size (failed transcriptions). failedTranscriptions: number; // --- What is holding the audio this row can't reclaim -------------------- // Straight off the snapshot the brief already carries (CleanupRow.snapshot has // been loaded and unread since this file was written) — no new disk reads. A // report written before the accounting existed has `measured: false`, and its // figures must render "—", never "0": a zero would claim a measurement nobody // took. measured: boolean; totalAudioBytes: number; heldBytes: HeldAudio; heldCounts: HeldAudio; reclaimableAtRiskBytes: number; // Ids for the release prescriptions. `downloadedNoTranscript` is the bucket // the Transcribe button runs, and `downloadedAutoSubsOnly` its ASR-only twin. transcribeIds: string[]; autoSubsIds: string[]; }; export type CleanupSummary = { rows: CleanupRow[]; // Reclaimable across INCLUDED channels (the badge/hero number = primary sweep). includedReclaimBytes: number; // The same primary number across ALL channels (so the UI can show what's // excluded). includedReclaimBytes <= totalReclaimBytes. totalReclaimBytes: number; // Secondary reclaim across included channels, for the breakdown bar. These // overlap with the primary and each other, so they're shown as separate // available sweeps, never summed into the headline. includedExtraFormatsBytes: number; includedForeignBytes: number; includedCount: number; excludedCount: number; // --- The sieve, summed over MEASURED, INCLUDED channels ------------------ // Measured-only on purpose: a channel whose report predates the hold // accounting contributes zeros to every gate, and quietly adding its // reclaimable bytes to the tail would make the column stop adding up. It is // counted in `staleSnapshotCount` and named instead. totalAudioBytes: number; heldBytes: HeldAudio; heldCounts: HeldAudio; // The remainder the four gates leave — equal to the hero figure whenever every // included channel has been measured. measuredReclaimBytes: number; // Of that remainder, what the pre-clean availability check is likely to refuse // (cached gone-from-source / needs-auth / error). An annotation, not a gate. atRiskBytes: number; // Included channels with no hold accounting yet — the "—" case. staleSnapshotCount: number; // Reclaimable bytes sitting in EXCLUDED channels. Not a hold: excludeFromCleanup // is display-only (the sweep never reads it), so this space is already // reclaimable today — it is merely hidden from the total. excludedReclaimBytes: number; }; function transcribedBytesOf(snap: ChannelSnapshot | null): number { return snap?.cleanupBytes?.transcribedWithAudio ?? 0; } function rowOf( channel: ChannelBrief, snapshot: ChannelSnapshot | null, failedTranscriptions: number, ): CleanupRow { return { channel, snapshot, included: !channel.config.excludeFromCleanup, transcribedBytes: transcribedBytesOf(snapshot), extraFormatsBytes: snapshot?.cleanupBytes?.multipleAudioFormats ?? 0, foreignBytes: snapshot?.cleanupBytes?.foreignAudio ?? 0, hasTargetFormat: Boolean(channel.config.audioFormat), failedTranscriptions, measured: snapshot?.heldAudioBytes !== undefined, totalAudioBytes: snapshot?.totalAudioBytes ?? 0, heldBytes: snapshot?.heldAudioBytes ?? emptyHeldAudio(), heldCounts: snapshot?.heldAudioCounts ?? emptyHeldAudio(), reclaimableAtRiskBytes: snapshot?.reclaimableAtRiskBytes ?? 0, transcribeIds: snapshot?.buckets.downloadedNoTranscript ?? [], autoSubsIds: snapshot?.buckets.downloadedAutoSubsOnly ?? [], }; } function addHeld(into: HeldAudio, from: HeldAudio): void { into.noTranscript += from.noTranscript; into.keepLatest += from.keepLatest; into.doNotClean += from.doNotClean; into.awaitingDiarization += from.awaitingDiarization; into.diarizationNeverClears += from.diarizationNeverClears; } export function heldTotalOf(held: HeldAudio): number { // diarizationNeverClears is a SUBSET of awaitingDiarization, so it is never // added — summing it would double-count the one hold that matters most. return ( held.noTranscript + held.keepLatest + held.doNotClean + held.awaitingDiarization ); } // ONE channel's ledger row — the row loadCleanupSummary builds for it, from the // same two reads (its brief, its failed-transcriptions list). For // `GET /api/ops/cleanup/`: asking about one channel must not pay for // every channel's snapshot. Null when the channel does not exist. export async function loadCleanupRow( paths: Paths, brief: ChannelBrief | null, ): Promise { if (!brief) return null; const failedT = await loadFailedTranscriptions(paths, brief.slug); return rowOf(brief, brief.snapshot, failedT.length); } export async function loadCleanupSummary( paths: Paths, ): Promise { const channels = await getChannelBriefs(paths); const rows = await Promise.all( channels.map(async (channel) => { // The brief already carries the snapshot; only the failed-transcriptions // list is still a per-channel read. const failedT = await loadFailedTranscriptions(paths, channel.slug); return rowOf(channel, channel.snapshot, failedT.length); }), ); // Sort: channels with the most reclaimable first; excluded sink to the bottom. rows.sort((a, b) => { if (a.included !== b.included) return a.included ? -1 : 1; return b.transcribedBytes - a.transcribedBytes; }); let includedReclaimBytes = 0; let totalReclaimBytes = 0; let includedExtraFormatsBytes = 0; let includedForeignBytes = 0; let includedCount = 0; let excludedCount = 0; let totalAudioBytes = 0; let measuredReclaimBytes = 0; let atRiskBytes = 0; let staleSnapshotCount = 0; let excludedReclaimBytes = 0; const heldBytes = emptyHeldAudio(); const heldCounts = emptyHeldAudio(); for (const r of rows) { totalReclaimBytes += r.transcribedBytes; if (r.included) { includedCount++; includedReclaimBytes += r.transcribedBytes; includedExtraFormatsBytes += r.extraFormatsBytes; includedForeignBytes += r.foreignBytes; if (r.measured) { totalAudioBytes += r.totalAudioBytes; measuredReclaimBytes += r.transcribedBytes; atRiskBytes += r.reclaimableAtRiskBytes; addHeld(heldBytes, r.heldBytes); addHeld(heldCounts, r.heldCounts); } else { staleSnapshotCount++; } } else { excludedCount++; excludedReclaimBytes += r.transcribedBytes; } } return { rows, includedReclaimBytes, totalReclaimBytes, includedExtraFormatsBytes, includedForeignBytes, includedCount, excludedCount, totalAudioBytes, heldBytes, heldCounts, measuredReclaimBytes, atRiskBytes, staleSnapshotCount, excludedReclaimBytes, }; } // One channel's reclaimable-audio row for the sidebar badge and the widget API. export type CleanableChannelRow = { slug: string; // Videos in the transcribed-with-audio bucket (the sweep's targets). count: number; // Reclaim estimate for the primary "clean audio" sweep, net of protection. bytes: number; }; // Lean per-channel breakdown for the sidebar badge and the widget API — the // primary "clean audio" reclaim over channels NOT excluded from the cleanup // total. Reads only the small snapshot JSONs (unlike loadCleanupSummary, which // also reads the failed-transcriptions list per channel), so it's cheap on the AutoRefresh // cadence / per widget poll. Sorted by reclaim, descending. // // The comment above was true of the snapshot reads and false of the line that // fetched the channel list: it used the corpus walk, so describing ~98 videos // cost ~474,559 file touches — on every render of the root layout, every // widget poll, and every auto-refresh tick. It reads the 65 briefs now. export async function cleanableChannels( paths: Paths, ): Promise { const channels = await getChannelBriefs(paths); const rows = channels.map((channel) => { if (channel.config.excludeFromCleanup) return null; const snap = channel.snapshot; const bytes = transcribedBytesOf(snap); const count = snap?.buckets.transcribedWithAudio?.length ?? 0; // Nothing to reclaim and nothing to sweep — drop the row. Zero-byte rows // contribute nothing to the total, so dropping them keeps it unchanged. if (bytes <= 0 && count <= 0) return null; return { slug: channel.slug, count, bytes }; }); return rows .filter((r): r is CleanableChannelRow => r !== null) .sort((a, b) => b.bytes - a.bytes); } // The headline reclaim number — a sum over the same rows, so the // excludeFromCleanup rule and the lean read live in exactly one place. export async function cleanableTotalBytes(paths: Paths): Promise { const rows = await cleanableChannels(paths); return rows.reduce((sum, r) => sum + r.bytes, 0); } // --- The release ledger ------------------------------------------------------ // // The prescription under the sieve, split by WHAT IT COSTS to release, because // that is a true property of the data and the sharpest answer to "what frees // space quickest": a run of the GPU/CPU lanes, or a setting nobody has to wait // for. Channels rank by held bytes descending inside each group. // How many channels per group get an action button. The button hands the job a // LIST OF IDS, so the cap is a payload bound, not a display choice: past this // the row still ranks and still reports its bytes, and links to the channel page // where the same bucket control lives. Starting the biggest run is what this // page is for; the long tail belongs to the channel. export const LEDGER_ACTION_LIMIT = 6; export type LedgerRow = { slug: string; name: string; bytes: number; count: number; // Undefined past LEDGER_ACTION_LIMIT — the row renders a link instead. transcribeIds?: string[]; // Only when the channel HAS ASR-only videos: those carry an English VTT, so // they are absent from downloadedNoTranscript and need the auto-subs lane. autoSubsIds?: string[]; }; export type CleanupLedger = { // Costs a run. noTranscript: { bytes: number; count: number; rows: LedgerRow[] }; awaitingDiarization: { bytes: number; count: number; neverClearsBytes: number; rows: LedgerRow[]; }; // Frees on a setting. notCounted: { bytes: number; channels: number; rows: LedgerRow[] }; keepLatest: { bytes: number; count: number; channels: number }; doNotClean: { bytes: number; count: number; channels: number }; }; function rankRows( rows: CleanupRow[], bytesOf: (r: CleanupRow) => number, countOf: (r: CleanupRow) => number, opts: { withIds?: boolean } = {}, ): LedgerRow[] { return rows .filter((r) => bytesOf(r) > 0 || countOf(r) > 0) .sort((a, b) => bytesOf(b) - bytesOf(a)) .map((r, i) => ({ slug: r.channel.slug, name: r.channel.config.name ?? r.channel.slug, bytes: bytesOf(r), count: countOf(r), ...(opts.withIds && i < LEDGER_ACTION_LIMIT ? { transcribeIds: r.transcribeIds, ...(r.autoSubsIds.length > 0 ? { autoSubsIds: r.autoSubsIds } : {}), } : {}), })); } export function buildCleanupLedger(summary: CleanupSummary): CleanupLedger { const included = summary.rows.filter((r) => r.included && r.measured); const excluded = summary.rows.filter((r) => !r.included); return { noTranscript: { bytes: summary.heldBytes.noTranscript, count: summary.heldCounts.noTranscript, rows: rankRows( included, (r) => r.heldBytes.noTranscript, (r) => r.heldCounts.noTranscript, { withIds: true }, ), }, awaitingDiarization: { bytes: summary.heldBytes.awaitingDiarization, count: summary.heldCounts.awaitingDiarization, neverClearsBytes: summary.heldBytes.diarizationNeverClears, rows: rankRows( included, (r) => r.heldBytes.awaitingDiarization, (r) => r.heldCounts.awaitingDiarization, ), }, // NOT a hold, and stated precisely: excludeFromCleanup is display-only (the // sweep never reads it), so this space is reclaimable TODAY. Filing it under // "held" would be the page's first lie — and it is in fact the fastest win // on the page: no compute, no setting change, just run the sweep. notCounted: { bytes: summary.excludedReclaimBytes, channels: excluded.length, rows: rankRows( excluded, (r) => r.transcribedBytes, (r) => r.snapshot?.buckets.transcribedWithAudio?.length ?? 0, ), }, keepLatest: { bytes: summary.heldBytes.keepLatest, count: summary.heldCounts.keepLatest, channels: included.filter((r) => r.heldBytes.keepLatest > 0).length, }, doNotClean: { bytes: summary.heldBytes.doNotClean, count: summary.heldCounts.doNotClean, channels: included.filter((r) => r.heldBytes.doNotClean > 0).length, }, }; }