commit 25962b84275c0ee1d87495c66a4f483b0fef1c21
parent 4ff71bb2324b38f456e66a2c390a929d4bdfd392
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 4 Jun 2026 18:35:03 -0400
cleanup disk space estimate
Diffstat:
10 files changed, 163 insertions(+), 14 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -1,6 +1,6 @@
import path from "node:path";
import type { Dirent } from "node:fs";
-import { readdir, readFile, rename, writeFile } from "node:fs/promises";
+import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
import {
@@ -71,6 +71,15 @@ export type ChannelSnapshot = {
undownloadedIds: string[];
excludedFromDownload?: ExcludedFromDownload;
availability?: AvailabilitySnapshot;
+ // Estimated bytes each cleanup operation would reclaim, mirroring the bucket
+ // counts: transcribedWithAudio sums every audio.* file in those dirs (all are
+ // removed); multipleAudioFormats sums the non-target audio.* files (the target
+ // format is kept). Optional: snapshots written before this field existed lack
+ // it; readers must default to 0.
+ cleanupBytes?: {
+ transcribedWithAudio: number;
+ multipleAudioFormats: number;
+ };
};
export function emptyExcludedFromDownload(): ExcludedFromDownload {
@@ -191,6 +200,17 @@ export async function generateChannelSnapshot(
limit(async () => {
const dir = path.join(dataDir, id);
const files = await readVideoFiles(dir, { checkUntranscribable: true });
+ // Sizes of the real audio files, used to estimate how much disk a
+ // cleanup would reclaim. Best-effort: skip any file we can't stat.
+ const audioSizes: Record<string, number> = {};
+ for (const name of files.audioFiles) {
+ try {
+ const st = await stat(path.join(dir, name));
+ audioSizes[name] = st.size;
+ } catch {
+ // ignore — file vanished or is unreadable
+ }
+ }
let nativeId: string | null = null;
if (wantNativeIndex && files.hasMeta) {
try {
@@ -212,6 +232,7 @@ export async function generateChannelSnapshot(
return {
id,
files,
+ audioSizes,
nativeId,
availability,
effectiveAvailability,
@@ -263,9 +284,11 @@ export async function generateChannelSnapshot(
const noMetadata: string[] = [];
const partialDownloads: string[] = [];
const nonStandardVtt: string[] = [];
+ let transcribedWithAudioBytes = 0;
+ let multipleAudioFormatsBytes = 0;
let transcribed = 0;
let downloaded = 0;
- for (const { id, files } of perVideo) {
+ for (const { id, files, audioSizes } of perVideo) {
if (isVideoTranscribed(files)) transcribed++;
if (isVideoDownloaded(files)) downloaded++;
if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id);
@@ -283,6 +306,11 @@ export async function generateChannelSnapshot(
!doNotCleanIds.has(id)
) {
multipleAudioFormats.push(id);
+ for (const name of files.audioFiles) {
+ if (name !== targetAudioFile) {
+ multipleAudioFormatsBytes += audioSizes[name] ?? 0;
+ }
+ }
}
if (
files.hasWhisper &&
@@ -290,6 +318,9 @@ export async function generateChannelSnapshot(
!doNotCleanIds.has(id)
) {
transcribedWithAudio.push(id);
+ for (const name of files.audioFiles) {
+ transcribedWithAudioBytes += audioSizes[name] ?? 0;
+ }
}
if (
files.audioFiles.length === 0 &&
@@ -382,6 +413,10 @@ export async function generateChannelSnapshot(
undownloadedIds,
excludedFromDownload,
availability,
+ cleanupBytes: {
+ transcribedWithAudio: transcribedWithAudioBytes,
+ multipleAudioFormats: multipleAudioFormatsBytes,
+ },
};
await writeChannelSnapshot(paths, slug, snapshot);
diff --git a/common/lib/format.ts b/common/lib/format.ts
@@ -3,6 +3,14 @@ export function formatDate(yyyymmdd: string): string {
return `${yyyymmdd.slice(0, 4)}-${yyyymmdd.slice(4, 6)}-${yyyymmdd.slice(6, 8)}`;
}
+export function formatBytes(bytes: number): string {
+ if (bytes < 1024) return `${bytes} B`;
+ if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`;
+ if (bytes < 1024 * 1024 * 1024)
+ return `${(bytes / (1024 * 1024)).toFixed(1)} MB`;
+ return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`;
+}
+
export function formatDuration(seconds: number): string {
if (!seconds) return "";
const h = Math.floor(seconds / 3600);
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Cleanup sections now estimate how much disk space they'll reclaim.** Both the `/actionable` cleanup rows and a channel's **Cleanup** stage show an estimated size next to each cleanup operation — the **Actionable** page gained an **Est. reclaim** column for "cleanable transcribed audio" and "extra audio formats", and the channel Cleanup stage shows an "Estimated space to reclaim: ~N" line under each of its two actions. The estimate is computed when a channel's report is generated (it sums the `audio.*` files each cleanup would delete — all audio for transcribed-audio cleanup, every non-target format for extra-format cleanup — and respects "do not clean"), so refresh a channel's report to populate it. Older reports without the figure show `~0 B` until refreshed.
- **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. The video detail page now reflects the same fallback (it previously hardcoded `transcript.en.vtt`, so a regional-only video showed as untranscribed there).
- **Pick which subtitle track is a video's transcript.** The video page has a new **Transcript source** section listing every `transcript.<lang>.vtt` track, marking the current primary, with a **Set as transcript** button that promotes any track to the canonical `transcript.en.vtt` (the chosen track is copied, so the original stays and the choice is reversible — delete `transcript.en.vtt` to fall back to the automatic English pick, or pick another track to switch). Useful when the auto-picked track isn't the one you want, or when a video's only captions are a non-English track.
- **Diagnostics: "Non-standard transcript VTT name" bucket.** A channel's Diagnostics now lists videos whose transcript rides on a non-canonical VTT name (e.g. only `transcript.en-US.vtt`, or a foreign-language track) rather than the standard `transcript.en.vtt` — each links to the video so you can normalize/switch the primary transcript. Refresh the channel snapshot to populate it.
diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts
@@ -69,6 +69,16 @@ export function actionableCleanExtraFormatsCount(row: ActionableRow): number {
return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0;
}
+// Estimated bytes each cleanup would reclaim (default 0 for snapshots written
+// before cleanupBytes existed).
+export function actionableCleanTranscribedBytes(row: ActionableRow): number {
+ return row.snapshot?.cleanupBytes?.transcribedWithAudio ?? 0;
+}
+
+export function actionableCleanExtraFormatsBytes(row: ActionableRow): number {
+ return row.snapshot?.cleanupBytes?.multipleAudioFormats ?? 0;
+}
+
export async function loadActionableSummary(
paths: Paths,
): Promise<ActionableSummary> {
diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx
@@ -1,8 +1,11 @@
import type { Metadata } from "next";
import Link from "next/link";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import {
+ actionableCleanExtraFormatsBytes,
actionableCleanExtraFormatsCount,
+ actionableCleanTranscribedBytes,
actionableCleanTranscribedCount,
actionableUndownloadedCount,
actionableUntranscribedCount,
@@ -23,6 +26,12 @@ type SectionConfig = {
countLabel: string;
emptyLabel: string;
getCount: (row: ActionableRow) => number;
+ // Optional extra column rendered after the count (used by the cleanup
+ // sections to show estimated reclaimable disk space).
+ extraColumn?: {
+ label: string;
+ getValue: (row: ActionableRow) => string;
+ };
primaryAction: (row: ActionableRow) => React.ReactNode | null;
};
@@ -84,6 +93,10 @@ export default async function ActionablePage() {
countLabel: "cleanable",
emptyLabel: "Nothing to clean.",
getCount: actionableCleanTranscribedCount,
+ extraColumn: {
+ label: "Est. reclaim",
+ getValue: (r) => `~${formatBytes(actionableCleanTranscribedBytes(r))}`,
+ },
primaryAction: (r) => (
<InlineActionButton
variant={{ kind: "cleanTranscribedAudio", slug: r.channel.slug }}
@@ -101,6 +114,11 @@ export default async function ActionablePage() {
countLabel: "extra formats",
emptyLabel: "Nothing to clean.",
getCount: actionableCleanExtraFormatsCount,
+ extraColumn: {
+ label: "Est. reclaim",
+ getValue: (r) =>
+ `~${formatBytes(actionableCleanExtraFormatsBytes(r))}`,
+ },
primaryAction: (r) => (
<InlineActionButton
variant={{ kind: "cleanExtraFormats", slug: r.channel.slug }}
@@ -180,6 +198,11 @@ function Section({
<th className="text-right font-medium px-3 py-2">
{config.countLabel}
</th>
+ {config.extraColumn && (
+ <th className="text-right font-medium px-3 py-2 whitespace-nowrap">
+ {config.extraColumn.label}
+ </th>
+ )}
<th className="text-left font-medium px-3 py-2 whitespace-nowrap">
Last report
</th>
@@ -195,6 +218,7 @@ function Section({
key={row.channel.slug}
row={row}
count={config.getCount(row)}
+ extraValue={config.extraColumn?.getValue(row) ?? null}
primaryAction={config.primaryAction(row)}
sectionId={config.id}
/>
@@ -210,11 +234,13 @@ function Section({
function Row({
row,
count,
+ extraValue,
primaryAction,
sectionId,
}: {
row: ActionableRow;
count: number;
+ extraValue: string | null;
primaryAction: React.ReactNode;
sectionId: string;
}) {
@@ -247,6 +273,11 @@ function Row({
count
)}
</td>
+ {extraValue !== null && (
+ <td className="px-3 py-2 text-right tabular-nums whitespace-nowrap text-zinc-600 dark:text-zinc-400">
+ {extraValue}
+ </td>
+ )}
<td
className={`px-3 py-2 text-xs whitespace-nowrap ${
isStale
diff --git a/editor/app/channels/[slug]/components/stages/CleanupStage.tsx b/editor/app/channels/[slug]/components/stages/CleanupStage.tsx
@@ -2,6 +2,7 @@
import { useState } from "react";
import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog";
+import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import { QueueControl } from "../../../../components/QueueControl";
import { cancelJobAction } from "../../../../jobs/actions";
import {
@@ -15,6 +16,9 @@ type Props = {
existingQueues: string[];
multipleAudioFormatIds: string[];
transcodeApplies: boolean;
+ // Estimated bytes each cleanup would reclaim, as of the last report.
+ transcribedAudioBytes: number;
+ extraFormatsBytes: number;
};
export function CleanupStage({
@@ -22,6 +26,8 @@ export function CleanupStage({
existingQueues,
multipleAudioFormatIds,
transcodeApplies,
+ transcribedAudioBytes,
+ extraFormatsBytes,
}: Props) {
const defaultQueueKey = `channel:${slug}`;
const [cleanQueue, setCleanQueue] = useState(defaultQueueKey);
@@ -32,6 +38,7 @@ export function CleanupStage({
title="Clean audio for transcribed videos"
desc="Delete audio.* files in any video directory that has a transcript.json (whisper output). YouTube videos with only auto-sub .vtt files have no audio and are skipped automatically."
/>
+ <ReclaimEstimate bytes={transcribedAudioBytes} />
<StreamActionLog
trigger={() => cleanAudioAction(slug, cleanQueue)}
cancelAction={cancelJobAction}
@@ -58,6 +65,7 @@ export function CleanupStage({
<ExtraAudioFormatsSection
slug={slug}
ids={multipleAudioFormatIds}
+ bytes={extraFormatsBytes}
existingQueues={existingQueues}
defaultQueueKey={defaultQueueKey}
/>
@@ -70,11 +78,13 @@ export function CleanupStage({
function ExtraAudioFormatsSection({
slug,
ids,
+ bytes,
existingQueues,
defaultQueueKey,
}: {
slug: string;
ids: string[];
+ bytes: number;
existingQueues: string[];
defaultQueueKey: string;
}) {
@@ -95,6 +105,7 @@ function ExtraAudioFormatsSection({
every video dir on disk regardless of this list — the count below
reflects the last snapshot only.
</p>
+ <ReclaimEstimate bytes={bytes} />
</div>
<VideoIdList
slug={slug}
@@ -139,6 +150,17 @@ function ExtraAudioFormatsSection({
);
}
+function ReclaimEstimate({ bytes }: { bytes: number }) {
+ return (
+ <p className="text-sm text-zinc-600 dark:text-zinc-400">
+ Estimated space to reclaim:{" "}
+ <span className="font-medium tabular-nums">
+ {bytes > 0 ? `~${formatBytes(bytes)}` : "nothing to reclaim"}
+ </span>
+ </p>
+ );
+}
+
function Heading({ title, desc }: { title: string; desc: string }) {
return (
<div>
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -255,6 +255,8 @@ export default async function ChannelDetailPage({
existingQueues={existingQueues}
multipleAudioFormatIds={buckets.multipleAudioFormats}
transcodeApplies={transcodeApplies}
+ transcribedAudioBytes={snapshot.cleanupBytes?.transcribedWithAudio ?? 0}
+ extraFormatsBytes={snapshot.cleanupBytes?.multipleAudioFormats ?? 0}
/>
),
diagnostics: (
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -9,6 +9,7 @@ import {
type ChannelHandling,
} from "yt-dlp-transcript-common/lib/channelConfig";
import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome";
+import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import { QueueControl } from "../../../../../components/QueueControl";
import { cancelJobAction } from "../../../../../jobs/actions";
import { PipelineStageCard } from "../../../components/PipelineStageCard";
@@ -117,13 +118,6 @@ function mediaUrl(slug: string, videoId: string, filename: string): string {
return `/api/channels/${encodeURIComponent(slug)}/videos/${encodeURIComponent(videoId)}/files/${encodeURIComponent(filename)}`;
}
-function formatSize(bytes: number): string {
- if (bytes < 1024) return `${bytes} B`;
- if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`;
- if (bytes < 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024)).toFixed(1)} MB`;
- return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`;
-}
-
export function VideoPanel({
slug,
videoId,
@@ -678,7 +672,7 @@ function FilesList({
{f.name}
</span>
<span className="text-xs text-zinc-500">
- {formatSize(f.size)} · {new Date(f.mtime).toLocaleString()}
+ {formatBytes(f.size)} · {new Date(f.mtime).toLocaleString()}
</span>
</div>
{kind === "audio" || kind === "video" ? (
diff --git a/editor/e2e/cleanup-actionable.spec.ts b/editor/e2e/cleanup-actionable.spec.ts
@@ -33,6 +33,12 @@ async function seedCleanupSnapshot(): Promise<void> {
partialDownloads: [],
},
undownloadedIds: [],
+ // 2.5 MB transcribed-audio cleanup, 1 MB extra-format cleanup → exercises
+ // the "Est. reclaim" column formatting.
+ cleanupBytes: {
+ transcribedWithAudio: 2_621_440,
+ multipleAudioFormats: 1_048_576,
+ },
});
await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {});
}
@@ -54,15 +60,18 @@ test("surfaces the two cleanup sections with per-channel counts", async ({
);
await expect(transcribedRow).toBeVisible();
await expect(transcribedRow).toContainText("2");
+ // 2,621,440 bytes → 2.5 MB, shown in the new "Est. reclaim" column.
+ await expect(transcribedRow).toContainText("~2.5 MB");
const extras = page.getByRole("region", {
name: "clean-extra-formats",
exact: true,
});
await expect(extras).toBeVisible();
- await expect(extras.getByLabel(`clean-extra-formats row ${SLUG}`)).toContainText(
- "1",
- );
+ const extrasRow = extras.getByLabel(`clean-extra-formats row ${SLUG}`);
+ await expect(extrasRow).toContainText("1");
+ // 1,048,576 bytes → 1.0 MB.
+ await expect(extrasRow).toContainText("~1.0 MB");
});
test("inline 'Clean audio' queues a clean-audio-transcribed job when confirmed", async ({
diff --git a/editor/e2e/do-not-clean.spec.ts b/editor/e2e/do-not-clean.spec.ts
@@ -1,7 +1,17 @@
-import { writeFile } from "node:fs/promises";
+import { stat, writeFile } from "node:fs/promises";
import { test, expect } from "@playwright/test";
import { pathExists, readJson, resetData, resolvePath } from "./helpers";
+// Mirror common/lib/format.ts formatBytes so the e2e assertion matches the UI
+// without depending on the package path alias resolving in the test runner.
+function formatBytes(bytes: number): string {
+ if (bytes < 1024) return `${bytes} B`;
+ if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`;
+ if (bytes < 1024 * 1024 * 1024)
+ return `${(bytes / (1024 * 1024)).toFixed(1)} MB`;
+ return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`;
+}
+
const SLUG = "test-transcribe";
const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`;
@@ -72,3 +82,30 @@ test("'do not clean' protects a video's audio from cleanup; toggling off restore
expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false);
});
+
+test("snapshot records reclaimable cleanup bytes and the Cleanup stage shows the estimate", async ({
+ page,
+}) => {
+ await resetData("one-transcribe-channel-with-audio");
+ await seedTranscript("vidA");
+ await seedTranscript("vidB");
+ await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {});
+
+ // Sum the audio that the transcribed-audio cleanup would delete (both videos
+ // have a whisper transcript and an audio.m4a on disk).
+ const sizeA = (await stat(resolvePath(dataRel("vidA", "audio.m4a")))).size;
+ const sizeB = (await stat(resolvePath(dataRel("vidB", "audio.m4a")))).size;
+ const expectedBytes = sizeA + sizeB;
+
+ // Visiting the channel page regenerates the snapshot from disk.
+ await page.goto(`/channels/${SLUG}`);
+ const snapshot = await readJson<{
+ cleanupBytes?: { transcribedWithAudio?: number };
+ }>(SNAPSHOT_REL);
+ expect(snapshot.cleanupBytes?.transcribedWithAudio).toBe(expectedBytes);
+
+ await page.getByRole("button", { name: "Cleanup stage summary" }).click();
+ await expect(
+ page.getByText(`Estimated space to reclaim: ~${formatBytes(expectedBytes)}`),
+ ).toBeVisible();
+});