Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 25962b84275c0ee1d87495c66a4f483b0fef1c21
parent 4ff71bb2324b38f456e66a2c390a929d4bdfd392
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  4 Jun 2026 18:35:03 -0400

cleanup disk space estimate

Diffstat:
Mcommon/controller/channelSnapshot.ts | 39+++++++++++++++++++++++++++++++++++++--
Mcommon/lib/format.ts | 8++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/actionable/lib/loadActionable.ts | 10++++++++++
Meditor/app/actionable/page.tsx | 31+++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/stages/CleanupStage.tsx | 22++++++++++++++++++++++
Meditor/app/channels/[slug]/page.tsx | 2++
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 10++--------
Meditor/e2e/cleanup-actionable.spec.ts | 15++++++++++++---
Meditor/e2e/do-not-clean.spec.ts | 39++++++++++++++++++++++++++++++++++++++-
10 files changed, 163 insertions(+), 14 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -1,6 +1,6 @@ import path from "node:path"; import type { Dirent } from "node:fs"; -import { readdir, readFile, rename, writeFile } from "node:fs/promises"; +import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises"; import pLimit from "p-limit"; import { readArchive } from "../lib/archive"; import { @@ -71,6 +71,15 @@ export type ChannelSnapshot = { undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; availability?: AvailabilitySnapshot; + // Estimated bytes each cleanup operation would reclaim, mirroring the bucket + // counts: transcribedWithAudio sums every audio.* file in those dirs (all are + // removed); multipleAudioFormats sums the non-target audio.* files (the target + // format is kept). Optional: snapshots written before this field existed lack + // it; readers must default to 0. + cleanupBytes?: { + transcribedWithAudio: number; + multipleAudioFormats: number; + }; }; export function emptyExcludedFromDownload(): ExcludedFromDownload { @@ -191,6 +200,17 @@ export async function generateChannelSnapshot( limit(async () => { const dir = path.join(dataDir, id); const files = await readVideoFiles(dir, { checkUntranscribable: true }); + // Sizes of the real audio files, used to estimate how much disk a + // cleanup would reclaim. Best-effort: skip any file we can't stat. + const audioSizes: Record<string, number> = {}; + for (const name of files.audioFiles) { + try { + const st = await stat(path.join(dir, name)); + audioSizes[name] = st.size; + } catch { + // ignore — file vanished or is unreadable + } + } let nativeId: string | null = null; if (wantNativeIndex && files.hasMeta) { try { @@ -212,6 +232,7 @@ export async function generateChannelSnapshot( return { id, files, + audioSizes, nativeId, availability, effectiveAvailability, @@ -263,9 +284,11 @@ export async function generateChannelSnapshot( const noMetadata: string[] = []; const partialDownloads: string[] = []; const nonStandardVtt: string[] = []; + let transcribedWithAudioBytes = 0; + let multipleAudioFormatsBytes = 0; let transcribed = 0; let downloaded = 0; - for (const { id, files } of perVideo) { + for (const { id, files, audioSizes } of perVideo) { if (isVideoTranscribed(files)) transcribed++; if (isVideoDownloaded(files)) downloaded++; if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); @@ -283,6 +306,11 @@ export async function generateChannelSnapshot( !doNotCleanIds.has(id) ) { multipleAudioFormats.push(id); + for (const name of files.audioFiles) { + if (name !== targetAudioFile) { + multipleAudioFormatsBytes += audioSizes[name] ?? 0; + } + } } if ( files.hasWhisper && @@ -290,6 +318,9 @@ export async function generateChannelSnapshot( !doNotCleanIds.has(id) ) { transcribedWithAudio.push(id); + for (const name of files.audioFiles) { + transcribedWithAudioBytes += audioSizes[name] ?? 0; + } } if ( files.audioFiles.length === 0 && @@ -382,6 +413,10 @@ export async function generateChannelSnapshot( undownloadedIds, excludedFromDownload, availability, + cleanupBytes: { + transcribedWithAudio: transcribedWithAudioBytes, + multipleAudioFormats: multipleAudioFormatsBytes, + }, }; await writeChannelSnapshot(paths, slug, snapshot); diff --git a/common/lib/format.ts b/common/lib/format.ts @@ -3,6 +3,14 @@ export function formatDate(yyyymmdd: string): string { return `${yyyymmdd.slice(0, 4)}-${yyyymmdd.slice(4, 6)}-${yyyymmdd.slice(6, 8)}`; } +export function formatBytes(bytes: number): string { + if (bytes < 1024) return `${bytes} B`; + if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`; + if (bytes < 1024 * 1024 * 1024) + return `${(bytes / (1024 * 1024)).toFixed(1)} MB`; + return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`; +} + export function formatDuration(seconds: number): string { if (!seconds) return ""; const h = Math.floor(seconds / 3600); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Cleanup sections now estimate how much disk space they'll reclaim.** Both the `/actionable` cleanup rows and a channel's **Cleanup** stage show an estimated size next to each cleanup operation — the **Actionable** page gained an **Est. reclaim** column for "cleanable transcribed audio" and "extra audio formats", and the channel Cleanup stage shows an "Estimated space to reclaim: ~N" line under each of its two actions. The estimate is computed when a channel's report is generated (it sums the `audio.*` files each cleanup would delete — all audio for transcribed-audio cleanup, every non-target format for extra-format cleanup — and respects "do not clean"), so refresh a channel's report to populate it. Older reports without the figure show `~0 B` until refreshed. - **Videos whose English captions only exist under a regional/auto code are now indexed.** YouTube occasionally serves a video's English subtitles only as `en-US`, `en-en-US`, or `en-orig` with no plain `en` track, so yt-dlp wrote e.g. `transcript.en-US.vtt` but never `transcript.en.vtt`. The index only recognized the literal `transcript.en.vtt`, so such a video looked untranscribed and never appeared in search. Build index now falls back to the best available English VTT (preferring `en`, then `en-orig`, then regional `en-US`/`en-GB`, then auto-translated `en-en-*`) while ignoring true translation tracks like `es-en-US`; whisper also treats these as already-transcribed. Re-run **Build index** to pick up affected videos already on disk. The video detail page now reflects the same fallback (it previously hardcoded `transcript.en.vtt`, so a regional-only video showed as untranscribed there). - **Pick which subtitle track is a video's transcript.** The video page has a new **Transcript source** section listing every `transcript.<lang>.vtt` track, marking the current primary, with a **Set as transcript** button that promotes any track to the canonical `transcript.en.vtt` (the chosen track is copied, so the original stays and the choice is reversible — delete `transcript.en.vtt` to fall back to the automatic English pick, or pick another track to switch). Useful when the auto-picked track isn't the one you want, or when a video's only captions are a non-English track. - **Diagnostics: "Non-standard transcript VTT name" bucket.** A channel's Diagnostics now lists videos whose transcript rides on a non-canonical VTT name (e.g. only `transcript.en-US.vtt`, or a foreign-language track) rather than the standard `transcript.en.vtt` — each links to the video so you can normalize/switch the primary transcript. Refresh the channel snapshot to populate it. diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -69,6 +69,16 @@ export function actionableCleanExtraFormatsCount(row: ActionableRow): number { return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0; } +// Estimated bytes each cleanup would reclaim (default 0 for snapshots written +// before cleanupBytes existed). +export function actionableCleanTranscribedBytes(row: ActionableRow): number { + return row.snapshot?.cleanupBytes?.transcribedWithAudio ?? 0; +} + +export function actionableCleanExtraFormatsBytes(row: ActionableRow): number { + return row.snapshot?.cleanupBytes?.multipleAudioFormats ?? 0; +} + export async function loadActionableSummary( paths: Paths, ): Promise<ActionableSummary> { diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -1,8 +1,11 @@ import type { Metadata } from "next"; import Link from "next/link"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { + actionableCleanExtraFormatsBytes, actionableCleanExtraFormatsCount, + actionableCleanTranscribedBytes, actionableCleanTranscribedCount, actionableUndownloadedCount, actionableUntranscribedCount, @@ -23,6 +26,12 @@ type SectionConfig = { countLabel: string; emptyLabel: string; getCount: (row: ActionableRow) => number; + // Optional extra column rendered after the count (used by the cleanup + // sections to show estimated reclaimable disk space). + extraColumn?: { + label: string; + getValue: (row: ActionableRow) => string; + }; primaryAction: (row: ActionableRow) => React.ReactNode | null; }; @@ -84,6 +93,10 @@ export default async function ActionablePage() { countLabel: "cleanable", emptyLabel: "Nothing to clean.", getCount: actionableCleanTranscribedCount, + extraColumn: { + label: "Est. reclaim", + getValue: (r) => `~${formatBytes(actionableCleanTranscribedBytes(r))}`, + }, primaryAction: (r) => ( <InlineActionButton variant={{ kind: "cleanTranscribedAudio", slug: r.channel.slug }} @@ -101,6 +114,11 @@ export default async function ActionablePage() { countLabel: "extra formats", emptyLabel: "Nothing to clean.", getCount: actionableCleanExtraFormatsCount, + extraColumn: { + label: "Est. reclaim", + getValue: (r) => + `~${formatBytes(actionableCleanExtraFormatsBytes(r))}`, + }, primaryAction: (r) => ( <InlineActionButton variant={{ kind: "cleanExtraFormats", slug: r.channel.slug }} @@ -180,6 +198,11 @@ function Section({ <th className="text-right font-medium px-3 py-2"> {config.countLabel} </th> + {config.extraColumn && ( + <th className="text-right font-medium px-3 py-2 whitespace-nowrap"> + {config.extraColumn.label} + </th> + )} <th className="text-left font-medium px-3 py-2 whitespace-nowrap"> Last report </th> @@ -195,6 +218,7 @@ function Section({ key={row.channel.slug} row={row} count={config.getCount(row)} + extraValue={config.extraColumn?.getValue(row) ?? null} primaryAction={config.primaryAction(row)} sectionId={config.id} /> @@ -210,11 +234,13 @@ function Section({ function Row({ row, count, + extraValue, primaryAction, sectionId, }: { row: ActionableRow; count: number; + extraValue: string | null; primaryAction: React.ReactNode; sectionId: string; }) { @@ -247,6 +273,11 @@ function Row({ count )} </td> + {extraValue !== null && ( + <td className="px-3 py-2 text-right tabular-nums whitespace-nowrap text-zinc-600 dark:text-zinc-400"> + {extraValue} + </td> + )} <td className={`px-3 py-2 text-xs whitespace-nowrap ${ isStale diff --git a/editor/app/channels/[slug]/components/stages/CleanupStage.tsx b/editor/app/channels/[slug]/components/stages/CleanupStage.tsx @@ -2,6 +2,7 @@ import { useState } from "react"; import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { QueueControl } from "../../../../components/QueueControl"; import { cancelJobAction } from "../../../../jobs/actions"; import { @@ -15,6 +16,9 @@ type Props = { existingQueues: string[]; multipleAudioFormatIds: string[]; transcodeApplies: boolean; + // Estimated bytes each cleanup would reclaim, as of the last report. + transcribedAudioBytes: number; + extraFormatsBytes: number; }; export function CleanupStage({ @@ -22,6 +26,8 @@ export function CleanupStage({ existingQueues, multipleAudioFormatIds, transcodeApplies, + transcribedAudioBytes, + extraFormatsBytes, }: Props) { const defaultQueueKey = `channel:${slug}`; const [cleanQueue, setCleanQueue] = useState(defaultQueueKey); @@ -32,6 +38,7 @@ export function CleanupStage({ title="Clean audio for transcribed videos" desc="Delete audio.* files in any video directory that has a transcript.json (whisper output). YouTube videos with only auto-sub .vtt files have no audio and are skipped automatically." /> + <ReclaimEstimate bytes={transcribedAudioBytes} /> <StreamActionLog trigger={() => cleanAudioAction(slug, cleanQueue)} cancelAction={cancelJobAction} @@ -58,6 +65,7 @@ export function CleanupStage({ <ExtraAudioFormatsSection slug={slug} ids={multipleAudioFormatIds} + bytes={extraFormatsBytes} existingQueues={existingQueues} defaultQueueKey={defaultQueueKey} /> @@ -70,11 +78,13 @@ export function CleanupStage({ function ExtraAudioFormatsSection({ slug, ids, + bytes, existingQueues, defaultQueueKey, }: { slug: string; ids: string[]; + bytes: number; existingQueues: string[]; defaultQueueKey: string; }) { @@ -95,6 +105,7 @@ function ExtraAudioFormatsSection({ every video dir on disk regardless of this list — the count below reflects the last snapshot only. </p> + <ReclaimEstimate bytes={bytes} /> </div> <VideoIdList slug={slug} @@ -139,6 +150,17 @@ function ExtraAudioFormatsSection({ ); } +function ReclaimEstimate({ bytes }: { bytes: number }) { + return ( + <p className="text-sm text-zinc-600 dark:text-zinc-400"> + Estimated space to reclaim:{" "} + <span className="font-medium tabular-nums"> + {bytes > 0 ? `~${formatBytes(bytes)}` : "nothing to reclaim"} + </span> + </p> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -255,6 +255,8 @@ export default async function ChannelDetailPage({ existingQueues={existingQueues} multipleAudioFormatIds={buckets.multipleAudioFormats} transcodeApplies={transcodeApplies} + transcribedAudioBytes={snapshot.cleanupBytes?.transcribedWithAudio ?? 0} + extraFormatsBytes={snapshot.cleanupBytes?.multipleAudioFormats ?? 0} /> ), diagnostics: ( diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -9,6 +9,7 @@ import { type ChannelHandling, } from "yt-dlp-transcript-common/lib/channelConfig"; import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome"; +import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { QueueControl } from "../../../../../components/QueueControl"; import { cancelJobAction } from "../../../../../jobs/actions"; import { PipelineStageCard } from "../../../components/PipelineStageCard"; @@ -117,13 +118,6 @@ function mediaUrl(slug: string, videoId: string, filename: string): string { return `/api/channels/${encodeURIComponent(slug)}/videos/${encodeURIComponent(videoId)}/files/${encodeURIComponent(filename)}`; } -function formatSize(bytes: number): string { - if (bytes < 1024) return `${bytes} B`; - if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`; - if (bytes < 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024)).toFixed(1)} MB`; - return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`; -} - export function VideoPanel({ slug, videoId, @@ -678,7 +672,7 @@ function FilesList({ {f.name} </span> <span className="text-xs text-zinc-500"> - {formatSize(f.size)} · {new Date(f.mtime).toLocaleString()} + {formatBytes(f.size)} · {new Date(f.mtime).toLocaleString()} </span> </div> {kind === "audio" || kind === "video" ? ( diff --git a/editor/e2e/cleanup-actionable.spec.ts b/editor/e2e/cleanup-actionable.spec.ts @@ -33,6 +33,12 @@ async function seedCleanupSnapshot(): Promise<void> { partialDownloads: [], }, undownloadedIds: [], + // 2.5 MB transcribed-audio cleanup, 1 MB extra-format cleanup → exercises + // the "Est. reclaim" column formatting. + cleanupBytes: { + transcribedWithAudio: 2_621_440, + multipleAudioFormats: 1_048_576, + }, }); await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {}); } @@ -54,15 +60,18 @@ test("surfaces the two cleanup sections with per-channel counts", async ({ ); await expect(transcribedRow).toBeVisible(); await expect(transcribedRow).toContainText("2"); + // 2,621,440 bytes → 2.5 MB, shown in the new "Est. reclaim" column. + await expect(transcribedRow).toContainText("~2.5 MB"); const extras = page.getByRole("region", { name: "clean-extra-formats", exact: true, }); await expect(extras).toBeVisible(); - await expect(extras.getByLabel(`clean-extra-formats row ${SLUG}`)).toContainText( - "1", - ); + const extrasRow = extras.getByLabel(`clean-extra-formats row ${SLUG}`); + await expect(extrasRow).toContainText("1"); + // 1,048,576 bytes → 1.0 MB. + await expect(extrasRow).toContainText("~1.0 MB"); }); test("inline 'Clean audio' queues a clean-audio-transcribed job when confirmed", async ({ diff --git a/editor/e2e/do-not-clean.spec.ts b/editor/e2e/do-not-clean.spec.ts @@ -1,7 +1,17 @@ -import { writeFile } from "node:fs/promises"; +import { stat, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { pathExists, readJson, resetData, resolvePath } from "./helpers"; +// Mirror common/lib/format.ts formatBytes so the e2e assertion matches the UI +// without depending on the package path alias resolving in the test runner. +function formatBytes(bytes: number): string { + if (bytes < 1024) return `${bytes} B`; + if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`; + if (bytes < 1024 * 1024 * 1024) + return `${(bytes / (1024 * 1024)).toFixed(1)} MB`; + return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`; +} + const SLUG = "test-transcribe"; const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; @@ -72,3 +82,30 @@ test("'do not clean' protects a video's audio from cleanup; toggling off restore expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); }); + +test("snapshot records reclaimable cleanup bytes and the Cleanup stage shows the estimate", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedTranscript("vidA"); + await seedTranscript("vidB"); + await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {}); + + // Sum the audio that the transcribed-audio cleanup would delete (both videos + // have a whisper transcript and an audio.m4a on disk). + const sizeA = (await stat(resolvePath(dataRel("vidA", "audio.m4a")))).size; + const sizeB = (await stat(resolvePath(dataRel("vidB", "audio.m4a")))).size; + const expectedBytes = sizeA + sizeB; + + // Visiting the channel page regenerates the snapshot from disk. + await page.goto(`/channels/${SLUG}`); + const snapshot = await readJson<{ + cleanupBytes?: { transcribedWithAudio?: number }; + }>(SNAPSHOT_REL); + expect(snapshot.cleanupBytes?.transcribedWithAudio).toBe(expectedBytes); + + await page.getByRole("button", { name: "Cleanup stage summary" }).click(); + await expect( + page.getByText(`Estimated space to reclaim: ~${formatBytes(expectedBytes)}`), + ).toBeVisible(); +});