commit 54305634cbeb1a3b3bb50f8833df13213b299f62 parent 931b48980c8611e874bb1ba04e9bea7eabc9ef8b Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Fri, 29 May 2026 22:52:45 -0400 Global cleaning and space estimation Diffstat:
15 files changed, 536 insertions(+), 13 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -18,6 +18,7 @@ import { loadAvailability, resolveEffectiveAvailability, } from "../lib/availability-server"; +import { isDoNotClean } from "../lib/doNotClean-server"; import type { Paths } from "../lib/paths"; import { extractVideoId } from "../ytdlp/runYtdlp"; import { reconcileVideoDirs } from "./reconcileVideoDirs"; @@ -47,6 +48,11 @@ export type ChannelSnapshot = { downloadedNoTranscript: string[]; untranscoded: string[]; multipleAudioFormats: string[]; + // Dirs with a whisper transcript AND audio still on disk — the audio is + // redundant and can be cleaned. Mirrors what cleanAudioFromTranscribed + // removes. Excludes "do not clean"–marked dirs. Optional: snapshots written + // before this field existed lack it; readers must default to []. + transcribedWithAudio: string[]; untranscribable: string[]; noMetadata: string[]; failedListed: string[]; @@ -194,12 +200,14 @@ export async function generateChannelSnapshot( } const availability = await loadAvailability(dir); const effectiveAvailability = await resolveEffectiveAvailability(dir); + const doNotClean = await isDoNotClean(dir); return { id, files, nativeId, availability, effectiveAvailability, + doNotClean, }; }), ), @@ -231,10 +239,18 @@ export async function generateChannelSnapshot( } } + // Videos the user has opted out of cleanup for (archived media). Filtered out + // of the cleanup buckets only — NOT of the download/transcribe buckets. + const doNotCleanIds = new Set<string>(); + for (const v of perVideo) { + if (v.doNotClean) doNotCleanIds.add(v.id); + } + const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; const untranscoded: string[] = []; const multipleAudioFormats: string[] = []; + const transcribedWithAudio: string[] = []; const untranscribable: string[] = []; const noMetadata: string[] = []; const partialDownloads: string[] = []; @@ -254,11 +270,19 @@ export async function generateChannelSnapshot( if ( targetAudioFile && files.audioFiles.includes(targetAudioFile) && - files.audioFiles.length > 1 + files.audioFiles.length > 1 && + !doNotCleanIds.has(id) ) { multipleAudioFormats.push(id); } if ( + files.hasWhisper && + files.audioFiles.length > 0 && + !doNotCleanIds.has(id) + ) { + transcribedWithAudio.push(id); + } + if ( files.audioFiles.length === 0 && files.partAudioFiles.length > 0 && !excludedById.has(id) @@ -326,6 +350,7 @@ export async function generateChannelSnapshot( downloadedNoTranscript: downloadedNoTranscript.sort(), untranscoded: untranscoded.sort(), multipleAudioFormats: multipleAudioFormats.sort(), + transcribedWithAudio: transcribedWithAudio.sort(), untranscribable: untranscribable.sort(), noMetadata: noMetadata.sort(), failedListed: failedListed.filter((id) => !excludedById.has(id)), diff --git a/common/controller/cleanAudioFromTranscribed.ts b/common/controller/cleanAudioFromTranscribed.ts @@ -1,6 +1,7 @@ import path from "node:path"; import fs from "fs-extra"; import type { Paths } from "../lib/paths"; +import { isDoNotClean } from "../lib/doNotClean-server"; const { pathExists, readdir, remove } = fs; @@ -15,6 +16,7 @@ export type CleanAudioResult = { inspected: number; cleanedDirs: number; removedFiles: number; + skipped: number; }; export async function cleanAudioFromTranscribed({ @@ -27,12 +29,13 @@ export async function cleanAudioFromTranscribed({ const dataDir = path.join(paths.channelsDir, channelSlug, "data"); if (!(await pathExists(dataDir))) { log(`No data directory for ${channelSlug}`); - return { inspected: 0, cleanedDirs: 0, removedFiles: 0 }; + return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 }; } const dirs = await readdir(dataDir); let cleanedDirs = 0; let removedFiles = 0; + let skipped = 0; for (const id of dirs) { if (signal?.aborted) break; @@ -47,6 +50,11 @@ export async function cleanAudioFromTranscribed({ !e.endsWith(".part"), ); if (audioFiles.length === 0) continue; + if (await isDoNotClean(videoDir)) { + log(`Skipped ${id} (marked do not clean)`); + skipped++; + continue; + } for (const f of audioFiles) { await remove(path.join(videoDir, f)); log(`Removed ${id}/${f}`); @@ -55,8 +63,9 @@ export async function cleanAudioFromTranscribed({ cleanedDirs++; } + const skippedNote = skipped > 0 ? ` Skipped ${skipped} (do not clean).` : ""; log( - `Cleaned ${removedFiles} audio file(s) from ${cleanedDirs} of ${dirs.length} video dir(s).`, + `Cleaned ${removedFiles} audio file(s) from ${cleanedDirs} of ${dirs.length} video dir(s).${skippedNote}`, ); - return { inspected: dirs.length, cleanedDirs, removedFiles }; + return { inspected: dirs.length, cleanedDirs, removedFiles, skipped }; } diff --git a/common/controller/cleanExtraAudioFormats.ts b/common/controller/cleanExtraAudioFormats.ts @@ -1,6 +1,7 @@ import path from "node:path"; import fs from "fs-extra"; import type { Paths } from "../lib/paths"; +import { isDoNotClean } from "../lib/doNotClean-server"; import { readChannelConfig } from "./channels"; const { pathExists, readdir, remove } = fs; @@ -16,6 +17,7 @@ export type CleanExtraAudioFormatsResult = { inspected: number; cleanedDirs: number; removedFiles: number; + skipped: number; }; export async function cleanExtraAudioFormats({ @@ -30,18 +32,19 @@ export async function cleanExtraAudioFormats({ log( `No audioFormat configured for ${channelSlug}; nothing to clean (set the channel's audio format first).`, ); - return { inspected: 0, cleanedDirs: 0, removedFiles: 0 }; + return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 }; } const targetAudioFile = `audio.${config.audioFormat}`; const dataDir = path.join(paths.channelsDir, channelSlug, "data"); if (!(await pathExists(dataDir))) { log(`No data directory for ${channelSlug}`); - return { inspected: 0, cleanedDirs: 0, removedFiles: 0 }; + return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 }; } const dirs = await readdir(dataDir); let cleanedDirs = 0; let removedFiles = 0; + let skipped = 0; for (const id of dirs) { if (signal?.aborted) { @@ -60,6 +63,11 @@ export async function cleanExtraAudioFormats({ !e.endsWith(".part"), ); if (extras.length === 0) continue; + if (await isDoNotClean(videoDir)) { + log(`Skipped ${id} (marked do not clean)`); + skipped++; + continue; + } for (const f of extras) { await remove(path.join(videoDir, f)); log(`Removed ${id}/${f}`); @@ -68,8 +76,9 @@ export async function cleanExtraAudioFormats({ cleanedDirs++; } + const skippedNote = skipped > 0 ? ` Skipped ${skipped} (do not clean).` : ""; log( - `Cleaned ${removedFiles} extra audio file(s) from ${cleanedDirs} of ${dirs.length} video dir(s) (kept ${targetAudioFile}).`, + `Cleaned ${removedFiles} extra audio file(s) from ${cleanedDirs} of ${dirs.length} video dir(s) (kept ${targetAudioFile}).${skippedNote}`, ); - return { inspected: dirs.length, cleanedDirs, removedFiles }; + return { inspected: dirs.length, cleanedDirs, removedFiles, skipped }; } diff --git a/common/lib/doNotClean-server.ts b/common/lib/doNotClean-server.ts @@ -0,0 +1,52 @@ +import path from "node:path"; +import { readFile, rename, rm, writeFile } from "node:fs/promises"; +import { + DO_NOT_CLEAN_FILENAME, + type DoNotCleanRecord, +} from "./doNotClean"; + +export function doNotCleanPath(videoDir: string): string { + return path.join(videoDir, DO_NOT_CLEAN_FILENAME); +} + +export async function loadDoNotClean( + videoDir: string, +): Promise<DoNotCleanRecord | null> { + try { + const raw = await readFile(doNotCleanPath(videoDir), "utf8"); + const parsed = JSON.parse(raw) as Partial<DoNotCleanRecord>; + if (typeof parsed?.setAt === "string") { + return parsed as DoNotCleanRecord; + } + // A parseable-but-malformed marker still means "protected" — fall back to + // an empty record rather than treating it as absent. + return { setAt: "" }; + } catch { + return null; + } +} + +// Presence of a valid sidecar = protected. Cheap existence check for the +// cleanup controllers' per-dir loops. +export async function isDoNotClean(videoDir: string): Promise<boolean> { + return (await loadDoNotClean(videoDir)) !== null; +} + +export async function setDoNotClean( + videoDir: string, + enabled: boolean, + note?: string, +): Promise<void> { + const file = doNotCleanPath(videoDir); + if (!enabled) { + await rm(file, { force: true }); + return; + } + const record: DoNotCleanRecord = { + setAt: new Date().toISOString(), + ...(note ? { note } : {}), + }; + const tmp = `${file}.tmp-${process.pid}`; + await writeFile(tmp, JSON.stringify(record, null, 2) + "\n"); + await rename(tmp, file); +} diff --git a/common/lib/doNotClean.ts b/common/lib/doNotClean.ts @@ -0,0 +1,18 @@ +// Client-safe types and constants. No node-only imports — the editor's +// video panel is a "use client" file that may pull this in for the toggle UI. +// Server-only I/O lives in doNotClean-server.ts. + +// Per-video "do not clean" marker: when this sidecar exists in a video dir, the +// cleanup controllers (cleanAudioFromTranscribed / cleanExtraAudioFormats) skip +// the dir and the snapshot's cleanup buckets exclude its id. Used to archive +// media we want to keep (e.g. a deleted / members-only upload). +// +// Marker semantics: presence of the file = protected. There is no stored +// `enabled: false` state — toggling off removes the file — which keeps +// isDoNotClean a simple existence check for the controllers' hot loops. +export type DoNotCleanRecord = { + setAt: string; + note?: string; +}; + +export const DO_NOT_CLEAN_FILENAME = "do-not-clean.json"; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,8 @@ # Changelog ## [Unreleased] +- **Cleanup tasks on `/actionable`.** The global Actionable view now surfaces two housekeeping sections alongside the download/transcribe backlog: **channels with cleanable transcribed audio** (videos that have a whisper transcript but still keep their audio on disk) and **channels with extra audio formats** (videos with leftover audio files beside the channel's configured format), each with a per-channel count and an inline button. Because these delete files, the buttons pop a confirm dialog before queuing. The dashboard "Needs attention" card is unchanged — it still tracks only download/transcribe work. +- **Per-video "do not clean" archive toggle.** Each video detail page has a new **Archive media** section to mark a video "do not clean". Marked videos are skipped by both cleanup jobs (their audio is preserved for archiving) and excluded from the cleanup counts on `/actionable`. The marker is reversible from the same toggle, and an "archived" badge shows on the video page while it's set. - **Twitch.tv support.** Twitch is now a first-class platform: Twitch channel/VOD URLs are auto-detected, "Twitch" is selectable in the channel form's platform dropdown and the charts platform filter, videos play via an in-browser Twitch embed, and downloads are queued on a `platform:twitch` queue like the other platforms. - **Downloads always land in the canonical `data/<id>/` dir.** Previously yt-dlp chose the directory from its own extractor id, which diverges from the app's URL-derived id on Twitch (`v<id>` prefix), Rumble, and Odysee — so a video's audio/metadata and its `download-outcome.json` could end up split across two sibling dirs and the editor couldn't resolve the video. The app now pins yt-dlp's output path per video, and a reconcile pass (run automatically when a channel's report regenerates) merges any pre-existing split dirs back together. A one-time migration command, `reconcile-video-dirs`, repairs all existing channels (`--dry-run` to preview). Sync and "download missing subtitles" now fetch one video per yt-dlp run (sync walks the channel a page at a time, downloading the diff against the archive and stopping once it reaches already-synced videos). - **Per-video download logs.** Each managed download now writes a `download.log` alongside the video's data, capturing that run's full yt-dlp output for after-the-fact debugging. diff --git a/editor/app/actionable/components/InlineActionButton.tsx b/editor/app/actionable/components/InlineActionButton.tsx @@ -4,12 +4,18 @@ import { useState, useTransition } from "react"; import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand"; import type { AudioFormat } from "yt-dlp-transcript-common/lib/channelConfig"; import { downloadMissingAction } from "../../channels/[slug]/pipelineActions"; -import { transcribeMissingAction } from "../../channels/[slug]/whisperActions"; +import { + cleanAudioAction, + cleanExtraAudioFormatsAction, + transcribeMissingAction, +} from "../../channels/[slug]/whisperActions"; import { refreshChannelSnapshotAction } from "../../channels/actions"; type Variant = | { kind: "downloadMissing"; slug: string } | { kind: "transcribeMissing"; slug: string; audioFormat?: AudioFormat } + | { kind: "cleanTranscribedAudio"; slug: string } + | { kind: "cleanExtraFormats"; slug: string } | { kind: "refreshReport"; slug: string }; type Status = @@ -22,9 +28,20 @@ type Status = const LABEL: Record<Variant["kind"], { idle: string; running: string }> = { downloadMissing: { idle: "Download missing", running: "Queuing…" }, transcribeMissing: { idle: "Transcribe pending", running: "Queuing…" }, + cleanTranscribedAudio: { idle: "Clean audio", running: "Queuing…" }, + cleanExtraFormats: { idle: "Clean extra formats", running: "Queuing…" }, refreshReport: { idle: "Refresh report", running: "Refreshing…" }, }; +// Cleanup actions delete files, so they pop a confirm() before queuing. +// Download/transcribe/refresh are non-destructive and stay unguarded. +const CONFIRM: Partial<Record<Variant["kind"], (slug: string) => string>> = { + cleanTranscribedAudio: (slug) => + `Delete redundant audio from every transcribed video in "${slug}"? Videos marked "do not clean" are skipped. This cannot be undone.`, + cleanExtraFormats: (slug) => + `Delete non-target audio formats from every video in "${slug}" (keeping the channel's configured format)? Videos marked "do not clean" are skipped. This cannot be undone.`, +}; + async function runAction(variant: Variant): Promise<StreamActionResult> { if (variant.kind === "downloadMissing") { return downloadMissingAction(variant.slug); @@ -38,6 +55,12 @@ async function runAction(variant: Variant): Promise<StreamActionResult> { variant.audioFormat, ); } + if (variant.kind === "cleanTranscribedAudio") { + return cleanAudioAction(variant.slug); + } + if (variant.kind === "cleanExtraFormats") { + return cleanExtraAudioFormatsAction(variant.slug); + } // refreshReport: returns an ActionResult, not a StreamActionResult — adapt. const result = await refreshChannelSnapshotAction(variant.slug); if (result && "error" in result) { @@ -59,6 +82,8 @@ export function InlineActionButton({ const ariaLabel = `${labels.idle.toLowerCase()} ${slug}`; function handleClick() { + const confirmMessage = CONFIRM[variant.kind]?.(slug); + if (confirmMessage && !window.confirm(confirmMessage)) return; setStatus({ kind: "running" }); startTransition(async () => { try { diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -18,6 +18,8 @@ export type ActionableSummary = { rows: ActionableRow[]; undownloaded: ActionableRow[]; untranscribed: ActionableRow[]; + cleanTranscribedAudio: ActionableRow[]; + cleanExtraFormats: ActionableRow[]; staleOrMissing: ActionableRow[]; }; @@ -56,6 +58,17 @@ export function actionableUntranscribedCount(row: ActionableRow): number { ); } +// Cleanup buckets are filtered by "do not clean" at snapshot-generation time, +// so the length is the actionable count directly (default undefined → 0 for +// snapshots written before the bucket existed). +export function actionableCleanTranscribedCount(row: ActionableRow): number { + return row.snapshot?.buckets.transcribedWithAudio?.length ?? 0; +} + +export function actionableCleanExtraFormatsCount(row: ActionableRow): number { + return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0; +} + export async function loadActionableSummary( paths: Paths, ): Promise<ActionableSummary> { @@ -79,9 +92,31 @@ export async function loadActionableSummary( (a, b) => actionableUntranscribedCount(b) - actionableUntranscribedCount(a), ); + const cleanTranscribedAudio = rows + .filter((r) => actionableCleanTranscribedCount(r) > 0) + .sort( + (a, b) => + actionableCleanTranscribedCount(b) - actionableCleanTranscribedCount(a), + ); + + const cleanExtraFormats = rows + .filter((r) => actionableCleanExtraFormatsCount(r) > 0) + .sort( + (a, b) => + actionableCleanExtraFormatsCount(b) - + actionableCleanExtraFormatsCount(a), + ); + const staleOrMissing = rows .filter(isStaleOrMissing) .sort((a, b) => a.channel.slug.localeCompare(b.channel.slug)); - return { rows, undownloaded, untranscribed, staleOrMissing }; + return { + rows, + undownloaded, + untranscribed, + cleanTranscribedAudio, + cleanExtraFormats, + staleOrMissing, + }; } diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -2,6 +2,8 @@ import type { Metadata } from "next"; import Link from "next/link"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { + actionableCleanExtraFormatsCount, + actionableCleanTranscribedCount, actionableUndownloadedCount, actionableUntranscribedCount, loadActionableSummary, @@ -30,6 +32,8 @@ export default async function ActionablePage() { const nothingPending = summary.undownloaded.length === 0 && summary.untranscribed.length === 0 && + summary.cleanTranscribedAudio.length === 0 && + summary.cleanExtraFormats.length === 0 && summary.staleOrMissing.length === 0; const sections: { config: SectionConfig; rows: ActionableRow[] }[] = [ @@ -73,6 +77,40 @@ export default async function ActionablePage() { }, { config: { + id: "clean-transcribed-audio", + title: "Channels with cleanable transcribed audio", + description: + "Videos that already have a whisper transcript but still keep their audio on disk. Run “Clean audio” to reclaim space. Videos marked “do not clean” are excluded.", + countLabel: "cleanable", + emptyLabel: "Nothing to clean.", + getCount: actionableCleanTranscribedCount, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "cleanTranscribedAudio", slug: r.channel.slug }} + /> + ), + }, + rows: summary.cleanTranscribedAudio, + }, + { + config: { + id: "clean-extra-formats", + title: "Channels with extra audio formats", + description: + "Videos that have the channel's configured audio format plus other leftover audio files. Run “Clean extra formats” to keep only the target format. Videos marked “do not clean” are excluded.", + countLabel: "extra formats", + emptyLabel: "Nothing to clean.", + getCount: actionableCleanExtraFormatsCount, + primaryAction: (r) => ( + <InlineActionButton + variant={{ kind: "cleanExtraFormats", slug: r.channel.slug }} + /> + ), + }, + rows: summary.cleanExtraFormats, + }, + { + config: { id: "stale-reports", title: "Channels with stale or missing reports", description: diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -17,6 +17,7 @@ export function normalizeBuckets( downloadedNoTranscript: raw?.downloadedNoTranscript ?? [], untranscoded: raw?.untranscoded ?? [], multipleAudioFormats: raw?.multipleAudioFormats ?? [], + transcribedWithAudio: raw?.transcribedWithAudio ?? [], untranscribable: raw?.untranscribable ?? [], noMetadata: raw?.noMetadata ?? [], failedListed: raw?.failedListed ?? [], diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -19,6 +19,7 @@ import { deleteVideoFileAction, downloadVideoPipelineAction, markVideoUntranscribableAction, + toggleDoNotCleanAction, transcodeAudioAction, transcribeOneAction, whisperVideoAction, @@ -40,6 +41,7 @@ type Props = { existingQueues: string[]; downloadOutcome: DownloadOutcomeRecord | null; channelAudioFormat?: AudioFormat; + doNotClean?: boolean; prevHref?: string; nextHref?: string; position?: { index: number; total: number }; @@ -120,6 +122,7 @@ export function VideoPanel({ existingQueues, downloadOutcome, channelAudioFormat, + doNotClean = false, prevHref, nextHref, position, @@ -166,6 +169,15 @@ export function VideoPanel({ downloadFailed={downloadFailed} /> {downloadOutcome && <DownloadOutcomeBadge outcome={downloadOutcome} />} + {doNotClean && ( + <div + role="status" + aria-label="media archived" + className="rounded border px-3 py-2 text-sm border-sky-300 bg-sky-50 text-sky-900 dark:border-sky-900 dark:bg-sky-950 dark:text-sky-200" + > + Media archived — cleanup will skip this video. + </div> + )} <PipelineStageCard id="download" title={noAudio ? "Download audio" : "Redownload audio"} @@ -240,6 +252,24 @@ export function VideoPanel({ )} <PipelineStageCard + id="archive-media" + title="Archive media" + summary={ + doNotClean + ? "Protected — cleanup will skip this video." + : "Exclude this video's media from cleanup jobs." + } + defaultOpen={true} + tone={doNotClean ? "ok" : "neutral"} + > + <DoNotCleanSection + slug={slug} + videoId={videoId} + doNotClean={doNotClean} + /> + </PipelineStageCard> + + <PipelineStageCard id="files" title="Files" summary={`${files.length} file${files.length === 1 ? "" : "s"} in this directory.`} @@ -670,6 +700,60 @@ function MarkUntranscribableSection({ ); } +function DoNotCleanSection({ + slug, + videoId, + doNotClean, +}: { + slug: string; + videoId: string; + doNotClean: boolean; +}) { + const [pending, startTransition] = useTransition(); + const [error, setError] = useState<string | null>(null); + const next = !doNotClean; + return ( + <div className="flex flex-col gap-2"> + <p className="text-sm text-zinc-500"> + {doNotClean + ? "This video is marked “do not clean”. The transcribed-audio and extra-format cleanup jobs will skip it, so its media is preserved for archiving. Reversible — allow cleanup to remove the marker." + : "Mark this video “do not clean” so the cleanup jobs never delete its audio (e.g. a deleted or members-only upload you want to keep). Reversible at any time."} + </p> + <button + type="button" + disabled={pending} + aria-label={ + next + ? `mark video ${videoId} do not clean` + : `allow cleanup for video ${videoId}` + } + onClick={() => { + setError(null); + startTransition(async () => { + const res = await toggleDoNotCleanAction(slug, videoId, next); + if (!res.ok) setError(res.error); + }); + }} + className="self-start px-3 py-2 rounded-md bg-zinc-900 dark:bg-zinc-100 text-zinc-100 dark:text-zinc-900 text-sm font-medium hover:opacity-90 disabled:opacity-50" + > + {pending + ? "Saving…" + : doNotClean + ? "Allow cleanup" + : "Mark do not clean"} + </button> + {error && ( + <span + className="text-sm text-red-600 dark:text-red-400" + aria-label="do not clean error" + > + {error} + </span> + )} + </div> + ); +} + function DeleteFileButton({ slug, videoId, diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -6,6 +6,7 @@ import { readdir, readFile, stat } from "node:fs/promises"; import type { Dirent } from "node:fs"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server"; +import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { platformQueueKey, @@ -93,9 +94,9 @@ export default async function VideoDetailPage({ if (!config) notFound(); const dirData = await loadVideoDir(slug, id); const meta = await loadMeta(slug, id); - const downloadOutcome = await loadDownloadOutcome( - path.join(getPaths().channelsDir, slug, "data", id), - ); + const videoDir = path.join(getPaths().channelsDir, slug, "data", id); + const downloadOutcome = await loadDownloadOutcome(videoDir); + const doNotClean = await isDoNotClean(videoDir); const registry = getRegistry(); const existingQueues = registry.activeQueueNames(); @@ -171,6 +172,7 @@ export default async function VideoDetailPage({ existingQueues={existingQueues} downloadOutcome={downloadOutcome} channelAudioFormat={config.audioFormat} + doNotClean={doNotClean} /> </div> ); diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -15,6 +15,7 @@ import { queueKeyForUrl, } from "yt-dlp-transcript-common/lib/platform"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { setDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode"; import { transcribeOneVideo } from "yt-dlp-transcript-common/controller/transcribeOne"; @@ -340,3 +341,23 @@ export async function markVideoUntranscribableAction( revalidatePath(`/channels/${slug}`); return { ok: true }; } + +export async function toggleDoNotCleanAction( + slug: string, + videoId: string, + enabled: boolean, +): Promise<{ ok: true } | { ok: false; error: string }> { + const videoDir = videoDirOf(slug, videoId); + try { + const s = await stat(videoDir); + if (!s.isDirectory()) { + return { ok: false, error: `Video directory not found: ${videoId}` }; + } + } catch { + return { ok: false, error: `Video directory not found: ${videoId}` }; + } + await setDoNotClean(videoDir, enabled); + revalidatePath(`/channels/${slug}/videos/${videoId}`); + revalidatePath(`/channels/${slug}`); + return { ok: true }; +} diff --git a/editor/e2e/cleanup-actionable.spec.ts b/editor/e2e/cleanup-actionable.spec.ts @@ -0,0 +1,128 @@ +import { mkdir, writeFile } from "node:fs/promises"; +import { dirname } from "node:path"; +import { test, expect } from "@playwright/test"; +import { resetData, resolvePath } from "./helpers"; + +const SLUG = "test-transcribe"; +const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; + +async function writeJson(relPath: string, value: unknown): Promise<void> { + const full = resolvePath(relPath); + await mkdir(dirname(full), { recursive: true }); + await writeFile(full, JSON.stringify(value, null, 2) + "\n"); +} + +// Seed a snapshot with the two cleanup buckets populated. transcribedWithAudio +// drives the "cleanable transcribed audio" section; multipleAudioFormats drives +// the "extra audio formats" section. +async function seedCleanupSnapshot(): Promise<void> { + await writeJson(SNAPSHOT_REL, { + generatedAt: "2026-05-27T12:00:00.000Z", + totals: { videos: 3, transcribed: 2, downloaded: 3 }, + buckets: { + noTranscript: [], + downloadedNoTranscript: [], + untranscoded: [], + multipleAudioFormats: ["vidA"], + transcribedWithAudio: ["vidA", "vidB"], + untranscribable: [], + noMetadata: [], + failedListed: [], + missingFromArchive: [], + duplicateDirs: [], + partialDownloads: [], + }, + undownloadedIds: [], + }); + await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {}); +} + +test("surfaces the two cleanup sections with per-channel counts", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedCleanupSnapshot(); + await page.goto("/actionable"); + + const transcribed = page.getByRole("region", { + name: "clean-transcribed-audio", + exact: true, + }); + await expect(transcribed).toBeVisible(); + const transcribedRow = transcribed.getByLabel( + `clean-transcribed-audio row ${SLUG}`, + ); + await expect(transcribedRow).toBeVisible(); + await expect(transcribedRow).toContainText("2"); + + const extras = page.getByRole("region", { + name: "clean-extra-formats", + exact: true, + }); + await expect(extras).toBeVisible(); + await expect(extras.getByLabel(`clean-extra-formats row ${SLUG}`)).toContainText( + "1", + ); +}); + +test("inline 'Clean audio' queues a clean-audio-transcribed job when confirmed", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedCleanupSnapshot(); + await page.goto("/actionable"); + + page.on("dialog", (d) => d.accept()); + await page.getByRole("button", { name: `clean audio ${SLUG}` }).click(); + await expect(page.getByLabel(`clean audio ${SLUG} job`)).toBeVisible({ + timeout: 10_000, + }); + + await page.goto("/jobs"); + await expect( + page.getByRole("row").filter({ hasText: "clean-audio-transcribed" }).first(), + ).toBeVisible({ timeout: 10_000 }); +}); + +test("inline 'Clean extra formats' queues a clean-extra-audio-formats job when confirmed", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedCleanupSnapshot(); + await page.goto("/actionable"); + + page.on("dialog", (d) => d.accept()); + await page + .getByRole("button", { name: `clean extra formats ${SLUG}` }) + .click(); + await expect( + page.getByLabel(`clean extra formats ${SLUG} job`), + ).toBeVisible({ timeout: 10_000 }); + + await page.goto("/jobs"); + await expect( + page + .getByRole("row") + .filter({ hasText: "clean-extra-audio-formats" }) + .first(), + ).toBeVisible({ timeout: 10_000 }); +}); + +test("dismissing the confirm dialog does not queue a cleanup job", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedCleanupSnapshot(); + await page.goto("/actionable"); + + page.on("dialog", (d) => d.dismiss()); + await page.getByRole("button", { name: `clean audio ${SLUG}` }).click(); + + // No job indicator should appear after a dismissed confirm. + await expect(page.getByLabel(`clean audio ${SLUG} job`)).toHaveCount(0); + + await page.goto("/jobs"); + await expect( + page.getByRole("row").filter({ hasText: "clean-audio-transcribed" }), + ).toHaveCount(0); +}); diff --git a/editor/e2e/do-not-clean.spec.ts b/editor/e2e/do-not-clean.spec.ts @@ -0,0 +1,74 @@ +import { writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, readJson, resetData, resolvePath } from "./helpers"; + +const SLUG = "test-transcribe"; +const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; + +function dataRel(videoId: string, file: string): string { + return `test-transcripts/channels/${SLUG}/data/${videoId}/${file}`; +} + +// Give vidA and vidB a whisper transcript so both qualify for the +// transcribed-audio cleanup. vidA gets protected, vidB does not. +async function seedTranscript(videoId: string): Promise<void> { + await writeFile( + resolvePath(dataRel(videoId, "transcript.json")), + '{"transcription":[]}\n', + ); +} + +test("'do not clean' protects a video's audio from cleanup; toggling off restores it", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await seedTranscript("vidA"); + await seedTranscript("vidB"); + // The fixture seeds audio.m4a for each video; config audioFormat is m4a. + await fetch("http://localhost:3011/api/test/invalidate-cache").catch(() => {}); + + // Mark vidA "do not clean" on its video page. + await page.goto(`/channels/${SLUG}/videos/vidA`); + await page + .getByRole("button", { name: "mark video vidA do not clean" }) + .click(); + await expect(page.getByLabel("media archived")).toBeVisible(); + + // Regenerating the snapshot (visiting the channel page) must omit the + // protected id from transcribedWithAudio while keeping the unprotected one. + await page.goto(`/channels/${SLUG}`); + const snapshot = await readJson<{ + buckets: { transcribedWithAudio?: string[] }; + }>(SNAPSHOT_REL); + expect(snapshot.buckets.transcribedWithAudio).toContain("vidB"); + expect(snapshot.buckets.transcribedWithAudio).not.toContain("vidA"); + + // Run the transcribed-audio cleanup from the channel's Cleanup stage. + await page.getByRole("button", { name: "Cleanup stage summary" }).click(); + await page.getByRole("button", { name: "Clean audio", exact: true }).click(); + const log = page.getByLabel("Clean audio output"); + await expect(log).toContainText("Skipped 1 (do not clean)", { + timeout: 30_000, + }); + + // vidB's audio is gone; vidA's is preserved. + expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(false); + expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(true); + + // Toggle vidA back to cleanable, then re-run cleanup. + await page.goto(`/channels/${SLUG}/videos/vidA`); + await page + .getByRole("button", { name: "allow cleanup for video vidA" }) + .click(); + await expect(page.getByLabel("media archived")).toHaveCount(0); + + await page.goto(`/channels/${SLUG}`); + await page.getByRole("button", { name: "Cleanup stage summary" }).click(); + await page.getByRole("button", { name: "Clean audio", exact: true }).click(); + await expect(page.getByLabel("Clean audio output")).toContainText( + "Cleaned 1 audio file", + { timeout: 30_000 }, + ); + + expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); +});