import { removeMediaFile } from "../lib/mediaTier-server"; import path from "node:path"; import fs from "fs-extra"; import type { Paths } from "../lib/paths"; import { isDoNotClean } from "../lib/doNotClean-server"; import { readVttProvenance } from "../lib/subtitleProvenance"; import { WHISPER_FILENAME, isEnglishVtt, listTranscriptVtts, } from "../lib/videoStatus"; const { pathExists, readdir } = fs; // Delete the YouTube auto-caption VTTs that our own transcript has superseded — // the manual counterpart to the `supersededAutoSubs` snapshot bucket, and the // one irreversible step in the replace-auto-captions lane. It is therefore // never wired into the auto-queue: the backup only disappears on an explicit // click, so an AI-vs-YouTube comparison stays possible until then. // // Deliberately narrow. A track is removed only when ALL of these hold: // - the dir has transcript.json (our transcript already won the index pick), // - the track is an ENGLISH one (the en / en-orig / en-US / en-en-* family) — // foreign-language tracks are separate content and are left alone, // - a fresh 4 KB provenance sniff still says "asr" — a manual or unclassifiable // track is never touched, // - the dir is not marked do-not-clean (same shield as the Clean-audio sweep). // // Recovering a purged track means re-running "Download missing subs". export type PurgeSupersededAutoSubsOptions = { channelSlug: string; paths: Paths; // When set, restrict the sweep to these ids (the bucket the button showed). // Omitted = walk every video dir, like the Clean-audio sweep. ids?: string[]; onLog?: (msg: string) => void; signal?: AbortSignal; }; export type PurgeSupersededAutoSubsResult = { inspected: number; cleanedDirs: number; removedFiles: number; skipped: number; }; export async function purgeSupersededAutoSubs({ channelSlug, paths, ids, onLog, signal, }: PurgeSupersededAutoSubsOptions): Promise { const log = onLog ?? ((m: string) => console.log(m)); const dataDir = path.join(paths.channelsDir, channelSlug, "data"); if (!(await pathExists(dataDir))) { log(`No data directory for ${channelSlug}`); return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 }; } const onDisk = await readdir(dataDir); const dirs = ids ? ids.filter((id) => onDisk.includes(id)) : onDisk; let cleanedDirs = 0; let removedFiles = 0; let skipped = 0; for (const id of dirs) { if (signal?.aborted) break; const videoDir = path.join(dataDir, id); const entries = await readdir(videoDir).catch(() => [] as string[]); // No transcript of our own yet — nothing has been superseded, so the VTT is // still this video's only transcript. Never touch it. if (!entries.includes(WHISPER_FILENAME)) continue; const englishVtts = listTranscriptVtts(entries).filter(isEnglishVtt); if (englishVtts.length === 0) continue; if (await isDoNotClean(videoDir)) { log(`Skipped ${id} (marked do not clean)`); skipped++; continue; } let removedHere = 0; for (const name of englishVtts) { const provenance = await readVttProvenance(videoDir, name); if (provenance !== "asr") { log(`Kept ${id}/${name} (${provenance} captions — not auto-generated)`); skipped++; continue; } await removeMediaFile(videoDir, name); // the one rm of a video-dir entry (release 17) log(`Removed ${id}/${name}`); removedFiles++; removedHere++; } if (removedHere > 0) cleanedDirs++; } const skippedNote = skipped > 0 ? ` Skipped ${skipped}.` : ""; log( `Purged ${removedFiles} superseded auto-caption file(s) from ${cleanedDirs} of ${dirs.length} video dir(s).${skippedNote}`, ); return { inspected: dirs.length, cleanedDirs, removedFiles, skipped, }; }