// Shared per-video building blocks for fixing truncated ("incomplete") // transcripts, reused by the per-video action, the channel-level batch job, the // bulk-bar wrappers, and the global all-channels actions. Server-only (it does // fs + spawns yt-dlp/whisper) but NOT a "use server" module — it exports plain // helpers that take onLog/signal/tracker, which aren't serializable across the // server-action boundary. // // A truncated transcript means the audio download silently stopped early: the // audio on disk is itself short, so re-running whisper on it just reproduces the // short transcript. The fix MUST re-fetch the audio first. import { removeMediaFile } from "yt-dlp-transcript-common/lib/mediaTier-server"; import path from "node:path"; import { readdir } from "node:fs/promises"; import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { isRealAudioFile, isTranscriptVtt, } from "yt-dlp-transcript-common/lib/videoStatus"; import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transcribeOne"; import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos"; import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import type { TaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; function videoDirOf(paths: Paths, slug: string, videoId: string): string { return path.join(paths.channelsDir, slug, "data", videoId); } // The current snapshot's truncated-transcript ids for a channel. Empty when the // snapshot is missing or has no flagged videos (older snapshots lack the bucket). export async function incompleteIdsForChannel( slug: string, paths: Paths = getPaths(), ): Promise { const snap = await readChannelSnapshot(paths, slug); const ids = snap?.buckets?.incompleteTranscript; return Array.isArray(ids) ? ids : []; } // The current snapshot's short-audio ids for a channel (downloads the duration // guard flagged as truncated at the source). Empty when the snapshot is missing // or lacks the bucket. The same fixIncompleteTranscriptOne re-download path fixes // these — it deletes the kept stub and re-fetches (now via the per-source // default, i.e. `original` for Odysee). export async function shortAudioIdsForChannel( slug: string, paths: Paths = getPaths(), ): Promise { const snap = await readChannelSnapshot(paths, slug); const ids = snap?.buckets?.shortAudio; return Array.isArray(ids) ? ids : []; } // Re-download the full audio and re-transcribe one video in place. The new // transcript overwrites transcript.json (transcribeWithWorker) and normalize // regenerates transcript.cues.json, so there is never a window with no // transcript. Throws on failure so the batch loop can record it per-id. export async function fixIncompleteTranscriptOne(opts: { slug: string; videoId: string; config: ChannelConfig; paths: Paths; onLog: (line: string) => void; signal: AbortSignal; tracker?: TaskTracker; }): Promise { const { slug, videoId, config, paths, onLog, signal, tracker } = opts; const videoDir = videoDirOf(paths, slug, videoId); const audioFormat = config.audioFormat ?? "mp3"; // BEFORE the delete below, not after. This function removes the truncated // audio so the re-download doesn't see it as already present — refusing on a // full disk once that audio is gone would leave the video with neither the // stub nor a replacement. Latching, because the channel-level batch calls this // in a loop unattended. const gate = await diskGate(paths, getSettings(), { dir: path.join(paths.channelsDir, slug, "data"), }); if (!gate.ok) { throw new Error( `Cannot re-download audio for ${videoId}: ${gate.message}. ` + `Free up space or lower the floor in Settings.`, ); } const url = await findVideoSourceUrl(paths, slug, videoId, config); if (!url) { throw new Error( `Could not determine the video URL for ${videoId}: no metadata.info.json and the playlist does not contain a matching entry.`, ); } // Remove the truncated audio so the download re-fetches the full file rather // than seeing it as already present. const entries = await readdir(videoDir).catch(() => [] as string[]); for (const name of entries.filter(isRealAudioFile)) { await removeMediaFile(videoDir, name); onLog(`Removed truncated audio ${name}.`); } onLog(`Re-downloading audio for ${videoId}…`); await runYtdlp({ channelSlug: slug, mode: "download-one-audio", channelConfig: config, paths, onLog, signal, singleVideoUrl: url, audioFormatOverride: audioFormat, }); await transcribeWithWorker({ paths, videoDir, videoId, audioFilename: `audio.${audioFormat}`, tracker, onLog, signal, }); } // Clear a truncated transcript so the video drops back into the normal pending // pipeline: delete every real audio file, the whisper transcript.json + derived // transcript.cues.json, and any transcript VTT track. With no remaining artifact // the snapshot re-buckets it as undownloaded → (after re-download) // downloadedNoTranscript, where auto-download/auto-transcribe (or the manual // "Download missing" / "Transcribe pending" actions) reprocess it. Keeps // metadata.info.json (needed to resolve the URL on re-download). export async function clearIncompleteTranscriptOne(opts: { slug: string; videoId: string; paths: Paths; }): Promise<{ removed: number }> { const { slug, videoId, paths } = opts; const videoDir = videoDirOf(paths, slug, videoId); const dataDir = path.resolve(paths.channelsDir, slug, "data"); const resolved = path.resolve(videoDir); // Belt-and-suspenders: never delete outside the channel's data dir. if (path.dirname(resolved) !== dataDir) { throw new Error( `Refusing to clear: video path resolved outside the data dir (${videoId})`, ); } const entries = await readdir(resolved).catch(() => [] as string[]); const toRemove = entries.filter( (name) => isRealAudioFile(name) || isTranscriptVtt(name) || name === "transcript.json" || name === "transcript.cues.json", ); for (const name of toRemove) { // Through its link when tiered (release 17): the bytes go too. await removeMediaFile(resolved, name); } return { removed: toRemove.length }; }