"use server"; import path from "node:path"; import { lstat, readdir, readFile, rm, stat } from "node:fs/promises"; import { revalidatePath } from "next/cache"; import { safeRevalidate } from "../../../../lib/safeRevalidate"; import { redirect } from "next/navigation"; import type { AudioFormat, ChannelConfig, } from "yt-dlp-transcript-common/lib/channelConfig"; import { AUDIO_FORMAT_VALUES } from "yt-dlp-transcript-common/lib/channelConfig"; import { isDownloadFormatPreset, resolveDownloadFormatPreset, resolveSourceVideoQuality, type DownloadFormatPreset, type SourceVideoQuality, } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; import { diskGate } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { audioFilesToRemove, isRealAudioFile, isTranscriptVtt, VTT_FILENAME, } from "yt-dlp-transcript-common/lib/videoStatus"; import { clipWindowQueueKey, downloadQueueKey, resolveQueueKey, } from "yt-dlp-transcript-common/lib/queueKeys"; import { windowInFlight } from "yt-dlp-transcript-common/jobs/windowJobs"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; import { setDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { KeepVideosError, keepVideosMatching, type KeepVideosField, type KeepVideosResult, } from "yt-dlp-transcript-common/controller/keepVideosMatching"; import { ChannelMediaUnreachableError, assertChannelTextReadable, channelTextStall, inspectChannelMedia, } from "yt-dlp-transcript-common/lib/channelMedia"; import { onDrive } from "yt-dlp-transcript-common/lib/storageHealth"; import { isTierable } from "yt-dlp-transcript-common/lib/mediaTier"; import { setExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server"; import { pinTranscript } from "yt-dlp-transcript-common/lib/transcriptPin-server"; import { readVideoTracks } from "yt-dlp-transcript-common/lib/captionTracks-server"; import type { AltTrack } from "yt-dlp-transcript-common/lib/captionTracks"; import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode"; import { removeMediaFile, removeVideoDirMedia, } from "yt-dlp-transcript-common/lib/mediaTier-server"; import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transcribeOne"; import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos"; import { loadSavedVideo, unpersistSavedVideo, } from "yt-dlp-transcript-common/lib/savedVideo-server"; import { savedVideoPath, type SavedVideoOrigin, } from "yt-dlp-transcript-common/lib/savedVideo"; import { MAX_CLIP_WINDOW_SECONDS, isFetchMaxHeight, type ClipProvenance, } from "yt-dlp-transcript-common/lib/clipWindow"; import { clipWindowPath, findContainingClipWindow, } from "yt-dlp-transcript-common/lib/clipWindow-server"; import { cutSavedVideoWindow, savedWindowSource, } from "yt-dlp-transcript-common/lib/savedVideoWindow-server"; import { fetchWindowManaged, type FetchWindowProvenance, } from "yt-dlp-transcript-common/ytdlp/fetchWindowManaged"; import { probeVideoHeight } from "yt-dlp-transcript-common/ytdlp/ffprobeDuration"; import { detectPlatform } from "yt-dlp-transcript-common/lib/platform"; import { applyTagAssignmentsAction } from "../../../../tags/actions"; import { heldPlatformRefusal, platformCooldownRemainingMs, recordDownloadBackoff, } from "yt-dlp-transcript-common/jobs/downloadBackoff"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; import { downloadOneManaged, sourceFetchFailure, } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { runRefreshMetadataJob } from "yt-dlp-transcript-common/controller/refreshVideoMetadataJob"; import { runManagedFunction, type StreamActionResult, } from "yt-dlp-transcript-common/jobs/streamCommand"; import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; import { fixIncompleteTranscriptOne } from "../../lib/fixIncompleteTranscript"; import { writeFileAtomic, writeJsonAtomic } from "yt-dlp-transcript-common/lib/jsonFile-server"; import { formValues, type FormErrorState } from "../../../../lib/formState"; function videoQueueKey(config: ChannelConfig, override: string | undefined): string { return resolveQueueKey(downloadQueueKey(config), override); } function videoDirOf(slug: string, videoId: string): string { return path.join(getPaths().channelsDir, slug, "data", videoId); } async function loadConfigOrError( slug: string, ): Promise< | { ok: true; config: ChannelConfig } | { ok: false; error: string } > { const config = await readChannelConfig(getPaths(), slug); if (!config) return { ok: false, error: `Channel "${slug}" not found` }; return { ok: true, config }; } export async function transcodeAudioAction( slug: string, videoId: string, sourceFilename: string, targetFormat: AudioFormat, queueKey?: string, ): Promise { if (!AUDIO_FORMAT_VALUES.includes(targetFormat)) { return { ok: false, error: `Unsupported target format: ${targetFormat}` }; } const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); const videoDir = videoDirOf(slug, videoId); return runManagedFunction({ kind: "transcode-audio", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal) => { await transcodeAudio({ paths, videoDir, sourceFilename, targetFormat, onLog, signal, }); safeRevalidate([`/channels/${slug}/videos/${videoId}`]); }, }); } export async function transcribeOneAction( slug: string, videoId: string, audioFilename: string, queueKey?: string, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); const videoDir = videoDirOf(slug, videoId); return runManagedFunction({ kind: "transcribe-one", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { await transcribeWithWorker({ paths, videoDir, videoId, audioFilename, tracker: makeTaskTracker(ctx, onLog), onLog, signal, }); safeRevalidate([`/channels/${slug}/videos/${videoId}`]); }, }); } export async function downloadVideoPipelineAction( slug: string, videoId: string, queueKey?: string, downloadFormat?: string, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); // This action fetches a full container. Its sibling redownloadToArchiveAction // has always preflighted the disk; this one never did. const lowDisk = await lowDiskError(paths, slug); if (lowDisk) return lowDisk; const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { return { ok: false, error: "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", }; } const settings = getSettings(); // Per-download override (from the redownload picker), falling back to the // channel default then the global default. An unrecognized value is ignored. const formatOverride: DownloadFormatPreset | undefined = isDownloadFormatPreset(downloadFormat) ? downloadFormat : undefined; return runManagedFunction({ kind: "download-one-pipeline", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { const task = makeTaskTracker(ctx, onLog).start({ id: videoId, label: videoId, kind: "download", }); try { await downloadOneManaged({ channelSlug: slug, channelConfig: r.config, paths, videoUrl: url, onLog: task.onLog, signal, cookiePolicy: resolveCookiePolicy(settings, r.config), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, downloadFormatPreset: resolveDownloadFormatPreset({ override: formatOverride, channel: r.config.downloadFormat, global: settings.downloadFormat, }), appendArchive: true, }); safeRevalidate([ `/channels/${slug}/videos/${videoId}`, `/channels/${slug}`, ]); } finally { task.end(); } }, }); } // "Refresh metadata": re-read this one video's metadata.info.json from its // source — no subtitles, no media — for a source that changed after the first // fetch (a livestream VOD gaining its processed formats and captions). The job // and every refusal are common/controller/refreshVideoMetadataJob.ts, so the // ops route and a Retry from /jobs answer exactly as this click does; what is // the editor's own is the two revalidations. export async function refreshVideoMetadataAction( slug: string, videoId: string, queueKey?: string, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; return runRefreshMetadataJob({ paths: getPaths(), slug, videoId, channelConfig: r.config, cookiePolicy: resolveCookiePolicy(getSettings(), r.config), queueKey, afterRun: () => { safeRevalidate([ `/channels/${slug}/videos/${videoId}`, `/channels/${slug}`, ]); }, }); } // Re-fetch an already-downloaded video purely to archive its SOURCE container, // without disturbing the existing transcript. Forces the persistence rule on // (keepSourceVideoOverride=true) so downloadOneManaged downloads the full video // and keeps the container (Phase 3 moves it to the saved store), and forces the // media download itself (forceMedia) on a youtube-handling channel, whose only // media pass is otherwise skipped when a transcript or captions exist; there, // with a transcript on disk, no audio is extracted — the container is the // point. Works on a video already in the archive — downloadOneManaged has no // archive prefilter, so it always re-downloads. `quality` is the control's // per-persist choice ("original" / "video_720"); anything else, or nothing, // inherits the channel's, else the global, source-video quality. export async function redownloadToArchiveAction( slug: string, videoId: string, queueKey?: string, quality?: string, ): Promise { return archiveSourceVideo(slug, videoId, queueKey, undefined, quality); } // The body of both "Persist source video" (the button) and the full-source // branch of the clip-window fetch (umtool asking for a whole recording rather // than a window). ONE function, because the two differ in exactly one field: // who asked. Factored rather than copied — a second copy is a second place for // the keepSourceVideoOverride / appendArchive pair to drift. async function archiveSourceVideo( slug: string, videoId: string, queueKey: string | undefined, persistOrigin?: SavedVideoOrigin, // The per-persist quality override, unvalidated: resolveSourceVideoQuality // drops anything that is not a quality, falling through to channel/global. qualityOverride?: unknown, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { return { ok: false, error: "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", }; } const err = await lowDiskError(paths, slug); if (err) return err; const settings = getSettings(); const persistFormatPreset = resolveSourceVideoQuality({ override: qualityOverride, channel: r.config.sourceVideoQuality, global: settings.sourceVideoQuality, }); return runManagedFunction({ kind: "redownload-archive", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { const task = makeTaskTracker(ctx, onLog).start({ id: videoId, label: videoId, kind: "download", }); try { if (persistOrigin) { onLog(`${requesterLine(persistOrigin)}\n`); } onLog( `Re-downloading ${videoId} to archive its source video (quality ${persistFormatPreset})…\n`, ); const record = await downloadOneManaged({ channelSlug: slug, channelConfig: r.config, paths, videoUrl: url, onLog: task.onLog, signal, cookiePolicy: resolveCookiePolicy(settings, r.config), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, keepSourceVideoOverride: true, // The operator (or the tool behind a whole-recording fetch) asked for // the SOURCE: a transcript or captions on disk are not a reason to // skip it. Without this a youtube-handling video with a transcript // got two subtitle passes and no file (release 10 slice N). forceMedia: true, persistOrigin, persistFormatPreset, }); safeRevalidate([ `/channels/${slug}/videos/${videoId}`, `/channels/${slug}`, ]); // THE JOB IS THE SOURCE, SO ITS STATUS IS THE SOURCE'S (release 11 // slice O3). A forced media pass that fails over a transcript leaves // the download `ok` — the transcript is fine — so without this the job // ended `done` with no file, and the whole-recording fetch's caller was // told only that it "finished but named no file". Failing the job puts // yt-dlp's line in the log tail the poll returns. const failure = sourceFetchFailure(record); if (failure) { throw new Error(`The source video was not downloaded: ${failure}`); } } finally { task.end(); } }, }); } // Preflight disk-space gate shared by every per-video action that can pull // bytes down. `latch: false` — the operator clicked this, so only the floor // applies and the shared hysteresis latch is left alone (see diskGate). async function lowDiskError( paths: Paths, slug: string, ): Promise<{ ok: false; error: string } | null> { // The channel's own volume. A relocated channel's bytes land on the platter // through the data/ symlink, so a full SSD is not a reason to refuse them. const gate = await diskGate(paths, getSettings(), { mode: "manual", dir: path.join(paths.channelsDir, slug, "data"), }); if (gate.ok) return null; return { ok: false, error: `Low disk space: ${formatBytes(gate.freeBytes)} free, ` + `${formatBytes(gate.thresholdBytes)} required. Free up space or ` + `lower the floor in Settings.`, }; } export async function whisperVideoAction( slug: string, videoId: string, queueKey?: string, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); const videoDir = videoDirOf(slug, videoId); return runManagedFunction({ kind: "whisper-video", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { const audioFormat = r.config.audioFormat ?? "mp3"; const entries = await readdir(videoDir).catch(() => [] as string[]); const hasAudio = entries.some(isRealAudioFile); if (!hasAudio) { // Only this branch writes bytes. Transcribing audio that is already on // disk produces a transcript.json measured in kilobytes, so a low disk // is no reason to refuse it — the gate belongs on the fetch, not on the // whole action. const gate = await diskGate(paths, getSettings(), { mode: "manual", dir: path.join(paths.channelsDir, slug, "data"), }); if (!gate.ok) { throw new Error( `Cannot download audio for ${videoId}: ${gate.message}. ` + `Free up space or lower the floor in Settings.`, ); } const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { throw new Error( "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", ); } onLog(`No audio on disk for ${videoId}; downloading…`); await runYtdlp({ channelSlug: slug, mode: "download-one-audio", channelConfig: r.config, paths, onLog, signal, singleVideoUrl: url, audioFormatOverride: audioFormat, }); } else { onLog(`Audio already on disk for ${videoId}; skipping download.`); } await transcribeWithWorker({ paths, videoDir, videoId, audioFilename: `audio.${audioFormat}`, tracker: makeTaskTracker(ctx, onLog), onLog, signal, }); safeRevalidate([ `/channels/${slug}/videos/${videoId}`, `/channels/${slug}`, ]); }, }); } // Fix a truncated transcript: the audio download silently stopped early, so the // audio on disk is itself truncated and re-running whisper on it would reproduce // the short transcript. Delete the truncated audio first, then re-download the // full audio and re-transcribe in one job. The new transcript overwrites the // old transcript.json (transcribeWithWorker), and normalize regenerates cues. export async function redownloadIncompleteTranscriptAction( slug: string, videoId: string, queueKey?: string, ): Promise { const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); return runManagedFunction({ kind: "whisper-video", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { await fixIncompleteTranscriptOne({ slug, videoId, config: r.config, paths, onLog, signal, tracker: makeTaskTracker(ctx, onLog), }); safeRevalidate([ `/channels/${slug}/videos/${videoId}`, `/channels/${slug}`, ]); }, }); } function safeJoinUnderDir( baseDir: string, filename: string, ): string | null { if (!filename || filename === "." || filename === "..") return null; if (filename.includes("\0")) return null; // Reject anything that has path separators or parent-dir traversal. if (filename.includes("/") || filename.includes("\\")) return null; const base = path.resolve(baseDir); const resolved = path.resolve(base, filename); if (path.dirname(resolved) !== base) return null; return resolved; } export async function deleteVideoFileAction( slug: string, videoId: string, filename: string, ): Promise<{ ok: true } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); const target = safeJoinUnderDir(videoDir, filename); if (!target) { return { ok: false, error: `Refusing to delete suspicious filename "${filename}"` }; } // A TIERED file's stat — a link (the `lstat`) with a tierable name — is // asked of the media tier's drive, through its watchdog (release 17); a // real file, `source-media.*` included, is on the corpus disk (review N7). // // A LEGACY channel (review R1) is asked nothing here: its whole `data/` is a // link onto the retired drive, so the `lstat` would reach it — refused while // that drive is known not to answer, and otherwise read directly as before // (it has no tiered links). const config = await readChannelConfig(getPaths(), slug); if (channelTextStall(config)) { return { ok: false, error: `${filename} not deleted: this channel's drive is not answering.`, }; } const linked = config?.dataDir?.trim() ? false : await lstat(target) .then((l) => l.isSymbolicLink()) .catch(() => false); const mediaDrive = linked && isTierable(path.basename(target)) ? config?.mediaDir?.trim() : undefined; let s; try { s = await (mediaDrive ? onDrive(mediaDrive, () => stat(target)) : stat(target)); } catch { // A TIERED FILE WHOSE DRIVE IS NOT THERE: removing the link alone would // orphan its bytes on that drive, so nothing is removed. if (linked && mediaDrive) { return { ok: false, error: `${filename} is on this channel's media drive, which is not mounted ` + `or not answering — nothing was deleted.`, }; } // A dangling link on an in-place `media/`: its bytes are already gone, so // the link is all there is to remove (review N8). if (linked) { await removeMediaFile(videoDir, path.basename(target)); revalidatePath(`/channels/${slug}/videos/${videoId}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } return { ok: false, error: `File not found: ${filename}` }; } if (!s.isFile()) { return { ok: false, error: `Not a regular file: ${filename}` }; } // Through its link when the file is tiered (release 17): the bytes on the // media tier go too, never orphaned behind a removed link. await removeMediaFile(videoDir, path.basename(target)); revalidatePath(`/channels/${slug}/videos/${videoId}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } // Promote a transcript..vtt track to the canonical transcript.en.vtt so // the index, snapshot, and viewer all treat it as the primary transcript. The // chosen file is copied (not moved) so the original language-coded track is kept // and the choice stays reversible/repeatable — delete transcript-pin.json to // fall back to the automatic pick, or pick a different track to switch again. // The pin is what makes the copy win: the caption-track rule otherwise ranks // transcript.en-orig.vtt above transcript.en.vtt (lib/videoStatus.ts). export async function setPrimaryTranscriptAction( slug: string, videoId: string, filename: string, ): Promise<{ ok: true } | { ok: false; error: string }> { if (!isTranscriptVtt(filename)) { return { ok: false, error: `Not a transcript VTT file: ${filename}`, }; } const videoDir = videoDirOf(slug, videoId); const source = safeJoinUnderDir(videoDir, filename); if (!source) { return { ok: false, error: `Refusing to read suspicious filename "${filename}"` }; } let raw: string; try { raw = await readFile(source, "utf8"); } catch { return { ok: false, error: `File not found: ${filename}` }; } if (filename !== VTT_FILENAME) { await writeFileAtomic(path.join(videoDir, VTT_FILENAME), raw); } await pinTranscript(videoDir, filename); // The video page reads the dir directly, so it reflects the new primary right // away. The channel list + diagnostics bucket read the cached snapshot and // refresh on the next snapshot regeneration (same as the other video actions). revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } // The video's transcript tracks, primary first (lib/captionTracks.ts): the // primary and every other English track whose words differ from it. Read-only — // what the page's transcript reader shows and switches between. Choosing a // track there changes nothing on disk; "Set as transcript" above is how the // primary changes. export async function readTranscriptTracksAction( slug: string, videoId: string, ): Promise<{ ok: true; tracks: AltTrack[] } | { ok: false; error: string }> { const read = await readVideoTracks(videoDirOf(slug, videoId)); if (!read) return { ok: false, error: "This video has no transcript to read." }; return { ok: true, tracks: read.tracks }; } // A refusal carries what was submitted (lib/formState.ts). export type DeleteDirActionResult = FormErrorState; // Validated, non-redirecting directory delete shared by the single-video form // action and the bulk action. Pure filesystem op — queues no job, so it never // triggers a transcode. export async function deleteOneVideoDir( slug: string, videoId: string, ): Promise<{ ok: true } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); // Belt-and-suspenders: ensure videoDir is inside the channel's data dir. const dataDir = path.resolve(getPaths().channelsDir, slug, "data"); const resolved = path.resolve(videoDir); if (path.dirname(resolved) !== dataDir) { return { ok: false, error: "Refusing to delete: video path resolved outside the data dir", }; } // The video's media tier first (release 17): every tiered link's bytes and // its `media//`, which the recursive rm of the text dir cannot reach. // Only while the channel's media is reachable, and through the watchdog: on // an unmounted drive the bytes would be orphaned, on a stalled one the rm // would block. Refused before anything is touched. const config = await readChannelConfig(getPaths(), slug); const media = await inspectChannelMedia(getPaths(), slug, config, { fresh: true }); if (media.status !== "ok" && media.status !== "in-place") { return { ok: false, error: `Refusing to delete ${videoId}: its media is not reachable — ` + `${media.detail ?? media.status}. Nothing has been touched.`, }; } const drive = config?.mediaDir?.trim(); try { await (drive ? onDrive(drive, () => removeVideoDirMedia(resolved)) : removeVideoDirMedia(resolved)); } catch (err) { return { ok: false, error: `Refusing to delete ${videoId}: ${(err as Error).message}. The text was not touched.`, }; } await rm(resolved, { recursive: true, force: true }); return { ok: true }; } // Validated removal of finalized audio files from one video dir, shared by the // bulk "Remove audio files" and "Remove wrong-format audio" actions. Keeps .part // partials, transcripts, and metadata. With wrongFormatOnly, only audio files // other than the channel's target format are removed. Pure filesystem op — never // queues a job, so it can't trigger a transcode. Bulk selection is explicit, so // this deliberately ignores the "do not clean" marker. export async function removeAudioFilesForVideo( slug: string, videoId: string, opts: { wrongFormatOnly?: boolean; targetAudioFile?: string } = {}, ): Promise<{ ok: true; removed: number } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); const dataDir = path.resolve(getPaths().channelsDir, slug, "data"); const resolved = path.resolve(videoDir); if (path.dirname(resolved) !== dataDir) { return { ok: false, error: "Refusing to remove: video path resolved outside the data dir", }; } const entries = await readdir(resolved).catch(() => [] as string[]); const toRemove = audioFilesToRemove(entries, { targetAudioFile: opts.targetAudioFile, wrongFormatOnly: opts.wrongFormatOnly, }); for (const name of toRemove) { await removeMediaFile(resolved, name); // derefs a tiered link (release 17) } return { ok: true, removed: toRemove.length }; } export async function deleteVideoDirAction( slug: string, videoId: string, _prev: DeleteDirActionResult, formData: FormData, ): Promise { const values = formValues(formData); const confirm = String(formData.get("confirmId") ?? "").trim(); if (confirm !== videoId) { return { error: `Type the video id "${videoId}" exactly to confirm deletion`, values, }; } const result = await deleteOneVideoDir(slug, videoId); if (!result.ok) return { error: result.error, values }; revalidatePath(`/channels/${slug}`); requestChannelSnapshot(getPaths(), slug); redirect(`/channels/${slug}`); } export async function markVideoUntranscribableAction( slug: string, videoId: string, ): Promise<{ ok: true } | { ok: false; error: string }> { const r = await loadConfigOrError(slug); if (!r.ok) return r; const videoDir = videoDirOf(slug, videoId); try { const s = await stat(videoDir); if (!s.isDirectory()) { return { ok: false, error: `Video directory not found: ${videoId}` }; } } catch { return { ok: false, error: `Video directory not found: ${videoId}` }; } const transcriptPath = path.join(videoDir, "transcript.json"); try { await stat(transcriptPath); return { ok: false, error: "transcript.json already exists; delete it first to remark.", }; } catch { // good — file does not exist } // Compact + "\n": exactly the literal `{"transcription":[]}\n` this wrote. await writeJsonAtomic(transcriptPath, { transcription: [] }, { indent: 0 }); const failureListFile = path.join( getPaths().channelsDir, slug, "failed-transcriptions", ); await pruneFailedTranscriptions(failureListFile, new Set([videoId])); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } export async function toggleDoNotCleanAction( slug: string, videoId: string, enabled: boolean, ): Promise<{ ok: true } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); try { const s = await stat(videoDir); if (!s.isDirectory()) { return { ok: false, error: `Video directory not found: ${videoId}` }; } } catch { return { ok: false, error: `Video directory not found: ${videoId}` }; } await setDoNotClean(videoDir, enabled); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } // The BULK form of the toggle above: set the do-not-clean marker on every video // of one channel whose title / description matches `pattern` (the download // filter's matcher and subject — see common/controller/keepVideosMatching.ts). // No UI calls it yet; `pnpm ops keep-videos` does. Same revalidation as the // toggle, per marked video, and ONE snapshot request for the whole batch — the // snapshot's cleanup buckets exclude kept ids, so it has to be re-derived, but // once, not once per video. export async function keepVideosAction(input: { slug: string; pattern: string; fields?: KeepVideosField[]; note?: string; dryRun?: boolean; }): Promise<{ ok: true; result: KeepVideosResult } | { ok: false; error: string }> { let result: KeepVideosResult; try { result = await keepVideosMatching({ paths: getPaths(), channelSlug: input.slug, pattern: input.pattern, fields: input.fields, note: input.note, dryRun: input.dryRun, }); } catch (e) { if (e instanceof KeepVideosError || e instanceof ChannelMediaUnreachableError) { return { ok: false, error: e.message }; } throw e; } if (result.marked > 0) { for (const m of result.matched) { if (m.marked) revalidatePath(`/channels/${input.slug}/videos/${m.id}`); } revalidatePath(`/channels/${input.slug}`); requestChannelSnapshot(getPaths(), input.slug); } return { ok: true, result }; } // Opt this video in/out of the truncated/incomplete-transcript check. When // enabled, the snapshot's incompleteTranscript + shortAudio buckets and the // video panel's "looks truncated" banner suppress this id (for videos that // legitimately have little speech). Mirrors toggleDoNotCleanAction. export async function toggleExcludeTruncatedCheckAction( slug: string, videoId: string, enabled: boolean, ): Promise<{ ok: true } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); try { const s = await stat(videoDir); if (!s.isDirectory()) { return { ok: false, error: `Video directory not found: ${videoId}` }; } } catch { return { ok: false, error: `Video directory not found: ${videoId}` }; } await setExcludedFromTruncatedCheck(videoDir, enabled); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); requestChannelSnapshot(getPaths(), slug); return { ok: true }; } // One video's curated tags. THE ONE WRITER, reached: this calls // applyTagAssignmentsAction, which is what the /tags preview rows, the bulk bar, // /api/ops/tag-videos and umtool all call — so a pin made here and a pin made by // an agent are the same write, differing only in the provenance recorded. // // The id passed in is the one the INDEX uses (loadVideoTags resolves it from // metadata, because a Rumble directory is named for the URL slug while the // record is keyed by the embed id). Nothing here re-derives it. export async function toggleVideoTagAction( slug: string, videoId: string, tag: string, op: "add" | "remove" | "suppress" | "unsuppress", // The DIRECTORY name, when the caller has it. Only used to revalidate the // page the operator is on; the write never touches it. dirId?: string, ): Promise<{ ok: true; changed: number } | { ok: false; error: string }> { const result = await applyTagAssignmentsAction({ op, tag, videos: [{ channelSlug: slug, id: videoId }], }); if (!result.ok) return result; // THE ROUTE SEGMENT IS THE DIRECTORY NAME, and `videoId` here is the RECORD // id — the two differ on Rumble, Odysee and Twitch, so revalidating // `/videos/` would refresh a path that does not exist while the // page the operator is looking at kept its cached panel. The caller passes // the directory name when it knows it; the list path covers both either way. revalidatePath(`/channels/${slug}/videos/${dirId ?? videoId}`); revalidatePath(`/channels/${slug}/videos`); return result; } // Reverse persistence: move this video's stored source container back into its // data dir and drop the pointer (Phase 5). The companion "persist" direction is // redownloadToArchiveAction, which re-fetches the container when it's not on // disk. Returns false-shaped error if there was nothing persisted to reverse. export async function unpersistVideoAction( slug: string, videoId: string, ): Promise<{ ok: true } | { ok: false; error: string }> { const videoDir = videoDirOf(slug, videoId); const reversed = await unpersistSavedVideo(videoDir); if (!reversed) { return { ok: false, error: "This video has no persisted source to restore." }; } revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); revalidatePath("/saved-videos"); return { ok: true }; } // --------------------------------------------------------------------------- // Sourcing media FOR ANOTHER TOOL, through this app's managed download path. // // The operator's rule is that no yt-dlp runs by hand. umtool's clip bench used // to fetch its own clip windows; it now asks here, so the fetch inherits the // channel's cookie policy and extra args, the per-platform 429 cooldown, the // low-disk gate and a job log — and the bytes land in the corpus where the next // tool (and the next build) can reuse them, with a note saying who asked. // --------------------------------------------------------------------------- // The one sentence a job log and a video page both show: who wanted this, for // what, and why. Built in one place so the two cannot disagree. function requesterLine(o: SavedVideoOrigin | FetchWindowProvenance): string { const what = [o.manifest, o.clipId].filter(Boolean).join("/"); const head = what ? `${o.requestedBy} · ${what}` : o.requestedBy; return o.reason ? `${head} — ${o.reason}` : head; } export type FetchMediaOutcome = // Already on disk. No job, no bytes, no politeness owed. `height` is how // tall the file on disk is, when that is known (a saved pointer's recorded // format, a window's probed stream) — a cached file was fetched for an // earlier ask, so it may be taller than this one's cap, and the caller is // the one who can tell whether that matters. | { ok: true; cached: true; file: string; from: number; to: number; bytes: number; provenance: ClipProvenance | SavedVideoOrigin | null; height?: number; } // Queued. `file` is where the window WILL be; for a full source it is not // knowable until the container's real extension is (null until then). | { ok: true; cached: false; jobId: string; file: string | null; from: number; to: number; // The job was ALREADY queued or running for this window (or a wider one // that covers it) — no second job was started (release 19, A5). `from` // and `to` are then that job's window, which `file` will hold. existing?: true; } // `stream: true` was asked for: the caller reads the job's log itself, so the // StreamActionResult is handed over rather than cancelled. | { ok: true; cached: false; started: StreamActionResult & { ok: true }; from: number; to: number; } // `status` is the HTTP status the route should answer with, so the mapping // from a refusal to a code lives with the refusal rather than in a switch // over error strings. | { ok: false; status: number; error: string; cooldownMs?: number; platform?: string; }; // A started job's stream has no reader here: the caller polls the job instead. // Cancelling the stream is the documented way to say so — it marks the // controller closed and leaves the job running into its on-disk log. function detach(res: StreamActionResult & { ok: true }): void { void res.stream.cancel().catch(() => {}); } // Fetch ONE window of a video's source media into data//clips/. // // TWO CALLERS, TWO SHAPES. The HTTP route wants a JSON-able outcome and polls // the job; Retry (jobReplayRegistry) wants the StreamActionResult every other // action returns, so the /jobs page can stream the log. So the work is one // function and the two entry points differ only in what they do with the // started job — never in which cache it consulted or which refusal it hit. export async function fetchWindowAction(req: { slug: string; videoId: string; // The page URL, when the caller has it. Otherwise it is resolved from the // video's own metadata or the stored playlist, exactly as the archive // re-download does. webpageUrl?: string; from: number; to: number; provenance: FetchWindowProvenance; queueKey?: string; // The source height to cap the fetch at (clipFormatSelector's), absent for // the default (DEFAULT_CLIP_MAX_HEIGHT). Checked again here for the same // reason the window is: a replayed spec is a file on disk. maxHeight?: number; // Hand back the StreamActionResult instead of detaching it. Set by Retry, // which renders the log; the HTTP route leaves it off and polls. stream?: boolean; }): Promise { const { slug, videoId, from, to } = req; // THE CAP IS CHECKED HERE TOO, not only at the HTTP door. Retry re-runs from // a stored JobSpec, and a spec is a JSON file on disk — a hand-edited one // must not be able to ask for an hour of source through a path the route // already refused. if ( !Number.isFinite(from) || !Number.isFinite(to) || from < 0 || from >= to || to - from > MAX_CLIP_WINDOW_SECONDS ) { return { ok: false, status: 400, error: `${from}–${to} is not a fetchable window ` + `(at most ${MAX_CLIP_WINDOW_SECONDS}s, from < to, from >= 0).`, }; } if (req.maxHeight !== undefined && !isFetchMaxHeight(req.maxHeight)) { return { ok: false, status: 400, error: `maxHeight ${req.maxHeight} is not a source height to cap a fetch at.`, }; } const r = await loadConfigOrError(slug); if (!r.ok) return { ok: false, status: 404, error: r.error }; const paths = getPaths(); const videoDir = videoDirOf(slug, videoId); // ASK THE CACHE FIRST, and accept a WIDER file: a report cites the same // stream more than once, so a generous fetch for one clip must serve the // neighbour it already covers rather than being downloaded again. const hit = await findContainingClipWindow(videoDir, from, to); if (hit) { // The window's own height, read off its header: a window holds seconds, // so the probe is quick, and it is the only record of how tall a window // fetched under an earlier cap is. const height = await probeVideoHeight({ ffprobeBin: paths.ffprobeBin, file: hit.path, }); return { ok: true, cached: true, file: hit.path, from: hit.from, to: hit.to, bytes: hit.bytes, provenance: hit.provenance, ...(height !== null ? { height } : {}), }; } // ...THEN THE JOBS ALREADY FETCHING IT (release 19, A5): an ask repeated // while its window is still queued or downloading is answered with that // job, not a second one for the same seconds (jobs/windowJobs.ts). Retry // (`stream: true`) is an explicit re-run and is not deduplicated. if (!req.stream) { const inFlight = windowInFlight(getRegistry().list(), { slug, videoId, from, to, maxHeight: req.maxHeight, }); if (inFlight) { return { ok: true, cached: false, jobId: inFlight.jobId, file: clipWindowPath(videoDir, inFlight.from, inFlight.to), from: inFlight.from, to: inFlight.to, existing: true, }; } } // ...THEN THE SAVED CONTAINER (release 21 D2), before any network: a video // whose whole source is in the saved-video store is CUT here, not fetched, // and the answer is a cached window like the one above // (lib/savedVideoWindow-server.ts). No // pointer, or a container that ends before the window, goes on to the // network as before; a pointer whose container cannot be read is REFUSED — // falling through would spend the source's patience (or, for a deleted // channel, fail) on seconds that are on a drive somebody can plug in. const saved = await savedWindowSource({ paths, slug, videoDir, from, to }); if (saved.kind === "unreadable") { return { ok: false, status: 503, error: saved.error }; } if (saved.kind === "covers") { // The window lands in data//clips/, the channel's TEXT tier: the guard a // fetch-window job's `needsText` would have asked, asked here because no // job runs. try { await assertChannelTextReadable(paths, slug); } catch (e) { return { ok: false, status: 503, error: (e as Error).message }; } let cut; try { cut = await cutSavedVideoWindow({ paths, videoDir, source: saved, from, to, provenance: req.provenance, }); } catch (e) { return { ok: false, status: 500, error: (e as Error).message }; } const height = await probeVideoHeight({ ffprobeBin: paths.ffprobeBin, file: cut.path, }); safeRevalidate([`/channels/${slug}/videos/${videoId}`]); return { ok: true, cached: true, file: cut.path, from: cut.from, to: cut.to, bytes: cut.bytes, provenance: cut.provenance, ...(height !== null ? { height } : {}), }; } const url = req.webpageUrl?.trim() || (await findVideoSourceUrl(paths, slug, videoId, r.config)); if (!url) { return { ok: false, status: 404, error: "Could not determine the video URL: no metadata.info.json and the " + "playlist does not contain a matching entry.", }; } // The same per-platform cooldown a clicked Sync respects. A window is a // handful of seconds, but a 429 is a fact about the SOURCE, not about the // size of the request. const platform = detectPlatform(url) ?? "unknown"; // A held platform's refusal names the hold and the next probe (release 17). const held = await heldPlatformRefusal(platform, "This fetch", paths); const cooldownMs = await platformCooldownRemainingMs(platform, paths); if (held && cooldownMs > 0) { return { ok: false, status: 409, cooldownMs, platform, error: held }; } if (cooldownMs > 0) { return { ok: false, status: 409, cooldownMs, platform, error: `${platform} is in a rate-limit cooldown ` + `(${Math.ceil(cooldownMs / 1000)}s remaining).`, }; } const disk = await lowDiskError(paths, slug); if (disk) return { ok: false, status: 507, error: disk.error }; const settings = getSettings(); // NOT GATED BY THE DOWNLOAD PAUSE. The pause exists to stop the lanes — the // auto-download runner and a channel-wide sync — from spending the source's // patience on their own schedule. This is an operator asking, by hand, // through a second tool, for one window they are about to watch; refusing it // would make "pause downloads" mean "stop working", which is not what the // control says and not why it gets used. const res = await runManagedFunction({ kind: "fetch-window", // The platform's CLIP queue, not its download queue: a window must not // wait behind a multi-hour persist or sync (lib/queueKeys.ts, // clipWindowQueueKey). An explicit override still wins. queueKey: resolveQueueKey(clipWindowQueueKey(downloadQueueKey(r.config)), req.queueKey), paths, channelSlug: slug, videoId, spec: { kind: "fetch-window", slug, params: { videoId, from, to, webpageUrl: url, queueKey: req.queueKey, ...(req.maxHeight !== undefined ? { maxHeight: req.maxHeight } : {}), ...req.provenance, }, }, fn: async (onLog, signal) => { onLog(`${requesterLine(req.provenance)}\n`); await fetchWindowManaged({ channelSlug: slug, channelConfig: r.config, paths, videoDir, videoId, videoUrl: url, // The channel root, as every other yt-dlp invocation here runs. cwd: path.join(paths.channelsDir, slug), from, to, provenance: req.provenance, maxHeight: req.maxHeight, cookiePolicy: resolveCookiePolicy(settings, r.config), onLog, signal, onPlatformBackoff: () => recordDownloadBackoff(platform, paths), }); safeRevalidate([`/channels/${slug}/videos/${videoId}`]); }, }); if (!res.ok) { // An unreachable-media refusal (a relocated channel whose drive is not // mounted, or one mid-relocation) arrives here verbatim from // runManagedFunction's guard, and is passed on verbatim: the caller's // operator is the person who can plug the drive in. return { ok: false, status: 503, error: res.error }; } if (req.stream) return { ok: true, cached: false, started: res, from, to }; detach(res); return { ok: true, cached: false, jobId: res.jobId, file: clipWindowPath(videoDir, from, to), from, to, }; } // Retry, from the /jobs page. A replayed fetch re-derives from the CURRENT // disk: a window somebody has since fetched (or that a wider one now covers) // replays as a neutral "already there" rather than downloading it twice. export async function replayFetchWindowAction(req: { slug: string; videoId: string; webpageUrl?: string; from: number; to: number; provenance: FetchWindowProvenance; queueKey?: string; maxHeight?: number; }): Promise { const r = await fetchWindowAction({ ...req, stream: true }); if (!r.ok) return { ok: false, error: r.error, info: r.status === 409 }; if ("started" in r) return r.started; if (r.cached) { return { ok: false, info: true, error: `${r.from.toFixed(2)}–${r.to.toFixed(2)} is already fetched ` + `(${r.file}); nothing to download.`, }; } // Unreachable: a non-cached result with `stream: true` always carries // `started`. Kept total rather than asserted. return { ok: false, info: true, error: "Nothing to do." }; } // The WHOLE recording, when a window will not do (a tool that needs to re-cut // freely, or a source whose windows would tile the entire runtime). It reuses // the saved-video store rather than clips/: that is where big containers // already live, with a retention rule that leaves an explicitly-requested one // alone. export async function fetchFullSourceAction(req: { slug: string; videoId: string; provenance: FetchWindowProvenance; queueKey?: string; // Per-fetch quality; absent = the channel's, else the global setting. quality?: SourceVideoQuality; }): Promise { const { slug, videoId } = req; const r = await loadConfigOrError(slug); if (!r.ok) return { ok: false, status: 404, error: r.error }; const paths = getPaths(); const videoDir = videoDirOf(slug, videoId); const pointer = await loadSavedVideo(videoDir); if (pointer) { return { ok: true, cached: true, file: savedVideoPath(pointer), from: 0, to: 0, bytes: pointer.bytes, provenance: pointer.origin ?? null, ...(pointer.format?.height !== undefined ? { height: pointer.format.height } : {}), }; } const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { return { ok: false, status: 404, error: "Could not determine the video URL: no metadata.info.json and the " + "playlist does not contain a matching entry.", }; } const platform = detectPlatform(url) ?? "unknown"; // A held platform's refusal names the hold and the next probe (release 17). const held = await heldPlatformRefusal(platform, "This fetch", paths); const cooldownMs = await platformCooldownRemainingMs(platform, paths); if (held && cooldownMs > 0) { return { ok: false, status: 409, cooldownMs, platform, error: held }; } if (cooldownMs > 0) { return { ok: false, status: 409, cooldownMs, platform, error: `${platform} is in a rate-limit cooldown ` + `(${Math.ceil(cooldownMs / 1000)}s remaining).`, }; } const origin: SavedVideoOrigin = { requestedBy: req.provenance.requestedBy, ...(req.provenance.manifest ? { manifest: req.provenance.manifest } : {}), ...(req.provenance.clipId ? { clipId: req.provenance.clipId } : {}), ...(req.provenance.reason ? { reason: req.provenance.reason } : {}), requestedAt: req.provenance.requestedAt ?? new Date().toISOString(), }; const res = await archiveSourceVideo( slug, videoId, req.queueKey, origin, req.quality, ); if (!res.ok) { // archiveSourceVideo's own low-disk refusal is the only 507-shaped one it // returns; everything else is a lookup failure or an unreachable drive. const status = res.error.startsWith("Low disk space") ? 507 : 503; return { ok: false, status, error: res.error }; } detach(res); return { ok: true, cached: false, jobId: res.jobId, // The container's extension is decided by the format yt-dlp picks, so the // path is not knowable until the job has finished. The poll endpoint reads // it off the saved-video pointer. file: null, from: 0, to: 0, }; }