Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a9edbf4bce84592dc90d86e7ecfc4914f3ce9936
parent 7614b292e5e274558b528443581e94c8addf8033
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 30 Jun 2026 10:59:32 -0400

Merge feat/download-format-and-duration-guard: download-time short-audio guard + selectable/per-source download format (Odysee -> original)

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/autoRunner.ts | 18+++++++++++++++---
Mcommon/controller/channelSnapshot.ts | 20++++++++++++++++++++
Mcommon/jobs/jobSpec.ts | 4+++-
Mcommon/lib/channelConfig.ts | 13+++++++++++++
Mcommon/lib/downloadOutcome.ts | 15+++++++++++++++
Mcommon/lib/paths.ts | 5+++++
Mcommon/lib/settings.ts | 16++++++++++++++++
Mcommon/lib/transcriptCoverage.ts | 29+++++++++++++++++++++++++++++
Mcommon/lib/transcripts-server.ts | 28+++++++++++++++++++---------
Acommon/ytdlp/downloadFormat.ts | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/downloadOneManaged.ts | 145++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Acommon/ytdlp/ffprobeDuration.ts | 48++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 32+++++++++++++++++++++++++++++++-
Meditor/CHANGELOG.md | 1+
Meditor/app/actionable/components/InlineActionButton.tsx | 6++++++
Meditor/app/actionable/lib/loadActionable.ts | 12++++++++++++
Meditor/app/actionable/page.tsx | 28++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/bulkVideoActions.ts | 12++++++++++++
Meditor/app/channels/[slug]/components/VideoListPane.tsx | 33++++++++++++++++++++++++++++++++-
Meditor/app/channels/[slug]/incompleteTranscriptActions.ts | 65+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/lib/fixIncompleteTranscript.ts | 14++++++++++++++
Meditor/app/channels/[slug]/lib/stageStatus.ts | 1+
Meditor/app/channels/[slug]/lib/videoRows.ts | 9+++++++++
Meditor/app/channels/[slug]/lib/videoRowsServer.ts | 2++
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 106++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 15+++++++++++++++
Meditor/app/channels/components/ChannelForm.tsx | 25+++++++++++++++++++++++++
Meditor/app/channels/components/parseChannelForm.ts | 6++++++
Meditor/app/jobs/jobReplayRegistry.ts | 13++++++++++---
Meditor/app/settings/actions.ts | 6++++++
Meditor/app/settings/components/SettingsForm.tsx | 26++++++++++++++++++++++++++
Aeditor/e2e/download-format-guard.spec.ts | 191+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/fixtures/bin/fake-ffprobe.mjs | 32++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 33++++++++++++++++++++++++++-------
Meditor/package.json | 4++--
35 files changed, 1044 insertions(+), 37 deletions(-)

diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -39,6 +39,7 @@ import { readChannelSnapshot } from "./channelSnapshot"; import { transcribeOneFromQueue } from "./transcribeOneFromQueue"; import { findVideoSourceUrl } from "./undownloadedVideos"; import { downloadOneManaged } from "../ytdlp/downloadOneManaged"; +import { resolveDownloadFormatPreset } from "../ytdlp/downloadFormat"; // The automatic priority-queue runners. Each kind (auto-transcribe / // auto-download) is ONE long-lived registry job (queueKey "" so it runs in @@ -624,17 +625,28 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, + downloadFormatPreset: resolveDownloadFormatPreset({ + channel: config.downloadFormat, + global: settings.downloadFormat, + }), appendArchive: true, }); unitStatus = record.status; unitFailureClass = record.failureClass; // Land a failed download as a failed JOB (red row) rather than a silent - // "done"; the runner reads the captured status/class regardless. + // "done"; the runner reads the captured status/class regardless. A + // short-audio download kept its file (so it won't be re-queued) but + // produced no usable audio, so surface it as failed too. if ( record.status === "failed" || - record.status === "failed-corrupt-source" + record.status === "failed-corrupt-source" || + record.status === "failed-short-audio" ) { - throw new Error(`download failed (${record.failureClass ?? "unknown"})`); + throw new Error( + record.status === "failed-short-audio" + ? "download produced truncated audio (short-audio)" + : `download failed (${record.failureClass ?? "unknown"})`, + ); } } finally { task.end(); diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -96,6 +96,15 @@ export type ChannelSnapshot = { // the user can re-download/re-transcribe. Optional: older snapshots lack it; // readers must default to []. incompleteTranscript: string[]; + // Videos whose download COMPLETED (yt-dlp exit 0) but whose audio is far + // shorter than the metadata duration — the source served a truncated stream + // (download-outcome.json "failed-short-audio"). Caught at download time by + // the duration guard BEFORE transcription. The short file is KEPT on disk + // (so it isn't re-downloaded into a loop) and excluded from + // downloadedNoTranscript (so it isn't auto-transcribed). Surfaced so the + // user can re-download with a different format (e.g. Original). Optional: + // older snapshots lack it; readers must default to []. + shortAudio: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; @@ -362,6 +371,7 @@ export async function generateChannelSnapshot( const nonStandardVtt: string[] = []; const skippedByFilter: string[] = []; const incompleteTranscript: string[] = []; + const shortAudio: string[] = []; let transcribedWithAudioBytes = 0; let multipleAudioFormatsBytes = 0; let foreignAudioBytes = 0; @@ -380,6 +390,15 @@ export async function generateChannelSnapshot( corruptFullSource.push(id); continue; } + // A download whose audio was far shorter than the video (truncated source). + // Like corrupt-full-source: the stub is KEPT on disk but is NOT usable audio, + // so short-circuit it out of downloadedNoTranscript (no auto-transcribe) and + // the wrong-format cleanup (don't delete the kept file). If the user later + // transcribes it anyway, it falls through to the incompleteTranscript path. + if (outcome?.status === "failed-short-audio" && !isVideoTranscribed(files)) { + shortAudio.push(id); + continue; + } if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); // A transcribed video whose cues stop far short of its duration — the audio // download truncated silently. Threshold lives in transcriptCoverage. @@ -556,6 +575,7 @@ export async function generateChannelSnapshot( nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), incompleteTranscript: incompleteTranscript.sort(), + shortAudio: shortAudio.sort(), }, undownloadedIds, excludedFromDownload, diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts @@ -19,7 +19,8 @@ export type ReplayBucket = | "partialDownloads" | "noTranscript" | "downloadedNoTranscript" - | "incompleteTranscript"; + | "incompleteTranscript" + | "shortAudio"; export type JobSpec = { kind: string; @@ -38,6 +39,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([ "noTranscript", "downloadedNoTranscript", "incompleteTranscript", + "shortAudio", ]); // Defensive parse for a spec read back from JSON (a sidecar or the bookmarks diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -1,4 +1,8 @@ import { PLATFORM_VALUES, type Platform } from "./platform"; +import { + isDownloadFormatPreset, + type DownloadFormatPreset, +} from "../ytdlp/downloadFormat"; export type ChannelHandling = "youtube" | "transcribe"; @@ -37,6 +41,12 @@ export type ChannelConfig = { name?: string; url?: string; audioFormat?: AudioFormat; + // Per-channel override for the yt-dlp `-f` download format (see + // common/ytdlp/downloadFormat.ts). Omitted -> inherit the global default + // (SiteSettings.downloadFormat), which itself falls back to the per-source + // "auto" selector. Lets a channel whose source only serves full-length audio + // in its `original` format (e.g. Odysee) force it regardless of the global. + downloadFormat?: DownloadFormatPreset; keepSourceVideo?: boolean; // Keep-latest retention/persistence window. The newest N videos (by upload // date) are protected from the Clean-audio sweep AND have their source video @@ -149,6 +159,9 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null { ) { config.audioFormat = r.audioFormat; } + if (isDownloadFormatPreset(r.downloadFormat)) { + config.downloadFormat = r.downloadFormat; + } if (typeof r.keepSourceVideo === "boolean") { config.keepSourceVideo = r.keepSourceVideo; } diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts @@ -17,6 +17,13 @@ export type DownloadOutcomeStatus = // downloaded container is KEPT on disk for inspection. Distinct from // failed-corrupt-source, which means the download never finished. | "corrupt-full-source" + // A download that COMPLETED (yt-dlp exit 0) but whose audio is far shorter + // than the video's metadata duration — the source served a truncated stream + // (e.g. a CDN-truncated HLS rung). Caught by the download-time duration guard + // BEFORE transcription, so whisper never runs on the stub. The short audio is + // KEPT on disk (so it isn't silently re-downloaded into a loop) and surfaced + // for a manual re-download with a different format. See `shortAudio` below. + | "failed-short-audio" // The app-level filter pass (e.g. skip-live) declined to download this video. // Not a failure and not archived — the next sync/download-missing retries it // once the filter no longer matches (e.g. a live stream becomes a VOD). @@ -30,6 +37,7 @@ export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus "failed", "failed-corrupt-source", "corrupt-full-source", + "failed-short-audio", "skipped-filtered", ]; @@ -78,6 +86,13 @@ export type DownloadOutcomeRecord = { // on success / skipped-filtered. failureClass?: DownloadFailureClass; fellBackToTranscribe?: boolean; + // Set when status is "failed-short-audio": the measured shortfall, so the UI + // can explain it (e.g. "4 min of 114 min") without re-probing. + shortAudio?: { + audioDurationSec: number; + expectedDurationSec: number; + coverage: number; + }; // Set when status is "skipped-filtered": which app-level filter declined the // download and why. Recorded so the UI/log can explain the skip. filter?: { name: string; reason: string }; diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -79,6 +79,10 @@ export type Paths = { whisperBin: string; whisperModel: string; ffmpegBin: string; + // ffprobe binary, used to measure a downloaded audio file's actual duration + // for the download-time short-audio guard (common/ytdlp/ffprobeDuration.ts). + // Ships alongside ffmpeg. + ffprobeBin: string; // rsync binary used to mirror the saved-video store to a backup destination // (Phase 4 of the video-persistence feature). See // common/controller/backupSavedVideos.ts. @@ -158,6 +162,7 @@ export function getPaths(): Paths { "ggml-base.en.bin", ), ffmpegBin: process.env.FFMPEG_BIN ?? "ffmpeg", + ffprobeBin: process.env.FFPROBE_BIN ?? "ffprobe", rsyncBin: process.env.RSYNC_BIN ?? "rsync", parakeetBin: process.env.PARAKEET_STITCH_BIN ?? diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -3,6 +3,10 @@ import path from "node:path"; import { getPaths } from "./paths"; import { SYNC_INTERVAL_MAX_MINUTES } from "./channelConfig"; import { + isDownloadFormatPreset, + type DownloadFormatPreset, +} from "../ytdlp/downloadFormat"; +import { type AppInstanceConfig, DEFAULT_TRANSCRIBE_ARGS, DEFAULT_TRANSCRIPTION_APP_ID, @@ -67,6 +71,11 @@ export type SiteSettings = { // within one invocation, so without this the managed loop hammers the // source IP back-to-back. 0 disables. Per-channel override available. sleepBetweenDownloadsSeconds: number; + // Default yt-dlp `-f` download format for every channel that doesn't set its + // own (ChannelConfig.downloadFormat). "auto" picks per-source: `original` for + // Odysee (whose HLS rungs are CDN-truncated), `bestaudio/worst` elsewhere. See + // common/ytdlp/downloadFormat.ts. + downloadFormat: DownloadFormatPreset; // Minimum free disk space (GB) required on the transcripts data directory for // downloads to run. When free space is below this floor, a download job is // prevented from starting and a running batch stops launching new videos @@ -431,6 +440,7 @@ function defaults(): SiteSettings { workers: [], cookiesFromBrowser: "", sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, + downloadFormat: "auto", minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT, parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback: false, @@ -585,6 +595,9 @@ export function getSettings(): SiteSettings { merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds( merged.sleepBetweenDownloadsSeconds, ); + if (!isDownloadFormatPreset(merged.downloadFormat)) { + merged.downloadFormat = "auto"; + } merged.minFreeDiskGB = clampMinFreeDiskGB(merged.minFreeDiskGB); merged.parallelTranscriptions = clampParallelTranscriptions( merged.parallelTranscriptions, @@ -761,6 +774,9 @@ export async function writeSettings(next: SiteSettings): Promise<void> { sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds( next.sleepBetweenDownloadsSeconds, ), + downloadFormat: isDownloadFormatPreset(next.downloadFormat) + ? next.downloadFormat + : "auto", minFreeDiskGB: clampMinFreeDiskGB(next.minFreeDiskGB), parallelTranscriptions: clampParallelTranscriptions( next.parallelTranscriptions, diff --git a/common/lib/transcriptCoverage.ts b/common/lib/transcriptCoverage.ts @@ -38,6 +38,35 @@ export function transcriptCoverage( return { lastCueEnd, duration, coverage }; } +// The download-time analogue of isIncompleteTranscript: compare the DOWNLOADED +// audio's actual duration to the video's metadata duration. A large shortfall +// means the source served a truncated stream (e.g. a CDN-truncated HLS rung), +// so the audio is short before whisper ever runs. Shares the same thresholds so +// the two guards agree. Returns false (no judgement) for livestreams (unreliable +// durations), short videos, or when either duration is unusable. +export function isShortAudio( + audioDurationSec: number | null, + metaDurationSec: number | null | undefined, + opts?: { isLivestream?: boolean }, +): boolean { + if (opts?.isLivestream) return false; + if ( + typeof metaDurationSec !== "number" || + !Number.isFinite(metaDurationSec) || + metaDurationSec < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC + ) { + return false; + } + if ( + audioDurationSec === null || + !Number.isFinite(audioDurationSec) || + audioDurationSec <= 0 + ) { + return false; + } + return audioDurationSec / metaDurationSec < INCOMPLETE_TRANSCRIPT_MAX_COVERAGE; +} + export function isIncompleteTranscript( cov: TranscriptCoverage, opts?: { isLivestream?: boolean }, diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -59,7 +59,7 @@ export function loadRawMetadataFromDir( return loadRawMetadata(path.join(videoDir, "metadata.info.json")); } -function detectPlatform(meta: RawMetadata): Platform { +export function platformFromMetadata(meta: RawMetadata): Platform { const key = meta.extractor_key ?? meta.extractor ?? ""; if (/^rumble/i.test(key)) return "rumble"; if (/^lbry/i.test(key)) return "odysee"; @@ -67,20 +67,30 @@ function detectPlatform(meta: RawMetadata): Platform { return "youtube"; } +// The broad "is this a livestream (or stream VOD/upcoming)" notion used by the +// coverage detector and the download-time duration guard, so both skip the same +// content (stream captures have unreliable metadata durations). Distinct from +// the narrower download-filter `skipLive`, which intentionally lets finished +// stream VODs through. +export function isLivestreamMetadata(meta: RawMetadata): boolean { + const liveStatus = meta.live_status ?? ""; + return ( + meta.was_live === true || + meta.is_live === true || + liveStatus === "was_live" || + liveStatus === "is_live" || + liveStatus === "is_upcoming" + ); +} + export function summarize( channelSlug: string, videoDir: string, meta: RawMetadata, configName?: string, ): TranscriptSummary { - const liveStatus = meta.live_status ?? ""; - const isLivestream = - meta.was_live === true || - meta.is_live === true || - liveStatus === "was_live" || - liveStatus === "is_live" || - liveStatus === "is_upcoming"; - const platform = detectPlatform(meta); + const isLivestream = isLivestreamMetadata(meta); + const platform = platformFromMetadata(meta); const id = platform === "odysee" ? (meta.webpage_url_basename ?? meta.id ?? videoDir) diff --git a/common/ytdlp/downloadFormat.ts b/common/ytdlp/downloadFormat.ts @@ -0,0 +1,68 @@ +import type { Platform } from "../lib/platform"; + +// The user-selectable yt-dlp `-f` download format, as a small preset enum (kept +// an enum rather than a free-form `-f` string so it validates like audioFormat). +// "auto" is platform-aware: Odysee/LBRY only serves the full-length audio in its +// `original` format (every HLS rung is CDN-truncated to a few minutes), so auto +// prefers `original` there and the historical `bestaudio/worst` everywhere else. +export type DownloadFormatPreset = + | "auto" + | "original" + | "bestaudio" + | "bestvideo_audio"; + +export const DOWNLOAD_FORMAT_PRESETS: ReadonlyArray<DownloadFormatPreset> = [ + "auto", + "original", + "bestaudio", + "bestvideo_audio", +]; + +export function isDownloadFormatPreset( + value: unknown, +): value is DownloadFormatPreset { + return ( + typeof value === "string" && + (DOWNLOAD_FORMAT_PRESETS as ReadonlyArray<string>).includes(value) + ); +} + +// Human labels for the UI selects (global / channel / per-video). +export const DOWNLOAD_FORMAT_LABELS: Record<DownloadFormatPreset, string> = { + auto: "Auto (per-source)", + original: "Original (full file)", + bestaudio: "Best audio only", + bestvideo_audio: "Best video + audio", +}; + +// Resolve a preset into the concrete yt-dlp `-f` selector string. The platform +// only matters for "auto"; explicit presets apply literally regardless of +// source (so an explicit choice always wins over the per-source default). +export function resolveDownloadFormatSelector( + preset: DownloadFormatPreset, + platform: Platform | null | undefined, +): string { + switch (preset) { + case "original": + return "original/bestaudio/worst"; + case "bestaudio": + return "bestaudio/worst"; + case "bestvideo_audio": + return "bestvideo*+bestaudio/best"; + case "auto": + default: + return platform === "odysee" + ? "original/bestaudio/worst" + : "bestaudio/worst"; + } +} + +// The override chain mirrors audioFormat: per-run override beats the per-channel +// default beats the global default; "auto" is the baseline when nothing is set. +export function resolveDownloadFormatPreset(opts: { + override?: DownloadFormatPreset | null; + channel?: DownloadFormatPreset | null; + global?: DownloadFormatPreset | null; +}): DownloadFormatPreset { + return opts.override ?? opts.channel ?? opts.global ?? "auto"; +} diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -33,12 +33,27 @@ import { } from "../lib/downloadOutcome"; import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { recordAvailability } from "../lib/availability-server"; -import { loadRawMetadata } from "../lib/transcripts-server"; +import { + loadRawMetadata, + loadRawMetadataFromDir, + platformFromMetadata, + isLivestreamMetadata, +} from "../lib/transcripts-server"; import { evaluateDownloadFilters } from "../lib/downloadFilters"; +import { detectPlatform, type Platform } from "../lib/platform"; +import { probeMediaDurationSec } from "./ffprobeDuration"; +import { + isShortAudio, + INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC, +} from "../lib/transcriptCoverage"; import type { Paths } from "../lib/paths"; import { transcribeWithWorker } from "../controller/transcribeOne"; import { extractVideoId, outputArgsForUrl } from "./runYtdlp"; import { runAudioCheckedYtdlp } from "./audioCheckedDownload"; +import { + resolveDownloadFormatSelector, + type DownloadFormatPreset, +} from "./downloadFormat"; import { DOWNLOAD_PROGRESS_TEMPLATE } from "../jobs/progressParsers"; const STDERR_TAIL_BYTES = 64 * 1024; @@ -95,6 +110,11 @@ export type ManagedDownloadOpts = { extractImmediately?: boolean; // Per-run override of channelConfig.audioFormat for the extracted audio. audioFormatOverride?: AudioFormat; + // Resolved yt-dlp `-f` download-format preset (override > channel > global), + // already collapsed by the playlist-level caller. Expanded into a concrete + // selector here, where the per-video source platform is known (so "auto" + // picks `original` for Odysee). Defaults to "auto" when omitted. + downloadFormatPreset?: DownloadFormatPreset; }; // When `reuseInfoJson` is true, the real download reuses the metadata the @@ -126,15 +146,18 @@ function audioFormatSelectionArgs( plan: PersistenceDecision, fmt: AudioFormat, config: ChannelConfig, + selector: string, ): string[] { if (plan.extractionMode === "ytdlp") { - const args = ["-f", "bestaudio/worst", "-x", "--audio-format", fmt]; + const args = ["-f", selector, "-x", "--audio-format", fmt]; if (config.keepSourceVideo) args.push("-k"); return args; } + // Persisting in app mode keeps the full source container (always a complete + // video), so the truncation-prone audio-only selector doesn't apply. return plan.persist ? ["-f", "bestvideo*+bestaudio/best"] - : ["-f", "bestaudio/worst"]; + : ["-f", selector]; } // The media output + info-json + format args for a transcribe-handling download. @@ -147,12 +170,13 @@ function transcribeMediaArgs( plan: PersistenceDecision, fmt: AudioFormat, reuseInfoJson: boolean, + selector: string, ): string[] { const mediaName = plan.extractionMode === "app" ? "source-media" : "audio"; return [ ...outputArgsForUrl(url, { mediaName }), ...(reuseInfoJson ? [] : ["--write-info-json"]), - ...audioFormatSelectionArgs(plan, fmt, config), + ...audioFormatSelectionArgs(plan, fmt, config, selector), ]; } @@ -241,11 +265,14 @@ function sourceArgs(url: string, loadInfoJson: string | null): string[] { // Audio-checked mode: we own the audio extraction, so yt-dlp must NOT run // the ExtractAudio postprocessor. We always pass `-c` so resume picks up // where the last validated snapshot left off. -function transcribeHandlingArgsForAudioCheck(_config: ChannelConfig): string[] { +function transcribeHandlingArgsForAudioCheck( + _config: ChannelConfig, + selector: string, +): string[] { return [ "--write-info-json", "-f", - "bestaudio/worst", + selector, "-c", ]; } @@ -440,6 +467,9 @@ async function runManagedDownload( const attempts: DownloadAttempt[] = []; let status: DownloadOutcomeStatus = "failed"; let fellBackToTranscribe = false; + // Set when the download-time duration guard trips: the measured shortfall, + // recorded on the outcome so the UI can explain it without re-probing. + let shortAudioInfo: NonNullable<DownloadOutcomeRecord["shortAudio"]> | undefined; let lastArchiveLine: string | null = null; // Full (untruncated) stderr tail of the most recent attempt, so a failed // download can be classified (rate_limit/network) against everything yt-dlp @@ -461,6 +491,11 @@ async function runManagedDownload( // This video's recency key (upload_date), captured from the prefetched // metadata so the keep-latest cutoff can classify it even before it's on disk. let videoUploadKey = canonicalId ? uploadKeyFor(undefined, canonicalId) : ""; + // Source platform for the "auto" download-format selector. Seeded from the + // channel config / URL and refined to the metadata extractor once prefetched + // (e.g. lbry -> odysee), which is the authoritative signal. + let resolvedPlatform: Platform | null = + opts.channelConfig.platform ?? detectPlatform(opts.videoUrl); if (canonicalId && !opts.signal.aborted) { const videoDir = path.join(channelDir, "data", canonicalId); const prefetchArgs = [ @@ -498,6 +533,7 @@ async function runManagedDownload( // the metadata file; a failed prefetch falls through to the legacy path so // the existing auth-retry logic still gets a chance. if (metadata) infoJsonPath = metaPath; + if (metadata) resolvedPlatform = platformFromMetadata(metadata); videoUploadKey = uploadKeyFor(metadata?.upload_date, canonicalId); const decision = evaluateDownloadFilters({ @@ -540,6 +576,12 @@ async function runManagedDownload( // which switches to transcribe). Cheap: a set/cutoff compare + one stat. const fmt: AudioFormat = opts.audioFormatOverride ?? opts.channelConfig.audioFormat ?? "mp3"; + // The concrete yt-dlp `-f` selector, expanded from the resolved preset against + // the source platform (so "auto" downloads `original` for Odysee). + const downloadFormatSelector = resolveDownloadFormatSelector( + opts.downloadFormatPreset ?? "auto", + resolvedPlatform, + ); const pinned = canonicalId ? await isDoNotClean(path.join(channelDir, "data", canonicalId)) : false; @@ -574,6 +616,7 @@ async function runManagedDownload( plan, fmt, reuseInfoJson, + downloadFormatSelector, ); if (opts.channelConfig.handling === "transcribe") { opts.onLog( @@ -603,7 +646,10 @@ async function runManagedDownload( "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, ...outputArgsForUrl(opts.videoUrl), - ...transcribeHandlingArgsForAudioCheck(opts.channelConfig), + ...transcribeHandlingArgsForAudioCheck( + opts.channelConfig, + downloadFormatSelector, + ), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(opts.channelConfig), @@ -737,6 +783,43 @@ async function runManagedDownload( path.join(channelDir, "data", canonicalId ?? "unknown"); const videoId = path.basename(videoDir); + // Download-time duration guard: probe the produced audio.<fmt> and compare its + // actual length to the metadata duration. A large shortfall means the source + // served a truncated stream (e.g. a CDN-truncated HLS rung) even though yt-dlp + // exited 0. Returns the shortfall metrics when tripped, else null. Skips + // livestreams (unreliable durations), short videos, and unmeasurable files + // (probe failure -> null -> no false positive). The file is left on disk; the + // caller marks the download failed-short-audio so it isn't transcribed. + const probeShortAudio = async (): Promise< + NonNullable<DownloadOutcomeRecord["shortAudio"]> | null + > => { + const meta = await loadRawMetadataFromDir(videoDir); + const expected = meta?.duration; + if ( + !meta || + isLivestreamMetadata(meta) || + typeof expected !== "number" || + expected < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC + ) { + return null; + } + const audioDurationSec = await probeMediaDurationSec({ + ffprobeBin: opts.paths.ffprobeBin, + file: path.join(videoDir, `audio.${fmt}`), + signal: opts.signal, + onLog: opts.onLog, + }); + if (!isShortAudio(audioDurationSec, expected, { isLivestream: false })) { + return null; + } + const a = audioDurationSec as number; + return { + audioDurationSec: Math.round(a * 10) / 10, + expectedDurationSec: expected, + coverage: Math.round((a / expected) * 1000) / 1000, + }; + }; + // ---------- App-side extraction (transcribe handling) ---------- // After a successful non-audio-check transcribe download in app mode, produce // audio.<fmt> from the downloaded source-media container and keep or discard it @@ -794,7 +877,14 @@ async function runManagedDownload( "--ignore-config", "--restrict-filenames", ...FULL_LOG_PROGRESS_ARGS, - ...transcribeMediaArgs(opts.videoUrl, fallbackConfig, plan, fmt, true), + ...transcribeMediaArgs( + opts.videoUrl, + fallbackConfig, + plan, + fmt, + true, + downloadFormatSelector, + ), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(fallbackConfig, fallbackCookieOverride), @@ -839,7 +929,17 @@ async function runManagedDownload( signal: opts.signal, }); } - if (opts.inlineTranscribeOnFallback) { + const shortBeforeInline = await probeShortAudio(); + if (shortBeforeInline) { + shortAudioInfo = shortBeforeInline; + status = "failed-short-audio"; + lastSucceeded = false; + opts.onLog( + `Short audio (${shortBeforeInline.audioDurationSec}s of ` + + `${shortBeforeInline.expectedDurationSec}s); skipping whisper and ` + + `keeping the file for re-download with a different format.\n`, + ); + } else if (opts.inlineTranscribeOnFallback) { // Inline whisper: matches whisperVideoAction's shape. Routes through // the worker pool so the inline transcription respects worker config // and slot limits like any other. @@ -875,6 +975,32 @@ async function runManagedDownload( } } + // ---------- Duration guard (short-audio) ---------- + // Covers the audio-check and non-inline transcribe paths (the inline-whisper + // path checked before transcribing). A truncated download is marked failed so + // it isn't archived or transcribed; the short file is left on disk so it isn't + // immediately re-downloaded into a loop. + if ( + lastSucceeded && + status !== "failed-short-audio" && + (opts.channelConfig.handling === "transcribe" || fellBackToTranscribe) && + !opts.signal.aborted + ) { + const short = await probeShortAudio(); + if (short) { + shortAudioInfo = short; + status = "failed-short-audio"; + lastSucceeded = false; + opts.onLog( + `Short audio for ${videoId}: ${short.audioDurationSec}s of ` + + `${short.expectedDurationSec}s (${Math.round(short.coverage * 100)}% ` + + `of the video). The source served a truncated stream; keeping the ` + + `file and flagging it. Re-download with a different format ` + + `(e.g. Original) to fix.\n`, + ); + } + } + // ---------- Archive append ---------- if (lastSucceeded && opts.appendArchive !== false && lastArchiveLine) { try { @@ -904,6 +1030,7 @@ async function runManagedDownload( attempts, ...(failureClass ? { failureClass } : {}), ...(fellBackToTranscribe ? { fellBackToTranscribe: true } : {}), + ...(shortAudioInfo ? { shortAudio: shortAudioInfo } : {}), }; // Only write the sidecar if we know which dir to put it in. If the very first // attempt failed before metadata could be written, the data/<id> dir may not diff --git a/common/ytdlp/ffprobeDuration.ts b/common/ytdlp/ffprobeDuration.ts @@ -0,0 +1,48 @@ +import { execa } from "execa"; + +export type ProbeMediaDurationOptions = { + ffprobeBin: string; + file: string; + signal?: AbortSignal; + onLog?: (s: string) => void; +}; + +// Measure a media file's container duration (seconds) via ffprobe. Returns null +// when ffprobe is missing, errors, or prints something unparseable — callers +// treat null as "couldn't measure" and skip the duration guard rather than +// failing a download. Distinct from probeAudioStream (ffmpegStreamProbe), which +// is a pass/fail corruption check and yields no duration. +export async function probeMediaDurationSec( + opts: ProbeMediaDurationOptions, +): Promise<number | null> { + const args = [ + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + opts.file, + ]; + opts.onLog?.(`$ ${opts.ffprobeBin} ${args.join(" ")}\n`); + try { + const result = await execa(opts.ffprobeBin, args, { + cancelSignal: opts.signal, + reject: false, + }); + if (result.exitCode !== 0) { + opts.onLog?.( + `ffprobe exited ${result.exitCode}; skipping duration check.\n`, + ); + return null; + } + const sec = Number.parseFloat(String(result.stdout).trim()); + if (!Number.isFinite(sec) || sec <= 0) return null; + return sec; + } catch (err) { + opts.onLog?.( + `ffprobe failed (${(err as Error).message}); skipping duration check.\n`, + ); + return null; + } +} diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -25,6 +25,11 @@ import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailabi import { resolveShardItems } from "../controller/shard"; import { computeKeepWindow, type KeepWindow } from "../controller/keptVideos"; import { downloadOneManaged } from "./downloadOneManaged"; +import { + resolveDownloadFormatPreset, + resolveDownloadFormatSelector, + type DownloadFormatPreset, +} from "./downloadFormat"; import type { TaskTracker } from "../jobs/taskHooks"; import type { JobProgress } from "../jobs/registry"; @@ -74,6 +79,11 @@ export type RunYtdlpOpts = { // Override channelConfig.audioFormat for this run. Honored by download-one-audio // and by the managed download paths (passed through to downloadOneManaged). audioFormatOverride?: AudioFormat; + // Per-run override of the yt-dlp `-f` download format (see + // common/ytdlp/downloadFormat.ts). Honored by download-one-audio and the + // managed download paths (passed through to downloadOneManaged). Beats the + // per-channel and global defaults. + downloadFormatOverride?: DownloadFormatPreset; // Per-run persistence overrides (Phase 2), threaded to downloadOneManaged for // every managed download in this run. keepSourceVideoOverride forces keep // (true) / discard (false); extractImmediately forces extract-now + discard @@ -570,6 +580,14 @@ async function runManagedDownloads( const globalCookies = settings.cookiesFromBrowser; const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback; const globalSkipLiveDownloads = settings.skipLiveDownloads; + // Resolve the download format once per run (override > channel > global), + // leaving the platform-aware "auto" expansion to downloadOneManaged where the + // per-video extractor is known. + const downloadFormatPreset = resolveDownloadFormatPreset({ + override: opts.downloadFormatOverride, + channel: effectiveChannelConfig.downloadFormat, + global: settings.downloadFormat, + }); const sleepSeconds = opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds; @@ -634,6 +652,7 @@ async function runManagedDownloads( keepSourceVideoOverride: opts.keepSourceVideoOverride, extractImmediately: opts.extractImmediately, audioFormatOverride: opts.audioFormatOverride, + downloadFormatPreset, }); } finally { task?.end(); @@ -910,13 +929,24 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { await mkdir(root, { recursive: true }); const fmt = opts.audioFormatOverride ?? opts.channelConfig.audioFormat ?? "mp3"; + // Resolve the download format (override > channel > global) and expand it + // against the source platform so the single-video fix path also honors the + // per-source default (Odysee -> original). + const downloadFormatSelector = resolveDownloadFormatSelector( + resolveDownloadFormatPreset({ + override: opts.downloadFormatOverride, + channel: opts.channelConfig.downloadFormat, + global: getSettings().downloadFormat, + }), + detectPlatform(opts.singleVideoUrl) ?? opts.channelConfig.platform, + ); const args: string[] = [ "--ignore-config", "--restrict-filenames", ...outputArgsForUrl(opts.singleVideoUrl), "--write-info-json", "-f", - "bestaudio/worst", + downloadFormatSelector, "-x", "--audio-format", fmt, diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Source-truncated downloads are now caught at download time, and the yt-dlp download format is selectable (Odysee defaults to `original`).** A download that completed (yt-dlp exit 0) but whose audio is far shorter than the video — e.g. an Odysee/LBRY video whose every HLS rung is CDN-truncated to a few minutes while the full audio lives only in the multi-GB `original` format — used to pass the audio-check (the short file is structurally *clean*) and get transcribed, so only the **post-transcription** coverage detector caught it. Two changes fix this. **(1) A download-time duration guard:** after each managed download the produced `audio.<fmt>` is probed with **ffprobe** (new `common/ytdlp/ffprobeDuration.ts`, `paths.ffprobeBin` / `FFPROBE_BIN`) and compared to the metadata duration using the same thresholds as the transcript-coverage detector (`isShortAudio` in `common/lib/transcriptCoverage.ts`: non-livestream, ≥10min, <50% covered). A large shortfall is recorded as a new terminal **`failed-short-audio`** status with the measured `{audioDurationSec, expectedDurationSec, coverage}` — caught **before** transcription (the inline no-subs-fallback whisper is pre-checked too). The short stub is **kept on disk** (so it isn't silently re-downloaded into a loop) and is **not** classified for platform backoff (a truncated stream isn't a transient error). Surfaced as a new **`shortAudio`** snapshot bucket (mirroring `corrupt-full-source`: excluded from `downloadedNoTranscript` so it's never auto-transcribed, and from the wrong-format cleanup so the kept file isn't deleted) → a **`short_audio`** channel-list filter + **Select short-audio** quick-select + bulk **Re-download truncated** action, a **"Download was truncated at the source"** banner on the video page with a one-click **Re-download as Original**, a **"Channels with truncated downloads (short audio)"** `/actionable` section, and a **"Truncated download — audio far shorter than the video (file kept)"** download badge. The auto-download runner lands a `failed-short-audio` unit as a failed job. **(2) Selectable, platform-aware download format:** the previously-hardcoded `-f bestaudio/worst` is now a preset (`auto` | `original` | `bestaudio` | `bestvideo_audio`; `common/ytdlp/downloadFormat.ts`) resolved with the same override chain as `audioFormat` — per-download (the redownload **Format** picker) > per-channel (`ChannelConfig.downloadFormat`, a select in the channel form) > global (`SiteSettings.downloadFormat`, a select in **Settings**). **`auto` is platform-aware:** Odysee/LBRY resolves to `original/bestaudio/worst` (so the truncation is avoided at the source); everything else stays `bestaudio/worst`. The source platform is taken from the prefetched metadata extractor. The short-audio bucket's re-download reuses the per-video fixer (delete the stub → re-fetch with the per-source default → re-transcribe) via the existing `redownload-incomplete-bucket` job (now also driving `ReplayBucket: "shortAudio"`). See `common/ytdlp/{downloadFormat.ts,ffprobeDuration.ts,downloadOneManaged.ts,runYtdlp.ts}`, `common/lib/{transcriptCoverage.ts,downloadOutcome.ts,settings.ts,channelConfig.ts,paths.ts,transcripts-server.ts}`, `common/controller/{channelSnapshot.ts,autoRunner.ts}`, `common/jobs/jobSpec.ts`, `editor/app/channels/[slug]/{incompleteTranscriptActions.ts,bulkVideoActions.ts,lib/{fixIncompleteTranscript.ts,videoRows.ts,videoRowsServer.ts,stageStatus.ts},components/VideoListPane.tsx,videos/[id]/{videoActions.ts,components/VideoPanel.tsx}}`, `editor/app/{settings/{actions.ts,components/SettingsForm.tsx},channels/components/{ChannelForm.tsx,parseChannelForm.ts},actionable/{lib/loadActionable.ts,page.tsx,components/InlineActionButton.tsx},jobs/jobReplayRegistry.ts}`, and `editor/e2e/download-format-guard.spec.ts`. - **A complete-but-corrupt audio download is no longer re-downloaded forever — it's kept and flagged instead.** When an audio-checked download finished (yt-dlp exit 0, all bytes) but its final integrity probe came back `malformed`, the orchestrator rolled the whole file back and re-downloaded it — repeatedly. For a fast download that completes inside one checkpoint interval no `.good` baseline ever exists, so each "rollback" discarded the entire file and re-fetched it from scratch (a 2.46 GB Odysee video looped until the rollback cap, leaving no audio behind), and a cancel landing mid-loop was deferred behind the next full re-download. Now the final-probe path is bounded: a malformed final probe triggers **exactly one** re-download; if it's still malformed, the orchestrator **keeps the downloaded container on disk** (for inspection) and records a new terminal **`corrupt-full-source`** status — re-downloading a complete file can't change a deterministic verdict. This is distinct from `failed-corrupt-source` (a download that never finished, driven by the checkpoint rollback cap). Surfaced as a new **`corruptFullSource`** channel-snapshot bucket → a Download-stage summary line ("corrupt full source (file kept)"), an **amber download-dot** + `corrupt_full_source` row status in the per-channel video list (folded into the *No audio* filter), and a **"Corrupt full source — download completed but audio is malformed (file kept)"** badge on the video page. It's terminal and **not** retried (auto-runner treats it as skipped; the kept file is an artifact, so it isn't re-queued) and **not** counted as a failed transcription or a transcode-pending video. Cancellation is also fixed: a pending abort now wins over an in-flight rollback decision, and the loop re-checks the abort signal after the final probe so Cancel stops it promptly. See `common/ytdlp/audioCheckedDownload.ts`, `common/ytdlp/downloadOneManaged.ts`, `common/lib/downloadOutcome.ts`, `common/controller/{autoRunner.ts,channelSnapshot.ts}`, `common/ytdlp/runYtdlp.ts`, `editor/app/channels/[slug]/{lib/videoRows.ts,lib/videoRowsServer.ts,lib/stageStatus.ts,components/VideoListPane.tsx,videos/[id]/components/VideoPanel.tsx}`, and `editor/e2e/audio-check-scenarios.spec.ts`. - **The Deploy page is reworked around a clearer build/deploy lifecycle, with one-click build-then-deploy and batch multi-site builds.** The page now reads top-to-bottom as you'd actually ship: **Release notes** (the `## [Unreleased]` changelog preview + Cut release) → **Build & deploy** → optional **Individual steps** → **Build multiple sites**. A new **Build & deploy** button runs the build and, only if it succeeds (and wasn't cancelled), deploys it — as a single managed job with one combined streamed log and one Cancel (`buildAndDeployAction`, a composite `runManagedFunction`; cancelling mid-build skips the deploy). The new **Build multiple sites** panel kicks off a build (optionally build+deploy) for several sites at once, each rendered as its own live status-chipped log lane (`BuildSitesPanel` + `JobLane`); in Basic mode the jobs serialize on the shared build/deploy queue (the `export/` output tree is shared), with a note that true parallelism arrives with Docker mode. A **Build mode** toggle (Basic | Docker) on the page persists the choice as the default (`settings.buildPipeline`, also editable on Settings); Docker mode is a follow-up and currently falls back to a basic build with an inline notice. The build/deploy commands now share a child-streaming helper (`common/jobs/runChild.ts`) and mode-routing core (`editor/app/deploy/buildDeployCore.ts`). See `editor/app/deploy/{page.tsx,buildAction usage,components/*}`, `editor/app/build/buildAction.ts`, and `common/lib/settings.ts`. - **Truncated transcripts are now detected and flagged for re-download.** When an audio download silently stops early (yt-dlp exits `ok`, `download-outcome.json` records success), whisper transcribes only the few minutes that landed — so a 2h22m video ends up with a ~7-minute transcript and nothing warns you. A new coverage check (last cue end ÷ video duration) flags any non-livestream video ≥10min whose transcript covers <50% of its runtime. The single source of truth is `common/lib/transcriptCoverage.ts` (`transcriptCoverage` + `isIncompleteTranscript`, with named thresholds), read from each video's `transcript.cues.json` so the existing corpus is flagged with no migration. Surfaced everywhere: a new **`incompleteTranscript`** channel-snapshot bucket → an **"Incomplete transcript"** filter chip and an **amber transcribed-dot** in the per-channel video list; a warning banner on the video page ("Transcript covers 6:52 of 2:22:21 (4.8%)…") with a one-click **Re-download & re-transcribe** button; and an **"Channels with incomplete (truncated) transcripts"** section on `/actionable`. The fix action (`redownloadIncompleteTranscriptAction`) deletes the truncated audio first, then re-downloads and re-transcribes — re-running whisper alone would just reproduce the short transcript. See `common/controller/channelSnapshot.ts`, `editor/app/channels/[slug]/{lib/videoRows.ts,lib/videoRowsServer.ts,lib/stageStatus.ts,components/VideoListPane.tsx,videos/[id]/{components/VideoPanel.tsx,videoActions.ts,page.tsx},page.tsx}`, and `editor/app/actionable/{lib/loadActionable.ts,page.tsx}`. diff --git a/editor/app/actionable/components/InlineActionButton.tsx b/editor/app/actionable/components/InlineActionButton.tsx @@ -12,6 +12,7 @@ import { import { clearIncompleteTranscriptsAction, redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, } from "../../channels/[slug]/incompleteTranscriptActions"; import { refreshChannelSnapshotAction } from "../../channels/actions"; @@ -20,6 +21,7 @@ type Variant = | { kind: "transcribeMissing"; slug: string; audioFormat?: AudioFormat } | { kind: "redownloadIncomplete"; slug: string } | { kind: "clearIncomplete"; slug: string } + | { kind: "redownloadShortAudio"; slug: string } | { kind: "cleanTranscribedAudio"; slug: string } | { kind: "cleanExtraFormats"; slug: string } | { kind: "refreshReport"; slug: string }; @@ -36,6 +38,7 @@ const LABEL: Record<Variant["kind"], { idle: string; running: string }> = { transcribeMissing: { idle: "Transcribe pending", running: "Queuing…" }, redownloadIncomplete: { idle: "Re-download & re-transcribe", running: "Queuing…" }, clearIncomplete: { idle: "Clear & re-queue", running: "Clearing…" }, + redownloadShortAudio: { idle: "Re-download (corrected format)", running: "Queuing…" }, cleanTranscribedAudio: { idle: "Clean audio", running: "Queuing…" }, cleanExtraFormats: { idle: "Clean extra formats", running: "Queuing…" }, refreshReport: { idle: "Refresh report", running: "Refreshing…" }, @@ -68,6 +71,9 @@ async function runAction(variant: Variant): Promise<StreamActionResult> { if (variant.kind === "redownloadIncomplete") { return redownloadIncompleteBucketAction(variant.slug); } + if (variant.kind === "redownloadShortAudio") { + return redownloadShortAudioBucketAction(variant.slug); + } if (variant.kind === "cleanTranscribedAudio") { return cleanAudioAction(variant.slug); } diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -21,6 +21,7 @@ export type ActionableSummary = { undownloaded: ActionableRow[]; untranscribed: ActionableRow[]; incompleteTranscripts: ActionableRow[]; + shortAudio: ActionableRow[]; cleanTranscribedAudio: ActionableRow[]; cleanExtraFormats: ActionableRow[]; staleOrMissing: ActionableRow[]; @@ -68,6 +69,12 @@ export function actionableIncompleteTranscriptCount(row: ActionableRow): number return row.snapshot?.buckets.incompleteTranscript?.length ?? 0; } +// Downloads the duration guard flagged as truncated at the source (short audio +// kept on disk, not transcribed). Default 0 for snapshots predating the bucket. +export function actionableShortAudioCount(row: ActionableRow): number { + return row.snapshot?.buckets.shortAudio?.length ?? 0; +} + // Cleanup buckets are filtered by "do not clean" at snapshot-generation time, // so the length is the actionable count directly (default undefined → 0 for // snapshots written before the bucket existed). @@ -123,6 +130,10 @@ export async function loadActionableSummary( actionableIncompleteTranscriptCount(a), ); + const shortAudio = rows + .filter((r) => actionableShortAudioCount(r) > 0) + .sort((a, b) => actionableShortAudioCount(b) - actionableShortAudioCount(a)); + const cleanTranscribedAudio = rows .filter((r) => actionableCleanTranscribedCount(r) > 0) .sort( @@ -147,6 +158,7 @@ export async function loadActionableSummary( undownloaded, untranscribed, incompleteTranscripts, + shortAudio, cleanTranscribedAudio, cleanExtraFormats, staleOrMissing, diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -8,6 +8,7 @@ import { actionableCleanTranscribedBytes, actionableCleanTranscribedCount, actionableIncompleteTranscriptCount, + actionableShortAudioCount, actionableUndownloadedCount, actionableUntranscribedCount, loadActionableSummary, @@ -51,6 +52,7 @@ export default async function ActionablePage() { summary.undownloaded.length === 0 && summary.untranscribed.length === 0 && summary.incompleteTranscripts.length === 0 && + summary.shortAudio.length === 0 && summary.cleanTranscribedAudio.length === 0 && summary.cleanExtraFormats.length === 0 && summary.staleOrMissing.length === 0; @@ -126,6 +128,32 @@ export default async function ActionablePage() { }, { config: { + id: "short-audio", + title: "Channels with truncated downloads (short audio)", + description: + "Downloads that completed but whose audio is far shorter than the video — the source served a truncated stream, caught before transcription. The stub is kept (not transcribed). “Re-download (corrected format)” deletes it and re-fetches with the per-source default (Original for Odysee).", + countLabel: "truncated", + emptyLabel: "None detected.", + getCount: actionableShortAudioCount, + primaryAction: (r) => ( + <span className="inline-flex items-center justify-end gap-2 flex-wrap"> + <InlineActionButton + variant={{ kind: "redownloadShortAudio", slug: r.channel.slug }} + /> + <Link + href={`/channels/${r.channel.slug}?filter=short_audio`} + aria-label={`review short-audio downloads for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-zinc-300 dark:border-zinc-700 text-xs font-medium hover:bg-zinc-100 dark:hover:bg-zinc-800 whitespace-nowrap" + > + Review + </Link> + </span> + ), + }, + rows: summary.shortAudio, + }, + { + config: { id: "clean-transcribed-audio", title: "Channels with cleanable transcribed audio", description: diff --git a/editor/app/channels/[slug]/bulkVideoActions.ts b/editor/app/channels/[slug]/bulkVideoActions.ts @@ -18,6 +18,7 @@ import { retryBucketAction } from "./pipelineActions"; import { clearIncompleteTranscriptsAction, redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, } from "./incompleteTranscriptActions"; import { deleteOneVideoDir, @@ -65,6 +66,17 @@ export async function bulkRedownloadIncompleteAction( return redownloadIncompleteBucketAction(slug, videoIds, queueKey); } +// Bulk re-download short-audio (source-truncated) downloads: queues ONE batch +// job that deletes the kept stub and re-fetches each selected video with the +// per-source default format (Original for Odysee), then re-transcribes. +export async function bulkRedownloadShortAudioAction( + slug: string, + videoIds: string[], + queueKey?: string, +): Promise<StreamActionResult> { + return redownloadShortAudioBucketAction(slug, videoIds, queueKey); +} + // Bulk clear truncated transcripts: synchronous fs op (delete audio + transcript) // that enables the auto-runners, so it reports a per-id summary like the other // clear/remove bulk actions. Destructive — the caller confirms first. diff --git a/editor/app/channels/[slug]/components/VideoListPane.tsx b/editor/app/channels/[slug]/components/VideoListPane.tsx @@ -13,6 +13,7 @@ import { bulkDeleteVideoDirsAction, bulkMarkUntranscribableAction, bulkRedownloadIncompleteAction, + bulkRedownloadShortAudioAction, bulkRemoveAudioAction, bulkRemoveWrongFormatAudioAction, bulkRetryDownloadAction, @@ -25,6 +26,7 @@ type BulkAction = | "retry" | "redownload_incomplete" | "clear_incomplete" + | "redownload_short_audio" | "untranscribable" | "clear_failed" | "remove_audio" @@ -36,6 +38,7 @@ const BULK_ACTION_OPTIONS: { value: BulkAction; label: string }[] = [ { value: "retry", label: "Retry download" }, { value: "redownload_incomplete", label: "Re-download & re-transcribe" }, { value: "clear_incomplete", label: "Clear incomplete (audio+transcript)" }, + { value: "redownload_short_audio", label: "Re-download truncated (short audio)" }, { value: "untranscribable", label: "Mark untranscribable" }, { value: "clear_failed", label: "Clear failed markers" }, { value: "remove_audio", label: "Remove audio files" }, @@ -65,6 +68,7 @@ const FILTER_OPTIONS: { value: VideoFilter; label: string }[] = [ { value: "downloaded_no_transcript", label: "No transcript" }, { value: "partial", label: "Partial" }, { value: "incomplete_transcript", label: "Incomplete transcript" }, + { value: "short_audio", label: "Short audio" }, { value: "untranscribable", label: "Untranscribable" }, { value: "running", label: "Running" }, { value: "transcribed", label: "Transcribed" }, @@ -290,6 +294,18 @@ export function VideoListPane({ }); } + // Videos whose download was truncated at the source (short audio kept on disk). + const hasShortAudio = useMemo(() => rows.some((r) => r.shortAudio), [rows]); + function selectShortAudio() { + setSelected((prev) => { + const next = new Set(prev); + for (const r of rows) { + if (r.shortAudio) next.add(r.id); + } + return next; + }); + } + const deleteArmed = deleteConfirm.trim().toLowerCase() === "delete"; const applyDisabled = pending || (action === "delete" && !deleteArmed); @@ -310,6 +326,11 @@ export function VideoListPane({ bulkRedownloadIncompleteAction(s, ids, incompleteQueue), ); break; + case "redownload_short_audio": + doStreamingBulk((s, ids) => + bulkRedownloadShortAudioAction(s, ids, incompleteQueue), + ); + break; case "clear_incomplete": if ( !confirm( @@ -429,6 +450,15 @@ export function VideoListPane({ Select incomplete </button> )} + {hasShortAudio && ( + <button + type="button" + onClick={selectShortAudio} + className="rounded border border-zinc-200 dark:border-zinc-800 px-2 py-0.5 hover:bg-zinc-100 dark:hover:bg-zinc-800" + > + Select short-audio + </button> + )} {visibleRows.length > 0 && ( <> <button @@ -568,7 +598,8 @@ export function VideoListPane({ </label> </> )} - {action === "redownload_incomplete" && ( + {(action === "redownload_incomplete" || + action === "redownload_short_audio") && ( <QueueControl value={incompleteQueue} onChange={setIncompleteQueue} diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts @@ -19,6 +19,7 @@ import { clearIncompleteTranscriptOne, fixIncompleteTranscriptOne, incompleteIdsForChannel, + shortAudioIdsForChannel, } from "./lib/fixIncompleteTranscript"; import type { BulkActionSummary } from "./bulkVideoActions"; @@ -115,6 +116,70 @@ export async function redownloadIncompleteBucketAction( }); } +// Re-download the channel's short-audio bucket (downloads the duration guard +// flagged as truncated at the source). Reuses the same per-video fixer and the +// same bucket job kind, parameterized with bucket "shortAudio" so a bookmarked +// re-run re-derives the live members. The re-download deletes the kept stub and +// re-fetches with the per-source default format (Original for Odysee), which is +// what actually recovers the full audio. +export async function redownloadShortAudioBucketAction( + slug: string, + ids?: string[], + queueKey?: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const config = await readChannelConfig(paths, slug); + if (!config) return { ok: false, error: `Channel "${slug}" not found` }; + const source = ids && ids.length ? ids : await shortAudioIdsForChannel(slug, paths); + const cleaned = dedupeIds(source); + if (cleaned.length === 0) { + return { ok: false, error: "No truncated downloads to fix.", info: true }; + } + return runManagedFunction({ + kind: "redownload-incomplete-bucket", + queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey), + paths, + channelSlug: slug, + spec: { + kind: "redownload-incomplete-bucket", + slug, + bucket: "shortAudio", + params: { queueKey }, + }, + fn: async (onLog, signal, _setProgress, ctx) => { + const tracker = makeTaskTracker(ctx, onLog); + let succeeded = 0; + let failed = 0; + for (const id of cleaned) { + if (ctx.drainSignal?.aborted) { + onLog(`Drain requested; stopping before ${id}.`); + break; + } + try { + onLog(`Re-downloading & re-transcribing ${id}…`); + await fixIncompleteTranscriptOne({ + slug, + videoId: id, + config, + paths, + onLog, + signal, + tracker, + }); + succeeded++; + } catch (e) { + failed++; + onLog(`Failed ${id}: ${(e as Error).message}`); + } + } + onLog( + `Re-download short-audio: ${succeeded} fixed, ${failed} failed of ${cleaned.length}.`, + ); + revalidatePath(`/channels/${slug}`); + }, + }); +} + // Clear & release: synchronously delete the truncated audio + transcript for the // flagged videos so they fall back into the normal pending pipeline, then enable // the auto-runners so they reprocess automatically. Destructive — gate every diff --git a/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts b/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts @@ -38,6 +38,20 @@ export async function incompleteIdsForChannel( return Array.isArray(ids) ? ids : []; } +// The current snapshot's short-audio ids for a channel (downloads the duration +// guard flagged as truncated at the source). Empty when the snapshot is missing +// or lacks the bucket. The same fixIncompleteTranscriptOne re-download path fixes +// these — it deletes the kept stub and re-fetches (now via the per-source +// default, i.e. `original` for Odysee). +export async function shortAudioIdsForChannel( + slug: string, + paths: Paths = getPaths(), +): Promise<string[]> { + const snap = await readChannelSnapshot(paths, slug); + const ids = snap?.buckets?.shortAudio; + return Array.isArray(ids) ? ids : []; +} + // Re-download the full audio and re-transcribe one video in place. The new // transcript overwrites transcript.json (transcribeWithWorker) and normalize // regenerates transcript.cues.json, so there is never a window with no diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -29,6 +29,7 @@ export function normalizeBuckets( nonStandardVtt: raw?.nonStandardVtt ?? [], skippedByFilter: raw?.skippedByFilter ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], + shortAudio: raw?.shortAudio ?? [], }; } diff --git a/editor/app/channels/[slug]/lib/videoRows.ts b/editor/app/channels/[slug]/lib/videoRows.ts @@ -39,6 +39,11 @@ export type VideoRow = { // duration — the audio download truncated silently. Independent flag (not a // `status`) so it composes with `transcribed`. See transcriptCoverage. incompleteTranscript: boolean; + // Download completed but the audio was far shorter than the video — the source + // served a truncated stream (download-outcome "failed-short-audio"). The stub + // is kept on disk; re-download with a different format to fix. Independent flag + // (the video isn't transcribed), mirroring incompleteTranscript. + shortAudio: boolean; running: boolean; status: VideoRowStatus; }; @@ -58,6 +63,7 @@ export type VideoFilter = | "transcribed" | "partial" | "incomplete_transcript" + | "short_audio" | "untranscribable" | "running"; @@ -69,6 +75,7 @@ const VIDEO_FILTERS: readonly VideoFilter[] = [ "transcribed", "partial", "incomplete_transcript", + "short_audio", "untranscribable", "running", ]; @@ -100,6 +107,8 @@ function matchesFilter(r: VideoRow, filter: VideoFilter): boolean { return r.partial; case "incomplete_transcript": return r.incompleteTranscript; + case "short_audio": + return r.shortAudio; case "untranscribable": return r.untranscribable; case "running": diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts @@ -41,6 +41,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { const untranscribable = new Set(buckets.untranscribable); const partial = new Set(buckets.partialDownloads); const incompleteTranscript = new Set(buckets.incompleteTranscript); + const shortAudioSet = new Set(buckets.shortAudio ?? []); const corruptSourceSet = new Set(buckets.corruptSource); const corruptFullSourceSet = new Set(buckets.corruptFullSource); const failedTranscription = new Set(input.failedTranscriptionIds); @@ -111,6 +112,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { untranscoded.has(id) || multipleAudioFormats.has(id), excluded, incompleteTranscript: incompleteTranscript.has(id), + shortAudio: shortAudioSet.has(id), running: runningIds.has(id), status, }); diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -32,6 +32,10 @@ import { whisperVideoAction, type DeleteDirActionResult, } from "../videoActions"; +import { + DOWNLOAD_FORMAT_LABELS, + DOWNLOAD_FORMAT_PRESETS, +} from "yt-dlp-transcript-common/ytdlp/downloadFormat"; export type VideoFile = { name: string; @@ -213,6 +217,15 @@ export function VideoPanel({ existingQueues={existingQueues} /> )} + {downloadOutcome?.status === "failed-short-audio" && ( + <ShortAudioBanner + slug={slug} + videoId={videoId} + shortAudio={downloadOutcome.shortAudio} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + )} <PipelineStageCard id="availability-history" title="Availability history" @@ -538,6 +551,8 @@ function RedownloadSection({ }) { const [mode, setMode] = useState<DownloadMode>("pipeline"); const [queueKey, setQueueKey] = useState(defaultQueueKey); + // "" = inherit the channel/global default; otherwise a DownloadFormatPreset. + const [downloadFormat, setDownloadFormat] = useState(""); const desc = (() => { if (mode === "whisper") { @@ -566,7 +581,13 @@ function RedownloadSection({ const actionLabel = `${verb} for ${videoId}`; const trigger = mode === "whisper" ? () => whisperVideoAction(slug, videoId, queueKey) - : () => downloadVideoPipelineAction(slug, videoId, queueKey); + : () => + downloadVideoPipelineAction( + slug, + videoId, + queueKey, + downloadFormat || undefined, + ); return ( <div className="flex flex-col gap-3"> @@ -583,6 +604,24 @@ function RedownloadSection({ <option value="whisper">Audio + Whisper (skip pipeline)</option> </select> </label> + {mode === "pipeline" && ( + <label className="flex items-center gap-2 text-sm"> + Format + <select + value={downloadFormat} + onChange={(e) => setDownloadFormat(e.target.value)} + aria-label="download format" + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + <option value="">Inherit (channel / global)</option> + {DOWNLOAD_FORMAT_PRESETS.map((p) => ( + <option key={p} value={p}> + {DOWNLOAD_FORMAT_LABELS[p]} + </option> + ))} + </select> + </label> + )} <StreamActionLog trigger={trigger} cancelAction={cancelJobAction} @@ -1244,6 +1283,8 @@ function DownloadOutcomeBadge({ return "Corrupt source — download couldn't finish"; case "corrupt-full-source": return "Corrupt full source — download completed but audio is malformed (file kept)"; + case "failed-short-audio": + return "Truncated download — audio far shorter than the video (file kept)"; default: return outcome.status; } @@ -1337,6 +1378,69 @@ function IncompleteTranscriptBanner({ ); } +// The download completed but the audio was far shorter than the video — the +// source served a truncated stream (e.g. a CDN-truncated HLS rung). The stub is +// kept on disk; re-downloading with a different format (Original) fetches the +// full audio. Offers a one-click re-download forcing the Original format. +function ShortAudioBanner({ + slug, + videoId, + shortAudio, + defaultQueueKey, + existingQueues, +}: { + slug: string; + videoId: string; + shortAudio?: { + audioDurationSec: number; + expectedDurationSec: number; + coverage: number; + }; + defaultQueueKey: string; + existingQueues: string[]; +}) { + const [queueKey, setQueueKey] = useState(defaultQueueKey); + const actionLabel = `Re-download ${videoId} as Original`; + return ( + <div + role="alert" + aria-label="short audio" + className="flex flex-col gap-3 rounded border px-3 py-2 text-sm border-amber-300 bg-amber-50 text-amber-900 dark:border-amber-900 dark:bg-amber-950 dark:text-amber-200" + > + <div className="flex flex-col gap-1"> + <span className="font-medium">Download was truncated at the source</span> + <span> + {shortAudio + ? `The downloaded audio is only ${formatDuration(shortAudio.audioDurationSec)} of ${formatDuration(shortAudio.expectedDurationSec)} (${Math.round(shortAudio.coverage * 100)}%). ` + : "The downloaded audio is far shorter than the video. "} + The source served a truncated stream for the selected format, so it was + flagged before transcription (the file is kept, not transcribed). + Re-download with the <strong>Original</strong> format to fetch the full + audio. + </span> + </div> + <StreamActionLog + trigger={() => + downloadVideoPipelineAction(slug, videoId, queueKey, "original") + } + cancelAction={cancelJobAction} + buttonLabel="Re-download as Original" + runningLabel="Re-downloading…" + label={actionLabel} + extraControls={ + <QueueControl + value={queueKey} + onChange={setQueueKey} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={actionLabel} + /> + } + /> + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -9,6 +9,11 @@ import type { ChannelConfig, } from "yt-dlp-transcript-common/lib/channelConfig"; import { AUDIO_FORMAT_VALUES } from "yt-dlp-transcript-common/lib/channelConfig"; +import { + isDownloadFormatPreset, + resolveDownloadFormatPreset, + type DownloadFormatPreset, +} from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; @@ -128,6 +133,7 @@ export async function downloadVideoPipelineAction( slug: string, videoId: string, queueKey?: string, + downloadFormat?: string, ): Promise<StreamActionResult> { const r = await loadConfigOrError(slug); if (!r.ok) return r; @@ -141,6 +147,10 @@ export async function downloadVideoPipelineAction( }; } const settings = getSettings(); + // Per-download override (from the redownload picker), falling back to the + // channel default then the global default. An unrecognized value is ignored. + const formatOverride: DownloadFormatPreset | undefined = + isDownloadFormatPreset(downloadFormat) ? downloadFormat : undefined; return runManagedFunction({ kind: "download-one-pipeline", queueKey: videoQueueKey(r.config, queueKey), @@ -164,6 +174,11 @@ export async function downloadVideoPipelineAction( globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, + downloadFormatPreset: resolveDownloadFormatPreset({ + override: formatOverride, + channel: r.config.downloadFormat, + global: settings.downloadFormat, + }), appendArchive: true, }); revalidatePath(`/channels/${slug}/videos/${videoId}`); diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx @@ -1,5 +1,9 @@ import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import { + DOWNLOAD_FORMAT_LABELS, + DOWNLOAD_FORMAT_PRESETS, +} from "yt-dlp-transcript-common/ytdlp/downloadFormat"; +import { AUDIO_CHECK_COPY_TIMEOUT_DEFAULT_SECONDS, AUDIO_CHECK_COPY_TIMEOUT_MAX_SECONDS, AUDIO_CHECK_COPY_TIMEOUT_MIN_SECONDS, @@ -139,6 +143,27 @@ export function ChannelForm({ <option value="opus">opus</option> </select> </label> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Download format</span> + <select + name="downloadFormat" + defaultValue={c?.downloadFormat ?? ""} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + <option value="">Inherit (global default)</option> + {DOWNLOAD_FORMAT_PRESETS.map((p) => ( + <option key={p} value={p}> + {DOWNLOAD_FORMAT_LABELS[p]} + </option> + ))} + </select> + <span className="text-xs text-zinc-500"> + The yt-dlp <code>-f</code> selector for this channel&apos;s + downloads. Leave on Inherit to use the global default; choose{" "} + <strong>Original</strong> for sources that only serve full-length + audio in their original file (e.g. Odysee). + </span> + </label> <label className="flex items-start gap-2 text-sm"> <input type="checkbox" diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts @@ -17,6 +17,7 @@ import { detectPlatform, type Platform, } from "yt-dlp-transcript-common/lib/platform"; +import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; export type ParsedChannelForm = { name: string; @@ -32,6 +33,7 @@ export const CHANNEL_FORM_FIELDS = [ "platform", "url", "audioFormat", + "downloadFormat", "keepSourceVideo", "keepLatest", "extractionMode", @@ -69,6 +71,9 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm { audioFormatRaw === "opus" ? audioFormatRaw : undefined; + const downloadFormatRaw = String(formData.get("downloadFormat") ?? ""); + const downloadFormat: ChannelConfig["downloadFormat"] | undefined = + isDownloadFormatPreset(downloadFormatRaw) ? downloadFormatRaw : undefined; const keepSourceVideo = formData.get("keepSourceVideo") != null; // Keep-latest retention/persistence window. Blank = inherit-off (omit), so a @@ -204,6 +209,7 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm { if (platform) config.platform = platform; if (url) config.url = url; if (audioFormat) config.audioFormat = audioFormat; + if (downloadFormat) config.downloadFormat = downloadFormat; if (keepSourceVideo) config.keepSourceVideo = true; if (keepLatest != null) config.keepLatest = keepLatest; if (extractionMode) config.extractionMode = extractionMode; diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -34,7 +34,10 @@ import { transcribeMissingAction, } from "../channels/[slug]/whisperActions"; import { persistKeptAction } from "../channels/[slug]/persistActions"; -import { redownloadIncompleteBucketAction } from "../channels/[slug]/incompleteTranscriptActions"; +import { + redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, +} from "../channels/[slug]/incompleteTranscriptActions"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -100,9 +103,13 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." }; const ids = await idsForBucket(spec.slug, spec.bucket); if (ids.length === 0) { - return { ok: false, error: "No incomplete transcripts right now.", info: true }; + return { ok: false, error: "Nothing to re-download right now.", info: true }; } - return redownloadIncompleteBucketAction(spec.slug, ids, queueKey); + // The same job kind drives both the incompleteTranscript and shortAudio + // buckets; route by the captured bucket so the re-run re-stamps the right one. + return spec.bucket === "shortAudio" + ? redownloadShortAudioBucketAction(spec.slug, ids, queueKey) + : redownloadIncompleteBucketAction(spec.slug, ids, queueKey); }, "retry-bucket": async (spec) => { const { p, queueKey } = params(spec); diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -21,6 +21,7 @@ import { type SocialLink, } from "yt-dlp-transcript-common/lib/settings"; import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps"; +import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { sanitizeWorkers, validateWorkers, @@ -54,6 +55,10 @@ export async function saveSettingsAction( const reportDebouncePreset = isReportDebouncePreset(reportDebouncePresetRaw) ? reportDebouncePresetRaw : DEFAULT_REPORT_DEBOUNCE_PRESET; + const downloadFormatRaw = String(formData.get("downloadFormat") ?? "").trim(); + const downloadFormat = isDownloadFormatPreset(downloadFormatRaw) + ? downloadFormatRaw + : "auto"; if (!adminTitle) return { ok: false, error: "Admin title is required" }; @@ -201,6 +206,7 @@ export async function saveSettingsAction( workers, cookiesFromBrowser, sleepBetweenDownloadsSeconds: sleepParsed, + downloadFormat, minFreeDiskGB: minFreeDiskParsed, parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, inlineTranscribeOnFallback, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -6,6 +6,10 @@ import { type SaveResult, } from "../actions"; import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings"; +import { + DOWNLOAD_FORMAT_LABELS, + DOWNLOAD_FORMAT_PRESETS, +} from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import type { TranscriptionAppDescriptor } from "yt-dlp-transcript-common/lib/transcriptionApps"; import { SocialLinksField, @@ -75,6 +79,28 @@ export function SettingsForm({ initial, apps }: Props) { type="number" hint="Pause inserted between per-video yt-dlp invocations in managed batch downloads (download-from-playlist, download-missing). 0 disables. Default 10s. Each channel can override this in its Advanced settings." /> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Default download format</span> + <select + name="downloadFormat" + defaultValue={initial.downloadFormat} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + {DOWNLOAD_FORMAT_PRESETS.map((p) => ( + <option key={p} value={p}> + {DOWNLOAD_FORMAT_LABELS[p]} + </option> + ))} + </select> + <span className="text-xs text-zinc-500"> + The yt-dlp <code>-f</code> selector for managed downloads.{" "} + <strong>Auto</strong> picks per-source: Odysee/LBRY downloads its + full-length <code>original</code> file (its HLS streams are + CDN-truncated to a few minutes), everything else uses{" "} + <code>bestaudio/worst</code>. Each channel can override this in its + Advanced settings, and a single re-download can override it too. + </span> + </label> <Field label="Minimum free disk space (GB)" name="minFreeDiskGB" diff --git a/editor/e2e/download-format-guard.spec.ts b/editor/e2e/download-format-guard.spec.ts @@ -0,0 +1,191 @@ +// Download-time duration guard + selectable / per-source download format. +// +// The guard compares a freshly-downloaded audio file's actual duration (via +// ffprobe) to the video's metadata duration; a large shortfall is recorded as +// "failed-short-audio" BEFORE transcription, the stub is kept on disk, and the +// video is surfaced (channel filter, per-video banner, actionable section). +// The format selector is platform-aware: Odysee "auto" downloads `original`. +// +// The fake yt-dlp embeds a `__DUR=120__` marker in audio for ids containing +// `truncaud` (fake-ffprobe reports 120s); ids with `longvideo` get a 6000s +// metadata duration; ids with `odyseevid` advertise the lbry extractor. The +// fake logs the resolved `-f` selector to fake-ytdlp.invocations. + +import { mkdir, writeFile, readFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { baseUrl } from "./baseUrl"; +import { + pathExists, + readJson, + resetData, + resolvePath, +} from "./helpers"; + +type DownloadOutcome = { + status: string; + shortAudio?: { + audioDurationSec: number; + expectedDurationSec: number; + coverage: number; + }; +}; + +async function invalidate() { + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); +} + +// Create a transcribe-handling channel (no audio-check → the -x extraction +// path) with a playlist of the given video ids. +async function makeChannel( + slug: string, + ids: string[], + config: Record<string, unknown> = {}, +): Promise<void> { + const root = resolvePath(`test-transcripts/channels/${slug}`); + await mkdir(`${root}/data`, { recursive: true }); + await writeFile( + `${root}/config.json`, + JSON.stringify({ + handling: "transcribe", + name: slug, + platform: "youtube", + url: `https://www.youtube.com/@${slug}`, + audioFormat: "mp3", + ...config, + }), + ); + await writeFile( + `${root}/playlist`, + ids.map((id) => `https://www.youtube.com/watch?v=${id}\n`).join(""), + ); + await invalidate(); +} + +async function downloadAll(page: import("@playwright/test").Page, slug: string) { + await page.goto(`/channels/${slug}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText("download complete", { timeout: 30_000 }); +} + +async function outcome(slug: string, id: string): Promise<DownloadOutcome> { + return readJson<DownloadOutcome>( + `test-transcripts/channels/${slug}/data/${id}/download-outcome.json`, + ); +} + +async function waitForOutcome( + slug: string, + id: string, +): Promise<DownloadOutcome> { + let last: DownloadOutcome | null = null; + await expect + .poll( + async () => { + try { + last = await outcome(slug, id); + return last.status; + } catch { + return null; + } + }, + { timeout: 30_000 }, + ) + .not.toBe(null); + return last!; +} + +test("duration guard flags a source-truncated download and keeps the stub", async ({ + page, +}) => { + const SLUG = "guard-trips"; + // truncaud + longvideo: 6000s metadata, 120s audio → 2% → short. + const TRUNC = "truncaudlongvideo1"; + // longvideo only: 6000s metadata, full audio → passes the guard. + const FULL = "fulllongvideo1"; + await resetData(); + await makeChannel(SLUG, [TRUNC, FULL]); + await downloadAll(page, SLUG); + + // Truncated download: recorded failed-short-audio, audio KEPT, NOT transcribed. + const truncOutcome = await waitForOutcome(SLUG, TRUNC); + expect(truncOutcome.status).toBe("failed-short-audio"); + expect(truncOutcome.shortAudio?.expectedDurationSec).toBe(6000); + expect(truncOutcome.shortAudio?.audioDurationSec).toBe(120); + expect( + await pathExists(`test-transcripts/channels/${SLUG}/data/${TRUNC}/audio.mp3`), + ).toBe(true); + expect( + await pathExists( + `test-transcripts/channels/${SLUG}/data/${TRUNC}/transcript.json`, + ), + ).toBe(false); + + // The full-length download passed the guard (not flagged). + const fullOutcome = await waitForOutcome(SLUG, FULL); + expect(fullOutcome.status).not.toBe("failed-short-audio"); + + // Channel list: the short_audio chip filters to exactly the truncated video. + const list = page.getByLabel("videos", { exact: true }); + await expect + .poll( + async () => { + await page.goto(`/channels/${SLUG}?filter=short_audio`); + return list.getByLabel(`open ${TRUNC}`).count(); + }, + { timeout: 15_000 }, + ) + .toBe(1); + await expect(list.getByLabel(`open ${FULL}`)).toBeHidden(); + // The kept stub is excluded from auto-transcribe (downloadedNoTranscript). + await page.goto(`/channels/${SLUG}?filter=downloaded_no_transcript`); + await expect(list.getByLabel(`open ${TRUNC}`)).toBeHidden(); + + // Video page: the short-audio banner offers a re-download as Original. + await page.goto(`/channels/${SLUG}/videos/${TRUNC}`); + const banner = page.getByLabel("short audio"); + await expect(banner).toBeVisible(); + await expect( + banner.getByRole("button", { name: /Re-download as Original/ }), + ).toBeVisible(); + + // Actionable: the channel is listed under truncated downloads. + await page.goto(`/actionable`); + const section = page.getByRole("region", { name: "short-audio", exact: true }); + await expect(section).toBeVisible(); + await expect(section.getByLabel(`short-audio row ${SLUG}`)).toBeVisible(); +}); + +test("download format: Odysee auto picks original; YouTube uses bestaudio; channel override forces original", async ({ + page, +}) => { + const SLUG = "fmt-auto"; + // odyseevid → lbry extractor → auto resolves "odysee" → original. + const ODY = "odyseevidlongvideo2"; + // plain youtube → bestaudio/worst. + const YT = "ytlongvideo2"; + await resetData(); + await makeChannel(SLUG, [ODY, YT]); + await downloadAll(page, SLUG); + await waitForOutcome(SLUG, ODY); + await waitForOutcome(SLUG, YT); + + const invocations = await readFile( + resolvePath(`test-transcripts/channels/${SLUG}/fake-ytdlp.invocations`), + "utf8", + ); + expect(invocations).toContain("format:original/bestaudio/worst"); + expect(invocations).toContain("format:bestaudio/worst"); + + // A per-channel downloadFormat override forces original even for YouTube. + const SLUG2 = "fmt-override"; + const YT2 = "ytlongvideo3"; + await makeChannel(SLUG2, [YT2], { downloadFormat: "original" }); + await downloadAll(page, SLUG2); + await waitForOutcome(SLUG2, YT2); + const inv2 = await readFile( + resolvePath(`test-transcripts/channels/${SLUG2}/fake-ytdlp.invocations`), + "utf8", + ); + expect(inv2).toContain("format:original/bestaudio/worst"); +}); diff --git a/editor/e2e/fixtures/bin/fake-ffprobe.mjs b/editor/e2e/fixtures/bin/fake-ffprobe.mjs @@ -0,0 +1,32 @@ +#!/usr/bin/env node +// E2E fake ffprobe. The duration guard invokes: +// ffprobe -v error -show_entries format=duration -of default=...:nokey=1 <file> +// We report a duration read from a marker the fake-ytdlp embeds in the audio +// file (`__DUR=<seconds>__`). A file with no marker reports a large duration so +// the guard never falsely trips on it (matches a full-length download). +import { readFile } from "node:fs/promises"; + +const argv = process.argv.slice(2); +// The input file is the only non-flag argument (and not a value of -show_entries +// / -of). It's the last positional in the guard's invocation. +const file = argv[argv.length - 1]; + +async function main() { + if (!file || file.startsWith("-")) { + process.stderr.write("[fake-ffprobe] missing input file\n"); + process.exit(1); + } + let buf; + try { + buf = await readFile(file, "utf8"); + } catch { + process.stderr.write(`[fake-ffprobe] cannot read ${file}\n`); + process.exit(1); + } + const m = buf.match(/__DUR=([0-9]+(?:\.[0-9]+)?)__/); + // No marker -> report a long duration so isShortAudio() stays false. + const seconds = m ? m[1] : "100000"; + process.stdout.write(`${seconds}\n`); +} + +main(); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -70,14 +70,21 @@ async function writeMetadata(videoDir, id, opts = {}) { channel_url: "https://www.youtube.com/channel/UCfake", uploader: "Fake Channel", upload_date: "20240101", - duration: 60, + // Default a short clip; the `longvideo` sentinel makes it long enough + // (>=600s) to exercise the download-time duration guard. + duration: opts.longDuration ? 6000 : 60, description: `Synthetic video ${id}`, is_live: false, was_live: false, live_status: "not_live", age_limit: 0, - extractor_key: "Youtube", - webpage_url: `https://www.youtube.com/watch?v=${id}`, + // The `odyseevid` sentinel makes the metadata advertise the lbry extractor + // so platformFromMetadata() resolves "odysee" (and the "auto" download + // format picks `original`). Otherwise it's a normal YouTube video. + extractor_key: opts.odysee ? "lbry" : "Youtube", + webpage_url: opts.odysee + ? `https://odysee.com/${id}` + : `https://www.youtube.com/watch?v=${id}`, }; // For the no-subs-fallback tests we need yt-dlp's metadata to reflect // whether the video advertises caption tracks. The default (no opts) @@ -116,6 +123,8 @@ function urlSentinels(url) { isLive: lower.includes("islive"), isUpcoming: lower.includes("isupcoming"), wasLive: lower.includes("waslive"), + longDuration: lower.includes("longvideo"), + odysee: lower.includes("odyseevid"), }; } @@ -153,11 +162,16 @@ async function downloadOne(url, opts = {}) { await sleep(500); } } - if (!alreadyHasMetadata) await writeMetadata(videoDir, id); + if (!alreadyHasMetadata) await writeMetadata(videoDir, id, urlSentinels(url)); if (audioFmt) { + // The `truncaud` sentinel simulates a source-truncated download: embed a + // short-duration marker the fake-ffprobe reports, so the duration guard + // trips even though the "download" completed. Unmarked audio reports a long + // duration (no marker), so it passes the guard. + const durMarker = id.toLowerCase().includes("truncaud") ? " __DUR=120__" : ""; await writeFile( path.join(videoDir, `audio.${audioFmt}`), - `fake-ytdlp synthesised audio for ${id}\n`, + `fake-ytdlp synthesised audio for ${id}${durMarker}\n`, ); } else { await writeTranscript(videoDir); @@ -201,6 +215,9 @@ async function modeDownloadFromFile(file, opts = {}) { async function modeDownloadOneUrl(url, opts) { process.stdout.write(`[fake-ytdlp] single-url download ${url}\n`); await appendFile("fake-ytdlp.invocations", `download-one:${url}\n`); + // Record the resolved -f selector so format-selection tests can assert it + // (e.g. Odysee "auto" -> original/bestaudio/worst). + await appendFile("fake-ytdlp.invocations", `format:${arg("-f") ?? ""}\n`); await downloadOne(url, opts); process.stdout.write(`[fake-ytdlp] download complete\n`); } @@ -528,8 +545,10 @@ async function main() { return; } - // Audio-check mode: no -x but has -f bestaudio/worst and -c. - if (has("-c") && arg("-f") === "bestaudio/worst") { + // Audio-check mode: no -x but has -c and a bestaudio-style -f selector. The + // selector may be "bestaudio/worst" or, for Odysee, "original/bestaudio/worst" + // (the platform-aware download-format default), so match on the substring. + if (has("-c") && (arg("-f") ?? "").includes("bestaudio")) { const url = lastNonFlag(); if (!url) { process.stderr.write(`[fake-ytdlp] audio-check mode missing URL\n`); diff --git a/editor/package.json b/editor/package.json @@ -5,8 +5,8 @@ "type": "module", "scripts": { "dev": "next dev --port ${EDITOR_PORT:-3001}", - "dev:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port ${PORT:-3011}", - "start:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port ${PORT:-3011}", + "dev:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port ${PORT:-3011}", + "start:test": "WORKER_TOKEN=test-worker-token TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs FFPROBE_BIN=$(pwd)/e2e/fixtures/bin/fake-ffprobe.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port ${PORT:-3011}", "build": "next build", "start": "next start --port ${EDITOR_PORT:-3001}", "lint": "eslint",