commit 372a9b12c7da281090901063c14e7035a4723e50 parent b413851f5b822047e7c318853f53ab582dfe0107 Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Tue, 30 Jun 2026 10:44:10 -0400 Download-time duration guard: failed-short-audio + shortAudio surfacing Catch source-truncated downloads (yt-dlp exits 0 but the audio is a few minutes of a multi-hour video) BEFORE transcription, instead of only after, via the post-transcription coverage detector. - common/ytdlp/ffprobeDuration.ts: probeMediaDurationSec (first ffprobe use); paths.ffprobeBin (FFPROBE_BIN). - isShortAudio in transcriptCoverage.ts: download-time analogue of isIncompleteTranscript, same 600s/0.5 thresholds; skips livestreams. - downloadOneManaged probes the produced audio.<fmt> after the audio-check / transcribe / fallback paths. On a large shortfall: status "failed-short-audio", recorded shortAudio metrics, NOT transcribed (inline whisper pre-checked too), file KEPT on disk (so it isn't re-downloaded into a loop). Not classified for platform backoff (not a transient error). - channelSnapshot: new shortAudio bucket mirroring corrupt-full-source (push + continue) so the kept stub is excluded from downloadedNoTranscript (no auto-transcribe) and the wrong-format cleanup. - Surfacing: per-video ShortAudioBanner (one-click re-download as Original), channel-list short_audio filter + flag + Select short-audio + bulk re-download, actionable "truncated downloads" section, and a redownloadShortAudioBucketAction reusing the bucket job (ReplayBucket "shortAudio") + the fixIncompleteTranscriptOne re-download path. - autoRunner lands a failed-short-audio auto-download as a failed job. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Diffstat:
20 files changed, 484 insertions(+), 10 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -634,12 +634,19 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { unitStatus = record.status; unitFailureClass = record.failureClass; // Land a failed download as a failed JOB (red row) rather than a silent - // "done"; the runner reads the captured status/class regardless. + // "done"; the runner reads the captured status/class regardless. A + // short-audio download kept its file (so it won't be re-queued) but + // produced no usable audio, so surface it as failed too. if ( record.status === "failed" || - record.status === "failed-corrupt-source" + record.status === "failed-corrupt-source" || + record.status === "failed-short-audio" ) { - throw new Error(`download failed (${record.failureClass ?? "unknown"})`); + throw new Error( + record.status === "failed-short-audio" + ? "download produced truncated audio (short-audio)" + : `download failed (${record.failureClass ?? "unknown"})`, + ); } } finally { task.end(); diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -96,6 +96,15 @@ export type ChannelSnapshot = { // the user can re-download/re-transcribe. Optional: older snapshots lack it; // readers must default to []. incompleteTranscript: string[]; + // Videos whose download COMPLETED (yt-dlp exit 0) but whose audio is far + // shorter than the metadata duration — the source served a truncated stream + // (download-outcome.json "failed-short-audio"). Caught at download time by + // the duration guard BEFORE transcription. The short file is KEPT on disk + // (so it isn't re-downloaded into a loop) and excluded from + // downloadedNoTranscript (so it isn't auto-transcribed). Surfaced so the + // user can re-download with a different format (e.g. Original). Optional: + // older snapshots lack it; readers must default to []. + shortAudio: string[]; }; undownloadedIds: string[]; excludedFromDownload?: ExcludedFromDownload; @@ -362,6 +371,7 @@ export async function generateChannelSnapshot( const nonStandardVtt: string[] = []; const skippedByFilter: string[] = []; const incompleteTranscript: string[] = []; + const shortAudio: string[] = []; let transcribedWithAudioBytes = 0; let multipleAudioFormatsBytes = 0; let foreignAudioBytes = 0; @@ -380,6 +390,15 @@ export async function generateChannelSnapshot( corruptFullSource.push(id); continue; } + // A download whose audio was far shorter than the video (truncated source). + // Like corrupt-full-source: the stub is KEPT on disk but is NOT usable audio, + // so short-circuit it out of downloadedNoTranscript (no auto-transcribe) and + // the wrong-format cleanup (don't delete the kept file). If the user later + // transcribes it anyway, it falls through to the incompleteTranscript path. + if (outcome?.status === "failed-short-audio" && !isVideoTranscribed(files)) { + shortAudio.push(id); + continue; + } if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); // A transcribed video whose cues stop far short of its duration — the audio // download truncated silently. Threshold lives in transcriptCoverage. @@ -556,6 +575,7 @@ export async function generateChannelSnapshot( nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), incompleteTranscript: incompleteTranscript.sort(), + shortAudio: shortAudio.sort(), }, undownloadedIds, excludedFromDownload, diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts @@ -19,7 +19,8 @@ export type ReplayBucket = | "partialDownloads" | "noTranscript" | "downloadedNoTranscript" - | "incompleteTranscript"; + | "incompleteTranscript" + | "shortAudio"; export type JobSpec = { kind: string; @@ -38,6 +39,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([ "noTranscript", "downloadedNoTranscript", "incompleteTranscript", + "shortAudio", ]); // Defensive parse for a spec read back from JSON (a sidecar or the bookmarks diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts @@ -17,6 +17,13 @@ export type DownloadOutcomeStatus = // downloaded container is KEPT on disk for inspection. Distinct from // failed-corrupt-source, which means the download never finished. | "corrupt-full-source" + // A download that COMPLETED (yt-dlp exit 0) but whose audio is far shorter + // than the video's metadata duration — the source served a truncated stream + // (e.g. a CDN-truncated HLS rung). Caught by the download-time duration guard + // BEFORE transcription, so whisper never runs on the stub. The short audio is + // KEPT on disk (so it isn't silently re-downloaded into a loop) and surfaced + // for a manual re-download with a different format. See `shortAudio` below. + | "failed-short-audio" // The app-level filter pass (e.g. skip-live) declined to download this video. // Not a failure and not archived — the next sync/download-missing retries it // once the filter no longer matches (e.g. a live stream becomes a VOD). @@ -30,6 +37,7 @@ export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus "failed", "failed-corrupt-source", "corrupt-full-source", + "failed-short-audio", "skipped-filtered", ]; @@ -78,6 +86,13 @@ export type DownloadOutcomeRecord = { // on success / skipped-filtered. failureClass?: DownloadFailureClass; fellBackToTranscribe?: boolean; + // Set when status is "failed-short-audio": the measured shortfall, so the UI + // can explain it (e.g. "4 min of 114 min") without re-probing. + shortAudio?: { + audioDurationSec: number; + expectedDurationSec: number; + coverage: number; + }; // Set when status is "skipped-filtered": which app-level filter declined the // download and why. Recorded so the UI/log can explain the skip. filter?: { name: string; reason: string }; diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -79,6 +79,10 @@ export type Paths = { whisperBin: string; whisperModel: string; ffmpegBin: string; + // ffprobe binary, used to measure a downloaded audio file's actual duration + // for the download-time short-audio guard (common/ytdlp/ffprobeDuration.ts). + // Ships alongside ffmpeg. + ffprobeBin: string; // rsync binary used to mirror the saved-video store to a backup destination // (Phase 4 of the video-persistence feature). See // common/controller/backupSavedVideos.ts. @@ -158,6 +162,7 @@ export function getPaths(): Paths { "ggml-base.en.bin", ), ffmpegBin: process.env.FFMPEG_BIN ?? "ffmpeg", + ffprobeBin: process.env.FFPROBE_BIN ?? "ffprobe", rsyncBin: process.env.RSYNC_BIN ?? "rsync", parakeetBin: process.env.PARAKEET_STITCH_BIN ?? diff --git a/common/lib/transcriptCoverage.ts b/common/lib/transcriptCoverage.ts @@ -38,6 +38,35 @@ export function transcriptCoverage( return { lastCueEnd, duration, coverage }; } +// The download-time analogue of isIncompleteTranscript: compare the DOWNLOADED +// audio's actual duration to the video's metadata duration. A large shortfall +// means the source served a truncated stream (e.g. a CDN-truncated HLS rung), +// so the audio is short before whisper ever runs. Shares the same thresholds so +// the two guards agree. Returns false (no judgement) for livestreams (unreliable +// durations), short videos, or when either duration is unusable. +export function isShortAudio( + audioDurationSec: number | null, + metaDurationSec: number | null | undefined, + opts?: { isLivestream?: boolean }, +): boolean { + if (opts?.isLivestream) return false; + if ( + typeof metaDurationSec !== "number" || + !Number.isFinite(metaDurationSec) || + metaDurationSec < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC + ) { + return false; + } + if ( + audioDurationSec === null || + !Number.isFinite(audioDurationSec) || + audioDurationSec <= 0 + ) { + return false; + } + return audioDurationSec / metaDurationSec < INCOMPLETE_TRANSCRIPT_MAX_COVERAGE; +} + export function isIncompleteTranscript( cov: TranscriptCoverage, opts?: { isLivestream?: boolean }, diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -33,9 +33,19 @@ import { } from "../lib/downloadOutcome"; import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { recordAvailability } from "../lib/availability-server"; -import { loadRawMetadata, platformFromMetadata } from "../lib/transcripts-server"; +import { + loadRawMetadata, + loadRawMetadataFromDir, + platformFromMetadata, + isLivestreamMetadata, +} from "../lib/transcripts-server"; import { evaluateDownloadFilters } from "../lib/downloadFilters"; import { detectPlatform, type Platform } from "../lib/platform"; +import { probeMediaDurationSec } from "./ffprobeDuration"; +import { + isShortAudio, + INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC, +} from "../lib/transcriptCoverage"; import type { Paths } from "../lib/paths"; import { transcribeWithWorker } from "../controller/transcribeOne"; import { extractVideoId, outputArgsForUrl } from "./runYtdlp"; @@ -457,6 +467,9 @@ async function runManagedDownload( const attempts: DownloadAttempt[] = []; let status: DownloadOutcomeStatus = "failed"; let fellBackToTranscribe = false; + // Set when the download-time duration guard trips: the measured shortfall, + // recorded on the outcome so the UI can explain it without re-probing. + let shortAudioInfo: NonNullable<DownloadOutcomeRecord["shortAudio"]> | undefined; let lastArchiveLine: string | null = null; // Full (untruncated) stderr tail of the most recent attempt, so a failed // download can be classified (rate_limit/network) against everything yt-dlp @@ -770,6 +783,43 @@ async function runManagedDownload( path.join(channelDir, "data", canonicalId ?? "unknown"); const videoId = path.basename(videoDir); + // Download-time duration guard: probe the produced audio.<fmt> and compare its + // actual length to the metadata duration. A large shortfall means the source + // served a truncated stream (e.g. a CDN-truncated HLS rung) even though yt-dlp + // exited 0. Returns the shortfall metrics when tripped, else null. Skips + // livestreams (unreliable durations), short videos, and unmeasurable files + // (probe failure -> null -> no false positive). The file is left on disk; the + // caller marks the download failed-short-audio so it isn't transcribed. + const probeShortAudio = async (): Promise< + NonNullable<DownloadOutcomeRecord["shortAudio"]> | null + > => { + const meta = await loadRawMetadataFromDir(videoDir); + const expected = meta?.duration; + if ( + !meta || + isLivestreamMetadata(meta) || + typeof expected !== "number" || + expected < INCOMPLETE_TRANSCRIPT_MIN_DURATION_SEC + ) { + return null; + } + const audioDurationSec = await probeMediaDurationSec({ + ffprobeBin: opts.paths.ffprobeBin, + file: path.join(videoDir, `audio.${fmt}`), + signal: opts.signal, + onLog: opts.onLog, + }); + if (!isShortAudio(audioDurationSec, expected, { isLivestream: false })) { + return null; + } + const a = audioDurationSec as number; + return { + audioDurationSec: Math.round(a * 10) / 10, + expectedDurationSec: expected, + coverage: Math.round((a / expected) * 1000) / 1000, + }; + }; + // ---------- App-side extraction (transcribe handling) ---------- // After a successful non-audio-check transcribe download in app mode, produce // audio.<fmt> from the downloaded source-media container and keep or discard it @@ -879,7 +929,17 @@ async function runManagedDownload( signal: opts.signal, }); } - if (opts.inlineTranscribeOnFallback) { + const shortBeforeInline = await probeShortAudio(); + if (shortBeforeInline) { + shortAudioInfo = shortBeforeInline; + status = "failed-short-audio"; + lastSucceeded = false; + opts.onLog( + `Short audio (${shortBeforeInline.audioDurationSec}s of ` + + `${shortBeforeInline.expectedDurationSec}s); skipping whisper and ` + + `keeping the file for re-download with a different format.\n`, + ); + } else if (opts.inlineTranscribeOnFallback) { // Inline whisper: matches whisperVideoAction's shape. Routes through // the worker pool so the inline transcription respects worker config // and slot limits like any other. @@ -915,6 +975,32 @@ async function runManagedDownload( } } + // ---------- Duration guard (short-audio) ---------- + // Covers the audio-check and non-inline transcribe paths (the inline-whisper + // path checked before transcribing). A truncated download is marked failed so + // it isn't archived or transcribed; the short file is left on disk so it isn't + // immediately re-downloaded into a loop. + if ( + lastSucceeded && + status !== "failed-short-audio" && + (opts.channelConfig.handling === "transcribe" || fellBackToTranscribe) && + !opts.signal.aborted + ) { + const short = await probeShortAudio(); + if (short) { + shortAudioInfo = short; + status = "failed-short-audio"; + lastSucceeded = false; + opts.onLog( + `Short audio for ${videoId}: ${short.audioDurationSec}s of ` + + `${short.expectedDurationSec}s (${Math.round(short.coverage * 100)}% ` + + `of the video). The source served a truncated stream; keeping the ` + + `file and flagging it. Re-download with a different format ` + + `(e.g. Original) to fix.\n`, + ); + } + } + // ---------- Archive append ---------- if (lastSucceeded && opts.appendArchive !== false && lastArchiveLine) { try { @@ -944,6 +1030,7 @@ async function runManagedDownload( attempts, ...(failureClass ? { failureClass } : {}), ...(fellBackToTranscribe ? { fellBackToTranscribe: true } : {}), + ...(shortAudioInfo ? { shortAudio: shortAudioInfo } : {}), }; // Only write the sidecar if we know which dir to put it in. If the very first // attempt failed before metadata could be written, the data/<id> dir may not diff --git a/common/ytdlp/ffprobeDuration.ts b/common/ytdlp/ffprobeDuration.ts @@ -0,0 +1,48 @@ +import { execa } from "execa"; + +export type ProbeMediaDurationOptions = { + ffprobeBin: string; + file: string; + signal?: AbortSignal; + onLog?: (s: string) => void; +}; + +// Measure a media file's container duration (seconds) via ffprobe. Returns null +// when ffprobe is missing, errors, or prints something unparseable — callers +// treat null as "couldn't measure" and skip the duration guard rather than +// failing a download. Distinct from probeAudioStream (ffmpegStreamProbe), which +// is a pass/fail corruption check and yields no duration. +export async function probeMediaDurationSec( + opts: ProbeMediaDurationOptions, +): Promise<number | null> { + const args = [ + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + opts.file, + ]; + opts.onLog?.(`$ ${opts.ffprobeBin} ${args.join(" ")}\n`); + try { + const result = await execa(opts.ffprobeBin, args, { + cancelSignal: opts.signal, + reject: false, + }); + if (result.exitCode !== 0) { + opts.onLog?.( + `ffprobe exited ${result.exitCode}; skipping duration check.\n`, + ); + return null; + } + const sec = Number.parseFloat(String(result.stdout).trim()); + if (!Number.isFinite(sec) || sec <= 0) return null; + return sec; + } catch (err) { + opts.onLog?.( + `ffprobe failed (${(err as Error).message}); skipping duration check.\n`, + ); + return null; + } +} diff --git a/editor/app/actionable/components/InlineActionButton.tsx b/editor/app/actionable/components/InlineActionButton.tsx @@ -12,6 +12,7 @@ import { import { clearIncompleteTranscriptsAction, redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, } from "../../channels/[slug]/incompleteTranscriptActions"; import { refreshChannelSnapshotAction } from "../../channels/actions"; @@ -20,6 +21,7 @@ type Variant = | { kind: "transcribeMissing"; slug: string; audioFormat?: AudioFormat } | { kind: "redownloadIncomplete"; slug: string } | { kind: "clearIncomplete"; slug: string } + | { kind: "redownloadShortAudio"; slug: string } | { kind: "cleanTranscribedAudio"; slug: string } | { kind: "cleanExtraFormats"; slug: string } | { kind: "refreshReport"; slug: string }; @@ -36,6 +38,7 @@ const LABEL: Record<Variant["kind"], { idle: string; running: string }> = { transcribeMissing: { idle: "Transcribe pending", running: "Queuing…" }, redownloadIncomplete: { idle: "Re-download & re-transcribe", running: "Queuing…" }, clearIncomplete: { idle: "Clear & re-queue", running: "Clearing…" }, + redownloadShortAudio: { idle: "Re-download (corrected format)", running: "Queuing…" }, cleanTranscribedAudio: { idle: "Clean audio", running: "Queuing…" }, cleanExtraFormats: { idle: "Clean extra formats", running: "Queuing…" }, refreshReport: { idle: "Refresh report", running: "Refreshing…" }, @@ -68,6 +71,9 @@ async function runAction(variant: Variant): Promise<StreamActionResult> { if (variant.kind === "redownloadIncomplete") { return redownloadIncompleteBucketAction(variant.slug); } + if (variant.kind === "redownloadShortAudio") { + return redownloadShortAudioBucketAction(variant.slug); + } if (variant.kind === "cleanTranscribedAudio") { return cleanAudioAction(variant.slug); } diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -21,6 +21,7 @@ export type ActionableSummary = { undownloaded: ActionableRow[]; untranscribed: ActionableRow[]; incompleteTranscripts: ActionableRow[]; + shortAudio: ActionableRow[]; cleanTranscribedAudio: ActionableRow[]; cleanExtraFormats: ActionableRow[]; staleOrMissing: ActionableRow[]; @@ -68,6 +69,12 @@ export function actionableIncompleteTranscriptCount(row: ActionableRow): number return row.snapshot?.buckets.incompleteTranscript?.length ?? 0; } +// Downloads the duration guard flagged as truncated at the source (short audio +// kept on disk, not transcribed). Default 0 for snapshots predating the bucket. +export function actionableShortAudioCount(row: ActionableRow): number { + return row.snapshot?.buckets.shortAudio?.length ?? 0; +} + // Cleanup buckets are filtered by "do not clean" at snapshot-generation time, // so the length is the actionable count directly (default undefined → 0 for // snapshots written before the bucket existed). @@ -123,6 +130,10 @@ export async function loadActionableSummary( actionableIncompleteTranscriptCount(a), ); + const shortAudio = rows + .filter((r) => actionableShortAudioCount(r) > 0) + .sort((a, b) => actionableShortAudioCount(b) - actionableShortAudioCount(a)); + const cleanTranscribedAudio = rows .filter((r) => actionableCleanTranscribedCount(r) > 0) .sort( @@ -147,6 +158,7 @@ export async function loadActionableSummary( undownloaded, untranscribed, incompleteTranscripts, + shortAudio, cleanTranscribedAudio, cleanExtraFormats, staleOrMissing, diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -8,6 +8,7 @@ import { actionableCleanTranscribedBytes, actionableCleanTranscribedCount, actionableIncompleteTranscriptCount, + actionableShortAudioCount, actionableUndownloadedCount, actionableUntranscribedCount, loadActionableSummary, @@ -51,6 +52,7 @@ export default async function ActionablePage() { summary.undownloaded.length === 0 && summary.untranscribed.length === 0 && summary.incompleteTranscripts.length === 0 && + summary.shortAudio.length === 0 && summary.cleanTranscribedAudio.length === 0 && summary.cleanExtraFormats.length === 0 && summary.staleOrMissing.length === 0; @@ -126,6 +128,32 @@ export default async function ActionablePage() { }, { config: { + id: "short-audio", + title: "Channels with truncated downloads (short audio)", + description: + "Downloads that completed but whose audio is far shorter than the video — the source served a truncated stream, caught before transcription. The stub is kept (not transcribed). “Re-download (corrected format)” deletes it and re-fetches with the per-source default (Original for Odysee).", + countLabel: "truncated", + emptyLabel: "None detected.", + getCount: actionableShortAudioCount, + primaryAction: (r) => ( + <span className="inline-flex items-center justify-end gap-2 flex-wrap"> + <InlineActionButton + variant={{ kind: "redownloadShortAudio", slug: r.channel.slug }} + /> + <Link + href={`/channels/${r.channel.slug}?filter=short_audio`} + aria-label={`review short-audio downloads for ${r.channel.slug}`} + className="inline-flex items-center px-2.5 py-1 rounded-md border border-zinc-300 dark:border-zinc-700 text-xs font-medium hover:bg-zinc-100 dark:hover:bg-zinc-800 whitespace-nowrap" + > + Review + </Link> + </span> + ), + }, + rows: summary.shortAudio, + }, + { + config: { id: "clean-transcribed-audio", title: "Channels with cleanable transcribed audio", description: diff --git a/editor/app/channels/[slug]/bulkVideoActions.ts b/editor/app/channels/[slug]/bulkVideoActions.ts @@ -18,6 +18,7 @@ import { retryBucketAction } from "./pipelineActions"; import { clearIncompleteTranscriptsAction, redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, } from "./incompleteTranscriptActions"; import { deleteOneVideoDir, @@ -65,6 +66,17 @@ export async function bulkRedownloadIncompleteAction( return redownloadIncompleteBucketAction(slug, videoIds, queueKey); } +// Bulk re-download short-audio (source-truncated) downloads: queues ONE batch +// job that deletes the kept stub and re-fetches each selected video with the +// per-source default format (Original for Odysee), then re-transcribes. +export async function bulkRedownloadShortAudioAction( + slug: string, + videoIds: string[], + queueKey?: string, +): Promise<StreamActionResult> { + return redownloadShortAudioBucketAction(slug, videoIds, queueKey); +} + // Bulk clear truncated transcripts: synchronous fs op (delete audio + transcript) // that enables the auto-runners, so it reports a per-id summary like the other // clear/remove bulk actions. Destructive — the caller confirms first. diff --git a/editor/app/channels/[slug]/components/VideoListPane.tsx b/editor/app/channels/[slug]/components/VideoListPane.tsx @@ -13,6 +13,7 @@ import { bulkDeleteVideoDirsAction, bulkMarkUntranscribableAction, bulkRedownloadIncompleteAction, + bulkRedownloadShortAudioAction, bulkRemoveAudioAction, bulkRemoveWrongFormatAudioAction, bulkRetryDownloadAction, @@ -25,6 +26,7 @@ type BulkAction = | "retry" | "redownload_incomplete" | "clear_incomplete" + | "redownload_short_audio" | "untranscribable" | "clear_failed" | "remove_audio" @@ -36,6 +38,7 @@ const BULK_ACTION_OPTIONS: { value: BulkAction; label: string }[] = [ { value: "retry", label: "Retry download" }, { value: "redownload_incomplete", label: "Re-download & re-transcribe" }, { value: "clear_incomplete", label: "Clear incomplete (audio+transcript)" }, + { value: "redownload_short_audio", label: "Re-download truncated (short audio)" }, { value: "untranscribable", label: "Mark untranscribable" }, { value: "clear_failed", label: "Clear failed markers" }, { value: "remove_audio", label: "Remove audio files" }, @@ -65,6 +68,7 @@ const FILTER_OPTIONS: { value: VideoFilter; label: string }[] = [ { value: "downloaded_no_transcript", label: "No transcript" }, { value: "partial", label: "Partial" }, { value: "incomplete_transcript", label: "Incomplete transcript" }, + { value: "short_audio", label: "Short audio" }, { value: "untranscribable", label: "Untranscribable" }, { value: "running", label: "Running" }, { value: "transcribed", label: "Transcribed" }, @@ -290,6 +294,18 @@ export function VideoListPane({ }); } + // Videos whose download was truncated at the source (short audio kept on disk). + const hasShortAudio = useMemo(() => rows.some((r) => r.shortAudio), [rows]); + function selectShortAudio() { + setSelected((prev) => { + const next = new Set(prev); + for (const r of rows) { + if (r.shortAudio) next.add(r.id); + } + return next; + }); + } + const deleteArmed = deleteConfirm.trim().toLowerCase() === "delete"; const applyDisabled = pending || (action === "delete" && !deleteArmed); @@ -310,6 +326,11 @@ export function VideoListPane({ bulkRedownloadIncompleteAction(s, ids, incompleteQueue), ); break; + case "redownload_short_audio": + doStreamingBulk((s, ids) => + bulkRedownloadShortAudioAction(s, ids, incompleteQueue), + ); + break; case "clear_incomplete": if ( !confirm( @@ -429,6 +450,15 @@ export function VideoListPane({ Select incomplete </button> )} + {hasShortAudio && ( + <button + type="button" + onClick={selectShortAudio} + className="rounded border border-zinc-200 dark:border-zinc-800 px-2 py-0.5 hover:bg-zinc-100 dark:hover:bg-zinc-800" + > + Select short-audio + </button> + )} {visibleRows.length > 0 && ( <> <button @@ -568,7 +598,8 @@ export function VideoListPane({ </label> </> )} - {action === "redownload_incomplete" && ( + {(action === "redownload_incomplete" || + action === "redownload_short_audio") && ( <QueueControl value={incompleteQueue} onChange={setIncompleteQueue} diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts @@ -19,6 +19,7 @@ import { clearIncompleteTranscriptOne, fixIncompleteTranscriptOne, incompleteIdsForChannel, + shortAudioIdsForChannel, } from "./lib/fixIncompleteTranscript"; import type { BulkActionSummary } from "./bulkVideoActions"; @@ -115,6 +116,70 @@ export async function redownloadIncompleteBucketAction( }); } +// Re-download the channel's short-audio bucket (downloads the duration guard +// flagged as truncated at the source). Reuses the same per-video fixer and the +// same bucket job kind, parameterized with bucket "shortAudio" so a bookmarked +// re-run re-derives the live members. The re-download deletes the kept stub and +// re-fetches with the per-source default format (Original for Odysee), which is +// what actually recovers the full audio. +export async function redownloadShortAudioBucketAction( + slug: string, + ids?: string[], + queueKey?: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const config = await readChannelConfig(paths, slug); + if (!config) return { ok: false, error: `Channel "${slug}" not found` }; + const source = ids && ids.length ? ids : await shortAudioIdsForChannel(slug, paths); + const cleaned = dedupeIds(source); + if (cleaned.length === 0) { + return { ok: false, error: "No truncated downloads to fix.", info: true }; + } + return runManagedFunction({ + kind: "redownload-incomplete-bucket", + queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey), + paths, + channelSlug: slug, + spec: { + kind: "redownload-incomplete-bucket", + slug, + bucket: "shortAudio", + params: { queueKey }, + }, + fn: async (onLog, signal, _setProgress, ctx) => { + const tracker = makeTaskTracker(ctx, onLog); + let succeeded = 0; + let failed = 0; + for (const id of cleaned) { + if (ctx.drainSignal?.aborted) { + onLog(`Drain requested; stopping before ${id}.`); + break; + } + try { + onLog(`Re-downloading & re-transcribing ${id}…`); + await fixIncompleteTranscriptOne({ + slug, + videoId: id, + config, + paths, + onLog, + signal, + tracker, + }); + succeeded++; + } catch (e) { + failed++; + onLog(`Failed ${id}: ${(e as Error).message}`); + } + } + onLog( + `Re-download short-audio: ${succeeded} fixed, ${failed} failed of ${cleaned.length}.`, + ); + revalidatePath(`/channels/${slug}`); + }, + }); +} + // Clear & release: synchronously delete the truncated audio + transcript for the // flagged videos so they fall back into the normal pending pipeline, then enable // the auto-runners so they reprocess automatically. Destructive — gate every diff --git a/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts b/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts @@ -38,6 +38,20 @@ export async function incompleteIdsForChannel( return Array.isArray(ids) ? ids : []; } +// The current snapshot's short-audio ids for a channel (downloads the duration +// guard flagged as truncated at the source). Empty when the snapshot is missing +// or lacks the bucket. The same fixIncompleteTranscriptOne re-download path fixes +// these — it deletes the kept stub and re-fetches (now via the per-source +// default, i.e. `original` for Odysee). +export async function shortAudioIdsForChannel( + slug: string, + paths: Paths = getPaths(), +): Promise<string[]> { + const snap = await readChannelSnapshot(paths, slug); + const ids = snap?.buckets?.shortAudio; + return Array.isArray(ids) ? ids : []; +} + // Re-download the full audio and re-transcribe one video in place. The new // transcript overwrites transcript.json (transcribeWithWorker) and normalize // regenerates transcript.cues.json, so there is never a window with no diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -29,6 +29,7 @@ export function normalizeBuckets( nonStandardVtt: raw?.nonStandardVtt ?? [], skippedByFilter: raw?.skippedByFilter ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], + shortAudio: raw?.shortAudio ?? [], }; } diff --git a/editor/app/channels/[slug]/lib/videoRows.ts b/editor/app/channels/[slug]/lib/videoRows.ts @@ -39,6 +39,11 @@ export type VideoRow = { // duration — the audio download truncated silently. Independent flag (not a // `status`) so it composes with `transcribed`. See transcriptCoverage. incompleteTranscript: boolean; + // Download completed but the audio was far shorter than the video — the source + // served a truncated stream (download-outcome "failed-short-audio"). The stub + // is kept on disk; re-download with a different format to fix. Independent flag + // (the video isn't transcribed), mirroring incompleteTranscript. + shortAudio: boolean; running: boolean; status: VideoRowStatus; }; @@ -58,6 +63,7 @@ export type VideoFilter = | "transcribed" | "partial" | "incomplete_transcript" + | "short_audio" | "untranscribable" | "running"; @@ -69,6 +75,7 @@ const VIDEO_FILTERS: readonly VideoFilter[] = [ "transcribed", "partial", "incomplete_transcript", + "short_audio", "untranscribable", "running", ]; @@ -100,6 +107,8 @@ function matchesFilter(r: VideoRow, filter: VideoFilter): boolean { return r.partial; case "incomplete_transcript": return r.incompleteTranscript; + case "short_audio": + return r.shortAudio; case "untranscribable": return r.untranscribable; case "running": diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts @@ -41,6 +41,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { const untranscribable = new Set(buckets.untranscribable); const partial = new Set(buckets.partialDownloads); const incompleteTranscript = new Set(buckets.incompleteTranscript); + const shortAudioSet = new Set(buckets.shortAudio ?? []); const corruptSourceSet = new Set(buckets.corruptSource); const corruptFullSourceSet = new Set(buckets.corruptFullSource); const failedTranscription = new Set(input.failedTranscriptionIds); @@ -111,6 +112,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] { untranscoded.has(id) || multipleAudioFormats.has(id), excluded, incompleteTranscript: incompleteTranscript.has(id), + shortAudio: shortAudioSet.has(id), running: runningIds.has(id), status, }); diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -217,6 +217,15 @@ export function VideoPanel({ existingQueues={existingQueues} /> )} + {downloadOutcome?.status === "failed-short-audio" && ( + <ShortAudioBanner + slug={slug} + videoId={videoId} + shortAudio={downloadOutcome.shortAudio} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + )} <PipelineStageCard id="availability-history" title="Availability history" @@ -1274,6 +1283,8 @@ function DownloadOutcomeBadge({ return "Corrupt source — download couldn't finish"; case "corrupt-full-source": return "Corrupt full source — download completed but audio is malformed (file kept)"; + case "failed-short-audio": + return "Truncated download — audio far shorter than the video (file kept)"; default: return outcome.status; } @@ -1367,6 +1378,69 @@ function IncompleteTranscriptBanner({ ); } +// The download completed but the audio was far shorter than the video — the +// source served a truncated stream (e.g. a CDN-truncated HLS rung). The stub is +// kept on disk; re-downloading with a different format (Original) fetches the +// full audio. Offers a one-click re-download forcing the Original format. +function ShortAudioBanner({ + slug, + videoId, + shortAudio, + defaultQueueKey, + existingQueues, +}: { + slug: string; + videoId: string; + shortAudio?: { + audioDurationSec: number; + expectedDurationSec: number; + coverage: number; + }; + defaultQueueKey: string; + existingQueues: string[]; +}) { + const [queueKey, setQueueKey] = useState(defaultQueueKey); + const actionLabel = `Re-download ${videoId} as Original`; + return ( + <div + role="alert" + aria-label="short audio" + className="flex flex-col gap-3 rounded border px-3 py-2 text-sm border-amber-300 bg-amber-50 text-amber-900 dark:border-amber-900 dark:bg-amber-950 dark:text-amber-200" + > + <div className="flex flex-col gap-1"> + <span className="font-medium">Download was truncated at the source</span> + <span> + {shortAudio + ? `The downloaded audio is only ${formatDuration(shortAudio.audioDurationSec)} of ${formatDuration(shortAudio.expectedDurationSec)} (${Math.round(shortAudio.coverage * 100)}%). ` + : "The downloaded audio is far shorter than the video. "} + The source served a truncated stream for the selected format, so it was + flagged before transcription (the file is kept, not transcribed). + Re-download with the <strong>Original</strong> format to fetch the full + audio. + </span> + </div> + <StreamActionLog + trigger={() => + downloadVideoPipelineAction(slug, videoId, queueKey, "original") + } + cancelAction={cancelJobAction} + buttonLabel="Re-download as Original" + runningLabel="Re-downloading…" + label={actionLabel} + extraControls={ + <QueueControl + value={queueKey} + onChange={setQueueKey} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={actionLabel} + /> + } + /> + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -34,7 +34,10 @@ import { transcribeMissingAction, } from "../channels/[slug]/whisperActions"; import { persistKeptAction } from "../channels/[slug]/persistActions"; -import { redownloadIncompleteBucketAction } from "../channels/[slug]/incompleteTranscriptActions"; +import { + redownloadIncompleteBucketAction, + redownloadShortAudioBucketAction, +} from "../channels/[slug]/incompleteTranscriptActions"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -100,9 +103,13 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." }; const ids = await idsForBucket(spec.slug, spec.bucket); if (ids.length === 0) { - return { ok: false, error: "No incomplete transcripts right now.", info: true }; + return { ok: false, error: "Nothing to re-download right now.", info: true }; } - return redownloadIncompleteBucketAction(spec.slug, ids, queueKey); + // The same job kind drives both the incompleteTranscript and shortAudio + // buckets; route by the captured bucket so the re-run re-stamps the right one. + return spec.bucket === "shortAudio" + ? redownloadShortAudioBucketAction(spec.slug, ids, queueKey) + : redownloadIncompleteBucketAction(spec.slug, ids, queueKey); }, "retry-bucket": async (spec) => { const { p, queueKey } = params(spec);