Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fb6f6cb96eeea02f1100d3e1ed45b3ea2a528d50
parent af80ae1efde681097533f9e5ad1dc641dfbb04d3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 18 May 2026 09:29:26 -0400

retry pipelines

Diffstat:
Mcommon/controller/channelSnapshot.ts | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
Mcommon/controller/whisperBatch.ts | 12+++++++++++-
Mcommon/jobs/streamCommand.ts | 108+++++++++++++++++++++++++++++++++++++++++++++++++++++++------------------------
Mcommon/lib/availability-server.ts | 33+++++++++++++++++++++++++++++++++
Mcommon/lib/availability.ts | 10++++++++++
Mcommon/lib/settings.ts | 9+++++++++
Mcommon/ytdlp/downloadOneManaged.ts | 82+++++++++++++++++++++++++++++++++++++++++++++++++++----------------------------
Mcommon/ytdlp/runYtdlp.ts | 122++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Aeditor/app/api/test/uncaught-count/route.ts | 72++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/channels/[slug]/components/RetryBucketControl.tsx | 89+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx | 45++++++++++++++++++++++++++++++++++++++++++++-
Meditor/app/channels/[slug]/components/stages/DownloadStage.tsx | 124+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Meditor/app/channels/[slug]/components/stages/TranscribeStage.tsx | 24+++++++++++++++++++++++-
Meditor/app/channels/[slug]/lib/stageStatus.ts | 23+++++++++++++++++++----
Meditor/app/channels/[slug]/page.tsx | 21++++++++++++++++++++-
Meditor/app/channels/[slug]/pipelineActions.ts | 42++++++++++++++++++++++++++++++++++++++++--
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 178++++++++++++++++++++++++++++++-------------------------------------------------
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 1-
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 27+++++++++++----------------
Meditor/app/settings/actions.ts | 3+++
Meditor/app/settings/components/SettingsForm.tsx | 20++++++++++++++++++++
Aeditor/e2e/exclude-from-counts.spec.ts | 102+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 55+++++++++++++++++++++++++++++++++++++++++++++++++------
Aeditor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/config.json | 5+++++
Aeditor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/playlist | 3+++
Aeditor/e2e/job-stream-cancel.spec.ts | 74++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/no-subs-fallback.spec.ts | 156+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/retry-bucket.spec.ts | 110+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/undownloaded.spec.ts | 11+++++------
Meditor/e2e/video-page.spec.ts | 79++++++++++++++++++++++++++++++++++++++++++-------------------------------------
Meditor/e2e/whisper-video.spec.ts | 22+++++++---------------
Meditor/e2e/whisper.spec.ts | 29+++++++++++++++++++++++++++++
32 files changed, 1485 insertions(+), 274 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -11,9 +11,13 @@ import { } from "../lib/videoStatus"; import { AVAILABILITY_VALUES, + EXCLUDED_FROM_DOWNLOAD, type Availability, } from "../lib/availability"; -import { loadAvailability } from "../lib/availability-server"; +import { + loadAvailability, + resolveEffectiveAvailability, +} from "../lib/availability-server"; import type { Paths } from "../lib/paths"; import { extractVideoId, isRumbleUrl } from "../ytdlp/runYtdlp"; import { loadFailedTranscriptions } from "./failedTranscriptions"; @@ -24,6 +28,12 @@ export type AvailabilitySnapshot = { unchecked: string[]; }; +export type ExcludedFromDownload = { + membersOnly: string[]; + deleted: string[]; + private: string[]; +}; + export type ChannelSnapshot = { generatedAt: string; totals: { @@ -43,9 +53,24 @@ export type ChannelSnapshot = { duplicateDirs: string[]; }; undownloadedIds: string[]; + excludedFromDownload?: ExcludedFromDownload; availability?: AvailabilitySnapshot; }; +export function emptyExcludedFromDownload(): ExcludedFromDownload { + return { membersOnly: [], deleted: [], private: [] }; +} + +export function normalizeExcludedFromDownload( + raw: Partial<ExcludedFromDownload> | undefined, +): ExcludedFromDownload { + return { + membersOnly: raw?.membersOnly ?? [], + deleted: raw?.deleted ?? [], + private: raw?.private ?? [], + }; +} + function emptyAvailability(): AvailabilitySnapshot { const byStatus = {} as Record<Availability, string[]>; for (const v of AVAILABILITY_VALUES) byStatus[v] = []; @@ -145,7 +170,14 @@ export async function generateChannelSnapshot( } } const availability = await loadAvailability(dir); - return { id, files, webpageUrl, availability }; + const effectiveAvailability = await resolveEffectiveAvailability(dir); + return { + id, + files, + webpageUrl, + availability, + effectiveAvailability, + }; }), ), ); @@ -160,6 +192,22 @@ export async function generateChannelSnapshot( } } + // Computed up-front so the buckets below can suppress IDs that can't be + // acted on (members_only / deleted / private). Surfacing them in + // "Missing metadata" or "Failed transcriptions" creates noise the user + // cannot resolve. + const excludedById = new Map<string, Availability>(); + for (const v of perVideo) { + if ( + v.effectiveAvailability && + (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes( + v.effectiveAvailability, + ) + ) { + excludedById.set(v.id, v.effectiveAvailability); + } + } + const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; const untranscoded: string[] = []; @@ -171,7 +219,7 @@ export async function generateChannelSnapshot( for (const { id, files } of perVideo) { if (isVideoTranscribed(files)) transcribed++; if (isVideoDownloaded(files)) downloaded++; - if (!files.hasMeta) noMetadata.push(id); + if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); if ( targetAudioFile && files.audioFiles.length > 0 && @@ -213,6 +261,16 @@ export async function generateChannelSnapshot( if (!filesById.has(id)) missingFromArchive.push(id); } + const excludedFromDownload = emptyExcludedFromDownload(); + for (const [id, cls] of excludedById) { + if (cls === "members_only") excludedFromDownload.membersOnly.push(id); + else if (cls === "deleted") excludedFromDownload.deleted.push(id); + else if (cls === "private") excludedFromDownload.private.push(id); + } + excludedFromDownload.membersOnly.sort(); + excludedFromDownload.deleted.sort(); + excludedFromDownload.private.sort(); + const undownloadedIds: string[] = []; for (const url of urls) { const slugId = extractVideoId(url); @@ -223,6 +281,7 @@ export async function generateChannelSnapshot( : slugId; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; + if (excludedById.has(dirId)) continue; undownloadedIds.push(dirId); } @@ -240,11 +299,12 @@ export async function generateChannelSnapshot( multipleAudioFormats: multipleAudioFormats.sort(), untranscribable: untranscribable.sort(), noMetadata: noMetadata.sort(), - failedListed, + failedListed: failedListed.filter((id) => !excludedById.has(id)), missingFromArchive: missingFromArchive.sort(), duplicateDirs: [], }, undownloadedIds, + excludedFromDownload, availability, }; diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts @@ -3,6 +3,7 @@ import fs from "fs-extra"; import pLimit from "p-limit"; import type { Paths } from "../lib/paths"; import type { AudioFormat } from "../lib/channelConfig"; +import { VTT_FILENAME, WHISPER_FILENAME } from "../lib/videoStatus"; import { transcribeOneVideo } from "./transcribeOne"; import { resolveShardItems } from "./shard"; @@ -90,7 +91,16 @@ export async function runWhisperBatch({ skipped++; return; } - if (await pathExists(path.join(videoPath, "transcript.json"))) { + // A transcript already exists if either whisper has run + // (transcript.json) OR yt-dlp wrote auto/manual English subs + // (transcript.en.vtt). Without the VTT branch, videos on a + // youtube-handling channel with only VTT auto-subs end up + // attempted -> fail with "no audio file found" -> get added to + // failed-transcriptions on every run. + if ( + (await pathExists(path.join(videoPath, WHISPER_FILENAME))) || + (await pathExists(path.join(videoPath, VTT_FILENAME))) + ) { log(`Transcription for ${videoDir} already exists`); skipped++; return; diff --git a/common/jobs/streamCommand.ts b/common/jobs/streamCommand.ts @@ -58,6 +58,51 @@ function makeJob( return { id, logPath, record }; } +// Safety contract shared by both stream creators. The consumer (RSC encoder) +// can close the stream at any time when the client disconnects; once closed, +// any further enqueue/close on the controller throws ERR_INVALID_STATE. The +// throw escapes our local try/catch because it surfaces inside the RSC +// encoder's own pipeline, not at our call site. Tracking a `closed` flag and +// gating all controller ops on it ensures we stop producing the moment the +// downstream goes away. Returns helpers plus a `markClosed()` for the +// cancel() / finally() paths. +function makeSafeController() { + let controller: ReadableStreamDefaultController<string> | null = null; + let closed = false; + const setController = (c: ReadableStreamDefaultController<string>) => { + controller = c; + }; + const safeEnqueue = (text: string) => { + if (closed) return; + try { + controller?.enqueue(text); + } catch (err) { + if ( + (err as NodeJS.ErrnoException | undefined)?.code === "ERR_INVALID_STATE" + ) { + closed = true; + } + } + }; + const safeClose = () => { + if (closed) return; + closed = true; + try { + controller?.close(); + } catch {} + }; + const markClosed = () => { + closed = true; + }; + return { setController, safeEnqueue, safeClose, markClosed }; +} + +// Best-effort error listener for the log WriteStream: writes can fail (disk +// full, permissions, etc.) and would otherwise become uncaughtExceptions. +function ignoreFileStreamErrors(stream: WriteStream): void { + stream.on("error", () => {}); +} + export async function runManagedCommand( opts: RunManagedCommandOpts, ): Promise<StreamActionResult> { @@ -71,27 +116,32 @@ export async function runManagedCommand( opts.videoId, ); - let controller: ReadableStreamDefaultController<string> | null = null; + const safe = makeSafeController(); let fileStream: WriteStream | null = null; let cancelledBeforeStart = false; const stream = new ReadableStream<string>({ start(c) { - controller = c; + safe.setController(c); }, cancel() { - // Client disconnected. Job continues to write to its log file. + // Consumer disconnected (page navigation, tab close, RSC response + // tear-down). Stop pushing into the controller — but keep the child + // and the on-disk log going so the job runs to completion and can be + // observed by another reconnecting client via /api/jobs/<id>/log. + // Explicit user cancels go through `registry.cancel()` (which kills + // the child), not through this source-cancel hook. + safe.markClosed(); }, }); const start = () => { if (cancelledBeforeStart) { - try { - controller?.close(); - } catch {} + safe.safeClose(); return; } fileStream = createWriteStream(logPath); + ignoreFileStreamErrors(fileStream); const child = execa(opts.command, opts.args, { cwd: opts.cwd, env: opts.env, @@ -103,9 +153,7 @@ export async function runManagedCommand( child.all?.on("data", (chunk: Buffer) => { fileStream!.write(chunk); - try { - controller?.enqueue(chunk.toString("utf8")); - } catch {} + safe.safeEnqueue(chunk.toString("utf8")); }); child.all?.on("error", () => {}); @@ -123,24 +171,18 @@ export async function runManagedCommand( registry.finalize(id, "cancelled"); return; } - try { - controller?.enqueue(`\n[error] ${(err as Error).message}\n`); - } catch {} + safe.safeEnqueue(`\n[error] ${(err as Error).message}\n`); registry.finalize(id, "failed"); }) .finally(() => { fileStream?.end(); - try { - controller?.close(); - } catch {} + safe.safeClose(); }); }; const onCancel = () => { cancelledBeforeStart = true; - try { - controller?.close(); - } catch {} + safe.safeClose(); }; registry.enqueue(record, { start, onCancel }); @@ -160,34 +202,38 @@ export async function runManagedFunction( opts.videoId, ); - let controller: ReadableStreamDefaultController<string> | null = null; + const safe = makeSafeController(); let fileStream: WriteStream | null = null; let cancelledBeforeStart = false; const stream = new ReadableStream<string>({ start(c) { - controller = c; + safe.setController(c); + }, + cancel() { + // Consumer disconnected (page navigation, tab close, RSC response + // tear-down). Stop pushing into the controller — but let the function + // run to completion, writing to disk; reconnecting clients can poll + // /api/jobs/<id>/log. Explicit user cancels go through + // `registry.cancel()` which aborts this controller. + safe.markClosed(); }, - cancel() {}, }); const start = () => { if (cancelledBeforeStart) { - try { - controller?.close(); - } catch {} + safe.safeClose(); return; } fileStream = createWriteStream(logPath); + ignoreFileStreamErrors(fileStream); const abort = new AbortController(); record.abortController = abort; const onLog = (line: string) => { const text = line.endsWith("\n") ? line : `${line}\n`; fileStream!.write(text); - try { - controller?.enqueue(text); - } catch {} + safe.safeEnqueue(text); }; opts @@ -209,17 +255,13 @@ export async function runManagedFunction( }) .finally(() => { fileStream?.end(); - try { - controller?.close(); - } catch {} + safe.safeClose(); }); }; const onCancel = () => { cancelledBeforeStart = true; - try { - controller?.close(); - } catch {} + safe.safeClose(); }; registry.enqueue(record, { start, onCancel }); diff --git a/common/lib/availability-server.ts b/common/lib/availability-server.ts @@ -3,8 +3,13 @@ import { readFile, rename, writeFile } from "node:fs/promises"; import { AVAILABILITY_FILENAME, AVAILABILITY_VALUES, + type Availability, type AvailabilityRecord, } from "./availability"; +import { + DOWNLOAD_OUTCOME_FILENAME, + type DownloadOutcomeRecord, +} from "./downloadOutcome"; export function availabilityPath(videoDir: string): string { return path.join(videoDir, AVAILABILITY_FILENAME); @@ -38,3 +43,31 @@ export async function writeAvailability( await writeFile(tmp, JSON.stringify(record, null, 2) + "\n"); await rename(tmp, file); } + +// Resolve a video's availability class, preferring an explicit availability.json +// (the authoritative record written by the availability-check or metadata-backfill +// passes) and falling back to the last attempt's availabilityClass in +// download-outcome.json. This second source matters for videos that have +// never successfully downloaded — their availability never reaches +// availability.json otherwise. +export async function resolveEffectiveAvailability( + videoDir: string, +): Promise<Availability | null> { + const explicit = await loadAvailability(videoDir); + if (explicit) return explicit.availability; + try { + const raw = await readFile( + path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME), + "utf8", + ); + const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>; + const attempts = parsed.attempts; + if (Array.isArray(attempts) && attempts.length > 0) { + const last = attempts[attempts.length - 1]; + if (last?.availabilityClass) return last.availabilityClass; + } + } catch { + // No outcome file or unreadable → unknown + } + return null; +} diff --git a/common/lib/availability.ts b/common/lib/availability.ts @@ -21,6 +21,16 @@ export const AVAILABILITY_VALUES: Availability[] = [ "error", ]; +// Availability classes treated as permanently blocking: videos in these +// states are filtered out of undownloadedIds and skipped by the managed +// download flows. needs_auth is intentionally excluded (auth-retry may +// recover it); error is excluded too (often transient: rate limits, network). +export const EXCLUDED_FROM_DOWNLOAD: ReadonlyArray<Availability> = [ + "members_only", + "deleted", + "private", +]; + export type AvailabilityRecord = { checkedAt: string; availability: Availability; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -19,6 +19,11 @@ export type SiteSettings = { // within one invocation, so without this the managed loop hammers the // source IP back-to-back. 0 disables. Per-channel override available. sleepBetweenDownloadsSeconds: number; + // When true, the no-subs fallback in the managed downloader runs whisper + // inline immediately after the audio download succeeds. When false + // (default), audio is left for the next "Transcribe missing" pass so a + // batch download finishes faster and whisper can parallelize. + inlineTranscribeOnFallback: boolean; }; export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600; @@ -61,6 +66,7 @@ function defaults(): SiteSettings { transcribeModel: paths.whisperModel, cookiesFromBrowser: "", sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, + inlineTranscribeOnFallback: false, }; } @@ -101,6 +107,9 @@ export function getSettings(): SiteSettings { merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds( merged.sleepBetweenDownloadsSeconds, ); + if (typeof merged.inlineTranscribeOnFallback !== "boolean") { + merged.inlineTranscribeOnFallback = false; + } return merged; } diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -37,6 +37,11 @@ export type ManagedDownloadOpts = { // When false, suppress appending to the channel archive file. Mirrors // `ignoreArchive` from the playlist-level callers. appendArchive?: boolean; + // When true, run whisper inline immediately after the no-subs fallback's + // audio download succeeds. When false (default), leave the audio for the + // next "Transcribe missing" pass. Resolved from the site setting by the + // playlist-level caller. + inlineTranscribeOnFallback?: boolean; }; const OUTPUT_ARGS: string[] = [ @@ -199,11 +204,16 @@ async function resolveVideoIdFromUrl( async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> { const entries = await readdir(videoDir).catch(() => [] as string[]); - return entries.some( - (e) => - e === "transcript.json" || - /^transcript\.[^.]+\.(?:vtt|json|json3|srv1|srv2|srv3)$/.test(e), - ); + return entries.some((e) => { + if (e === "transcript.json") return true; + const m = e.match( + /^transcript\.([^.]+)\.(?:vtt|json|json3|srv1|srv2|srv3)$/, + ); + if (!m) return false; + // live_chat is YouTube's chat replay, not a transcript of speech — don't + // let its presence suppress the no-subs fallback to audio + whisper. + return m[1] !== "live_chat"; + }); } async function metadataReportsNoCaptions( @@ -229,12 +239,13 @@ async function metadataReportsNoCaptions( } const subs = parsed.subtitles ?? {}; const auto = parsed.automatic_captions ?? {}; - return ( - typeof subs === "object" && - typeof auto === "object" && - Object.keys(subs).length === 0 && - Object.keys(auto).length === 0 - ); + if (typeof subs !== "object" || typeof auto !== "object") return false; + // live_chat is YouTube's chat replay, not a caption track. yt-dlp places + // it under `subtitles`, so a video whose only listed track is live_chat + // should be treated as having no captions for fallback purposes. + const realSubs = Object.keys(subs).filter((k) => k !== "live_chat"); + const realAuto = Object.keys(auto).filter((k) => k !== "live_chat"); + return realSubs.length === 0 && realAuto.length === 0; } function trimError(stderrTail: string): string | undefined { @@ -429,6 +440,10 @@ export async function downloadOneManaged( status === "ok-with-cookies" ? opts.globalCookiesFromBrowser : undefined; + // Feed yt-dlp the metadata it already wrote during the primary + // attempt instead of re-querying the extractor — saves a network + // round-trip per video, which adds up across batch runs and helps + // dodge rate limits we'd otherwise burn on info we already have. const fallbackArgs = [ "--ignore-config", "--restrict-filenames", @@ -437,8 +452,8 @@ export async function downloadOneManaged( "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, ...channelConfigArgs(fallbackConfig, fallbackCookieOverride), - "--", - opts.videoUrl, + "--load-info-json", + path.join(videoDir, "metadata.info.json"), ]; const fallbackRes = await runOneYtdlp(opts, channelDir, fallbackArgs); const fallbackAvail = attemptSucceeded(fallbackRes.exitCode) @@ -458,24 +473,33 @@ export async function downloadOneManaged( if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine; if (attemptSucceeded(fallbackRes.exitCode)) { - // Inline whisper: matches whisperVideoAction's shape. - try { - await transcribeOneVideo({ - paths: opts.paths, - videoDir, - videoId, - audioFilename: `audio.${audioFmt}`, - onLog: opts.onLog, - signal: opts.signal, - }); - fellBackToTranscribe = true; - status = "ok-auto-transcribed"; - } catch (err) { + fellBackToTranscribe = true; + if (opts.inlineTranscribeOnFallback) { + // Inline whisper: matches whisperVideoAction's shape. + try { + await transcribeOneVideo({ + paths: opts.paths, + videoDir, + videoId, + audioFilename: `audio.${audioFmt}`, + onLog: opts.onLog, + signal: opts.signal, + }); + status = "ok-auto-transcribed"; + } catch (err) { + opts.onLog( + `Whisper failed after no-subs fallback: ${(err as Error).message}\n`, + ); + status = "failed"; + lastSucceeded = false; + } + } else { opts.onLog( - `Whisper failed after no-subs fallback: ${(err as Error).message}\n`, + `Audio downloaded for ${videoId}; skipping inline whisper (inlineTranscribeOnFallback=off). Run "Transcribe missing" to transcribe.\n`, ); - status = "failed"; - lastSucceeded = false; + // status remains whatever the primary attempt set; the + // fellBackToTranscribe flag distinguishes from a normal "ok" + // so the UI and downstream jobs can tell what happened. } } else { status = "failed"; diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -18,6 +18,11 @@ import { } from "../lib/channelConfig"; import { getSettings } from "../lib/settings"; import type { Paths } from "../lib/paths"; +import { + EXCLUDED_FROM_DOWNLOAD, + type Availability, +} from "../lib/availability"; +import { resolveEffectiveAvailability } from "../lib/availability-server"; import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability"; import { resolveShardItems } from "../controller/shard"; import { downloadOneManaged } from "./downloadOneManaged"; @@ -28,6 +33,7 @@ export type YtdlpMode = | "download-missing" | "download-missing-subs" | "download-one-audio" + | "retry-bucket" | "sync"; export type RunYtdlpOpts = { @@ -54,6 +60,14 @@ export type RunYtdlpOpts = { audioFormatOverride?: AudioFormat; // download-one-audio only: appended after configArgs, before the URL. extraYtdlpArgs?: string[]; + // retry-bucket only: video IDs to limit the run to (matched against the + // saved playlist by extracted ID). IDs not present in the playlist are + // reported and skipped. + bucketIds?: ReadonlyArray<string>; + // retry-bucket only: override channelConfig.handling for this run without + // mutating the channel config on disk. Useful for retrying old "youtube" + // videos as "transcribe". + handlingOverride?: ChannelHandling; }; export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> { @@ -76,12 +90,27 @@ export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> { case "download-one-audio": await downloadOneAudio(opts); return; + case "retry-bucket": + await retryBucket(opts); + return; case "sync": await sync(opts); return; } } +async function retryBucket(opts: RunYtdlpOpts): Promise<void> { + if (!opts.bucketIds || opts.bucketIds.length === 0) { + opts.onLog("retry-bucket: no video IDs supplied. Nothing to do.\n"); + return; + } + await downloadPlaylistManaged(opts, { + prefilter: "destination-exists", + idAllowList: opts.bucketIds, + handlingOverride: opts.handlingOverride, + }); +} + function channelRoot(opts: RunYtdlpOpts): string { return path.join(opts.paths.channelsDir, opts.channelSlug); } @@ -203,8 +232,20 @@ type PrefilterMode = "archive" | "destination-exists"; async function downloadPlaylistManaged( opts: RunYtdlpOpts, - { prefilter }: { prefilter: PrefilterMode }, + cfg: { + prefilter: PrefilterMode; + idAllowList?: ReadonlyArray<string>; + handlingOverride?: ChannelHandling; + }, ): Promise<void> { + const { prefilter, idAllowList, handlingOverride } = cfg; + const effectiveHandling: ChannelHandling = + handlingOverride ?? opts.channelConfig.handling; + const effectiveChannelConfig: ChannelConfig = + handlingOverride && handlingOverride !== opts.channelConfig.handling + ? { ...opts.channelConfig, handling: handlingOverride } + : opts.channelConfig; + const root = channelRoot(opts); const dataDir = path.join(root, "data"); const playlistPath = path.join(root, "playlist"); @@ -218,7 +259,7 @@ async function downloadPlaylistManaged( `No saved playlist at ${playlistPath}. Run "Store playlist" first.`, ); } - const urls = playlistText + let urls = playlistText .split("\n") .map((s) => s.trim()) .filter(Boolean); @@ -227,6 +268,38 @@ async function downloadPlaylistManaged( ? await buildRumbleSlugIndex(dataDir) : null; + if (idAllowList) { + const allow = new Set(idAllowList); + const before = urls.length; + const matched: string[] = []; + const matchedIds = new Set<string>(); + for (const url of urls) { + const slugId = extractVideoId(url); + if (!slugId) continue; + const dirId = + rumbleIndex && isRumbleUrl(url) + ? (rumbleIndex.get(slugId) ?? slugId) + : slugId; + if (allow.has(dirId)) { + matched.push(url); + matchedIds.add(dirId); + } + } + const missing = [...allow].filter((id) => !matchedIds.has(id)); + opts.onLog( + `Retry allowlist: kept ${matched.length} of ${before} playlist URLs (${missing.length} requested IDs not in playlist).\n`, + ); + if (missing.length > 0) { + opts.onLog(` Missing: ${missing.join(", ")}\n`); + } + urls = matched; + } + if (handlingOverride) { + opts.onLog( + `Handling override: using ${handlingOverride} for this run (channel config unchanged).\n`, + ); + } + const tofetch: string[] = []; if (prefilter === "archive") { // Skip URLs whose ID is already recorded in the channel's archive file. @@ -264,7 +337,7 @@ async function downloadPlaylistManaged( rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; if ( dirId && - (await destinationExists(dataDir, dirId, opts.channelConfig.handling)) + (await destinationExists(dataDir, dirId, effectiveHandling)) ) { alreadyComplete++; continue; @@ -278,6 +351,45 @@ async function downloadPlaylistManaged( ); } + // Exclude videos whose effective availability marks them as permanently + // unavailable (members_only, deleted, private). Applies to both prefilter + // modes: failed download attempts don't write to the archive, so the + // archive-prefilter path would also keep retrying them. + const excludedCounts = { members_only: 0, deleted: 0, private: 0 }; + const filteredTofetch: string[] = []; + for (const url of tofetch) { + const slugId = extractVideoId(url); + const dirId = slugId + ? rumbleIndex && isRumbleUrl(url) + ? (rumbleIndex.get(slugId) ?? slugId) + : slugId + : null; + if (dirId) { + const cls = await resolveEffectiveAvailability( + path.join(dataDir, dirId), + ); + if ( + cls && + (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes(cls) + ) { + excludedCounts[cls as keyof typeof excludedCounts]++; + continue; + } + } + filteredTofetch.push(url); + } + const totalExcluded = + excludedCounts.members_only + + excludedCounts.deleted + + excludedCounts.private; + if (totalExcluded > 0) { + opts.onLog( + `Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`, + ); + } + tofetch.length = 0; + tofetch.push(...filteredTofetch); + // Sharding is only meaningful for the missing-files mode (matches prior // behavior where `download-missing` accepted shardTotal / shardIndex). const items = @@ -303,6 +415,7 @@ async function downloadPlaylistManaged( const settings = getSettings(); const globalCookies = settings.cookiesFromBrowser; + const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback; const sleepSeconds = opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds; @@ -329,13 +442,14 @@ async function downloadPlaylistManaged( if (firstFailure && abortOnError) return; const outcome = await downloadOneManaged({ channelSlug: opts.channelSlug, - channelConfig: opts.channelConfig, + channelConfig: effectiveChannelConfig, paths: opts.paths, videoUrl: url, onLog: opts.onLog, signal: opts.signal, globalCookiesFromBrowser: globalCookies || undefined, appendArchive: !opts.ignoreArchive, + inlineTranscribeOnFallback, }); if (outcome.status === "failed") { failedCount++; diff --git a/editor/app/api/test/uncaught-count/route.ts b/editor/app/api/test/uncaught-count/route.ts @@ -0,0 +1,72 @@ +import { NextResponse } from "next/server"; + +export const dynamic = "force-dynamic"; + +// E2E test harness only. Counts uncaughtException + unhandledRejection +// events seen by the dev-server's Node process so a spec can assert that +// "Controller is already closed" no longer leaks out of the streaming job +// pipeline when the consumer disconnects. +// +// The process-level handler is installed once on first module load. It's +// gated by the test-fixture TRANSCRIPTS_DIR path so production-shaped +// installs never see it. + +type CountState = { + uncaught: number; + unhandled: number; + messages: string[]; +}; + +declare global { + // eslint-disable-next-line no-var + var __yttUncaughtCount__: CountState | undefined; +} + +function isTestEnv(): boolean { + const dir = process.env.TRANSCRIPTS_DIR ?? ""; + return dir.endsWith("test-transcripts"); +} + +function getState(): CountState { + if (!globalThis.__yttUncaughtCount__) { + globalThis.__yttUncaughtCount__ = { + uncaught: 0, + unhandled: 0, + messages: [], + }; + if (isTestEnv()) { + process.on("uncaughtException", (err) => { + const s = globalThis.__yttUncaughtCount__!; + s.uncaught++; + s.messages.push(`uncaught: ${(err as Error)?.message ?? String(err)}`); + if (s.messages.length > 50) s.messages.shift(); + }); + process.on("unhandledRejection", (reason) => { + const s = globalThis.__yttUncaughtCount__!; + s.unhandled++; + s.messages.push( + `unhandled: ${(reason as Error)?.message ?? String(reason)}`, + ); + if (s.messages.length > 50) s.messages.shift(); + }); + } + } + return globalThis.__yttUncaughtCount__; +} + +export async function GET() { + const state = getState(); + return NextResponse.json({ + uncaught: state.uncaught, + unhandled: state.unhandled, + messages: state.messages.slice(), + }); +} + +export async function DELETE() { + const state = getState(); + state.uncaught = 0; + state.unhandled = 0; + state.messages.length = 0; + return NextResponse.json({ ok: true }); +} diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx @@ -0,0 +1,89 @@ +"use client"; + +import { useState } from "react"; +import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import { HANDLING_VALUES } from "yt-dlp-transcript-common/lib/channelConfig"; +import { QueueControl } from "../../../components/QueueControl"; +import { cancelJobAction } from "../../../jobs/actions"; +import { retryBucketAction } from "../pipelineActions"; + +type Props = { + slug: string; + ids: string[]; + actionLabel: string; + defaultQueueKey: string; + existingQueues: string[]; +}; + +export function RetryBucketControl({ + slug, + ids, + actionLabel, + defaultQueueKey, + existingQueues, +}: Props) { + const [queue, setQueue] = useState(defaultQueueKey); + const [handlingOverride, setHandlingOverride] = useState(""); + const [abortOnError, setAbortOnError] = useState(false); + + if (ids.length === 0) return null; + + return ( + <div + className="flex flex-col gap-2" + aria-label={`retry ${actionLabel} bucket`} + > + <StreamActionLog + trigger={() => + retryBucketAction( + slug, + ids, + queue, + abortOnError, + handlingOverride || undefined, + ) + } + cancelAction={cancelJobAction} + buttonLabel={`Retry (${ids.length})`} + runningLabel="Retrying…" + label={`Retry ${actionLabel}`} + extraControls={ + <> + <label className="flex items-center gap-1 text-xs text-zinc-500"> + <span>Retry as</span> + <select + value={handlingOverride} + onChange={(e) => setHandlingOverride(e.target.value)} + aria-label={`handling override for retry ${actionLabel}`} + className="font-mono px-2 py-1 rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 text-zinc-900 dark:text-zinc-100" + > + <option value="">channel default</option> + {HANDLING_VALUES.map((h) => ( + <option key={h} value={h}> + {h} + </option> + ))} + </select> + </label> + <QueueControl + value={queue} + onChange={setQueue} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel={`Retry ${actionLabel}`} + /> + <label className="flex items-center gap-1 text-xs text-zinc-500"> + <input + type="checkbox" + checked={abortOnError} + onChange={(e) => setAbortOnError(e.target.checked)} + aria-label={`abort on error for retry ${actionLabel}`} + /> + Abort on error + </label> + </> + } + /> + </div> + ); +} diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx @@ -17,6 +17,7 @@ import { parseShardField, type ShardConfigSummary, } from "../../../../components/ShardControl"; +import { RetryBucketControl } from "../RetryBucketControl"; import { VideoIdList } from "../VideoIdList"; function parseConcurrency(s: string): number | undefined { @@ -32,6 +33,7 @@ type DiagnosticBucket = { label: string; description: string; ariaLabel: string; + retry?: boolean; }; type Props = { @@ -43,6 +45,8 @@ type Props = { totals: { videos: number; transcribed: number; downloaded: number }; availability: AvailabilitySnapshot; availabilityShard: ShardConfigSummary | null; + existingQueues: string[]; + downloadDefaultQueueKey: string; }; export function DiagnosticsStage({ @@ -54,6 +58,8 @@ export function DiagnosticsStage({ totals, availability, availabilityShard, + existingQueues, + downloadDefaultQueueKey, }: Props) { const buckets: DiagnosticBucket[] = [ { @@ -62,6 +68,7 @@ export function DiagnosticsStage({ description: "Video directory exists without yt-dlp metadata; index will skip it.", ariaLabel: "missing metadata", + retry: true, }, { ids: missingFromArchiveIds, @@ -69,6 +76,7 @@ export function DiagnosticsStage({ description: "Video ID is in the archive but its data directory is missing.", ariaLabel: "archived without dir", + retry: true, }, { ids: duplicateDirIds, @@ -127,7 +135,12 @@ export function DiagnosticsStage({ desc="Probe yt-dlp to detect videos that have been deleted, made private, or restricted by the uploader. Results write to channels/<slug>/data/<id>/availability.json." /> <AvailabilityButtons slug={slug} shard={availabilityShard} /> - <AvailabilitySummary slug={slug} availability={availability} /> + <AvailabilitySummary + slug={slug} + availability={availability} + existingQueues={existingQueues} + downloadDefaultQueueKey={downloadDefaultQueueKey} + /> </div> <div className="flex flex-col gap-2" aria-label="channel health"> <div> @@ -166,6 +179,15 @@ export function DiagnosticsStage({ emptyMessage="None" itemAriaLabel={(id) => `${b.ariaLabel} ${id}`} /> + {b.retry ? ( + <RetryBucketControl + slug={slug} + ids={b.ids} + actionLabel={b.ariaLabel} + defaultQueueKey={downloadDefaultQueueKey} + existingQueues={existingQueues} + /> + ) : null} </div> ))} </div> @@ -390,12 +412,24 @@ const AVAILABILITY_LIST_STATUSES: Availability[] = [ "error", ]; +// Statuses where a retry-download is meaningful. needs_auth and error are the +// only availability buckets not in EXCLUDED_FROM_DOWNLOAD (see availability.ts); +// the others would be silent no-ops because the downloader filters them. +const AVAILABILITY_RETRY_STATUSES: ReadonlyArray<Availability> = [ + "needs_auth", + "error", +]; + function AvailabilitySummary({ slug, availability, + existingQueues, + downloadDefaultQueueKey, }: { slug: string; availability: AvailabilitySnapshot; + existingQueues: string[]; + downloadDefaultQueueKey: string; }) { const total = AVAILABILITY_VALUES.reduce( @@ -462,6 +496,15 @@ function AvailabilitySummary({ emptyMessage="None" itemAriaLabel={(id) => `availability ${v} ${id}`} /> + {AVAILABILITY_RETRY_STATUSES.includes(v) && ( + <RetryBucketControl + slug={slug} + ids={availability.byStatus[v]} + actionLabel={AVAILABILITY_LABELS[v].toLowerCase()} + defaultQueueKey={downloadDefaultQueueKey} + existingQueues={existingQueues} + /> + )} </div> ))} </div> diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx @@ -2,6 +2,7 @@ import { useState } from "react"; import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import type { ExcludedFromDownload } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { QueueControl } from "../../../../components/QueueControl"; import { ShardControl, @@ -14,6 +15,7 @@ import { downloadMissingAction, downloadMissingSubsAction, } from "../../pipelineActions"; +import { RetryBucketControl } from "../RetryBucketControl"; import { VideoIdList } from "../VideoIdList"; type Props = { @@ -22,6 +24,7 @@ type Props = { defaultQueueKey: string; existingQueues: string[]; undownloadedIds: string[]; + excludedFromDownload: ExcludedFromDownload; noTranscriptIds: string[]; missingShard: ShardConfigSummary | null; }; @@ -32,6 +35,7 @@ export function DownloadStage({ defaultQueueKey, existingQueues, undownloadedIds, + excludedFromDownload, noTranscriptIds, missingShard, }: Props) { @@ -59,6 +63,10 @@ export function DownloadStage({ return ( <div className="flex flex-col gap-6"> <UndownloadedList slug={slug} ids={undownloadedIds} /> + <ExcludedFromDownloadCard + slug={slug} + excluded={excludedFromDownload} + /> <div className="flex flex-col gap-2"> <Heading title="Download from playlist" @@ -194,7 +202,102 @@ export function DownloadStage({ } /> </div> - <NoTranscriptList slug={slug} ids={noTranscriptIds} /> + <NoTranscriptList + slug={slug} + ids={noTranscriptIds} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> + </div> + ); +} + +function ExcludedFromDownloadCard({ + slug, + excluded, +}: { + slug: string; + excluded: ExcludedFromDownload; +}) { + const total = + excluded.membersOnly.length + + excluded.deleted.length + + excluded.private.length; + if (total === 0) return null; + type Section = { + key: "membersOnly" | "deleted" | "private"; + label: string; + ariaPrefix: string; + ids: string[]; + }; + const allSections: Section[] = [ + { + key: "membersOnly", + label: "Members-only", + ariaPrefix: "members only", + ids: excluded.membersOnly, + }, + { + key: "deleted", + label: "Deleted", + ariaPrefix: "deleted", + ids: excluded.deleted, + }, + { + key: "private", + label: "Private", + ariaPrefix: "private", + ids: excluded.private, + }, + ]; + const sections = allSections.filter((s) => s.ids.length > 0); + return ( + <div + aria-label="excluded from download" + className="flex flex-col gap-2 rounded border border-zinc-200 dark:border-zinc-800 p-3" + > + <div> + <h4 className="text-sm font-semibold"> + Excluded from download ({total}) + </h4> + <p className="text-xs text-zinc-500"> + Permanently-unavailable videos (members-only, deleted, or private) + are skipped by the download flows. To retry one after it becomes + public again, re-check its availability in Diagnostics. + </p> + </div> + <div className="flex flex-wrap gap-x-3 gap-y-1 text-xs text-zinc-600 dark:text-zinc-400"> + {sections.map((s) => ( + <span key={s.key} aria-label={`excluded ${s.ariaPrefix} count`}> + <span className="font-medium text-zinc-800 dark:text-zinc-200"> + {s.label} + </span>{" "} + {s.ids.length} + </span> + ))} + </div> + <div className="grid grid-cols-1 lg:grid-cols-2 gap-3"> + {sections.map((s) => ( + <details + key={s.key} + className="rounded border border-zinc-200 dark:border-zinc-800 p-2" + > + <summary className="cursor-pointer text-xs font-semibold"> + {s.label} ({s.ids.length}) + </summary> + <div className="mt-2"> + <VideoIdList + slug={slug} + ids={s.ids} + ariaLabel={`excluded ${s.ariaPrefix} list`} + emptyAriaLabel={`excluded ${s.ariaPrefix} empty`} + emptyMessage="None" + itemAriaLabel={(id) => `excluded ${s.ariaPrefix} ${id}`} + /> + </div> + </details> + ))} + </div> </div> ); } @@ -217,7 +320,17 @@ function UndownloadedList({ slug, ids }: { slug: string; ids: string[] }) { ); } -function NoTranscriptList({ slug, ids }: { slug: string; ids: string[] }) { +function NoTranscriptList({ + slug, + ids, + defaultQueueKey, + existingQueues, +}: { + slug: string; + ids: string[]; + defaultQueueKey: string; + existingQueues: string[]; +}) { if (ids.length === 0) return null; return ( <div className="flex flex-col gap-2 rounded border border-zinc-200 dark:border-zinc-800 p-3"> @@ -237,6 +350,13 @@ function NoTranscriptList({ slug, ids }: { slug: string; ids: string[] }) { emptyMessage="None" itemAriaLabel={(id) => `no transcript or download ${id}`} /> + <RetryBucketControl + slug={slug} + ids={ids} + actionLabel="missing transcript and no download" + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + /> </div> ); } diff --git a/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx b/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx @@ -18,6 +18,7 @@ import { clearFailedTranscriptionsAction, transcribeMissingAction, } from "../../whisperActions"; +import { RetryBucketControl } from "../RetryBucketControl"; import { VideoIdList } from "../VideoIdList"; type AudioFormatChoice = AudioFormat | "any"; @@ -37,6 +38,11 @@ type Props = { downloadedNoTranscriptIds: string[]; defaultConcurrency: number; defaultQueueKey: string; + // Default queue used when the user clicks Retry on the + // downloadedNoTranscript bucket. The retry triggers a download flow, so + // it should run on the platform's download queue (not the transcription + // queue) to avoid hammering the source. + downloadDefaultQueueKey: string; missingShard: ShardConfigSummary | null; }; @@ -47,6 +53,7 @@ export function TranscribeStage({ downloadedNoTranscriptIds, defaultConcurrency, defaultQueueKey, + downloadDefaultQueueKey, missingShard, }: Props) { const channelQueueKey = `channel:${slug}`; @@ -133,6 +140,8 @@ export function TranscribeStage({ <DownloadedNoTranscriptList slug={slug} ids={downloadedNoTranscriptIds} + downloadDefaultQueueKey={downloadDefaultQueueKey} + existingQueues={existingQueues} /> </div> <FailedTranscriptionsSection @@ -224,9 +233,13 @@ function FailedTranscriptionsSection({ function DownloadedNoTranscriptList({ slug, ids, + downloadDefaultQueueKey, + existingQueues, }: { slug: string; ids: string[]; + downloadDefaultQueueKey: string; + existingQueues: string[]; }) { if (ids.length === 0) return null; return ( @@ -236,7 +249,9 @@ function DownloadedNoTranscriptList({ Downloaded but not transcribed ({ids.length}) </h4> <p className="text-xs text-zinc-500"> - Audio is on disk but no .vtt or transcript.json yet. + Audio is on disk but no .vtt or transcript.json yet. The retry runs + a download pass — pick &ldquo;youtube&rdquo; handling to fetch a + missing VTT, or leave default to verify nothing else is needed. </p> </div> <VideoIdList @@ -247,6 +262,13 @@ function DownloadedNoTranscriptList({ emptyMessage="None" itemAriaLabel={(id) => `downloaded without transcript ${id}`} /> + <RetryBucketControl + slug={slug} + ids={ids} + actionLabel="downloaded but not transcribed" + defaultQueueKey={downloadDefaultQueueKey} + existingQueues={existingQueues} + /> </div> ); } diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -1,5 +1,8 @@ import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; -import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; +import { + normalizeExcludedFromDownload, + type ChannelSnapshot, +} from "yt-dlp-transcript-common/controller/channelSnapshot"; import type { JobRecord } from "yt-dlp-transcript-common/jobs/registry"; export type SnapshotBuckets = ChannelSnapshot["buckets"]; @@ -95,6 +98,17 @@ export function computeStageStatuses( const buckets = normalizeBuckets(snapshot.buckets); const undownloadedIds = snapshot.undownloadedIds ?? []; + const excludedFromDownload = normalizeExcludedFromDownload( + snapshot.excludedFromDownload, + ); + const excludedDownloadIds = new Set<string>([ + ...excludedFromDownload.membersOnly, + ...excludedFromDownload.deleted, + ...excludedFromDownload.private, + ]); + const actionableNoTranscript = buckets.noTranscript.filter( + (id) => !excludedDownloadIds.has(id), + ); const runningByStage = new Set<StageId>(); for (const job of runningJobs) { @@ -106,7 +120,8 @@ export function computeStageStatuses( const transcodeApplies = config.handling === "transcribe" && !!config.audioFormat; - const downloadPending = undownloadedIds.length + buckets.noTranscript.length; + const downloadPending = + undownloadedIds.length + actionableNoTranscript.length; const transcodePending = transcodeApplies ? buckets.untranscoded.length : 0; const transcodeFailed = transcodeApplies ? failedTranscodingIds.length : 0; const transcribePending = buckets.downloadedNoTranscript.length; @@ -157,10 +172,10 @@ export function computeStageStatuses( if (undownloadedIds.length > 0) { downloadParts.push(pluralize(undownloadedIds.length, "undownloaded")); } - if (buckets.noTranscript.length > 0) { + if (actionableNoTranscript.length > 0) { downloadParts.push( pluralize( - buckets.noTranscript.length, + actionableNoTranscript.length, "dir missing transcript & audio", "dirs missing transcript & audio", ), diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -5,6 +5,7 @@ import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels" import { generateChannelSnapshot, normalizeAvailability, + normalizeExcludedFromDownload, readChannelSnapshot, } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { loadFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; @@ -85,7 +86,7 @@ export default async function ChannelDetailPage({ // The retry-failures panel is interactive (a click mutates the file), so // read it fresh on every render — the snapshot bucket only reflects state // at refresh time. - const failedVideoIds = await loadFailedTranscriptions(paths, slug); + const rawFailedVideoIds = await loadFailedTranscriptions(paths, slug); const failedTranscodingIds = await loadFailedTranscodings(paths, slug); const summarize = (c: ShardConfig | null) => @@ -106,6 +107,20 @@ export default async function ChannelDetailPage({ const buckets = normalizeBuckets(snapshot.buckets); const undownloadedIds = snapshot.undownloadedIds ?? []; + const excludedFromDownload = normalizeExcludedFromDownload( + snapshot.excludedFromDownload, + ); + // Hide failed-transcription entries whose video can no longer be acted on + // (members_only / deleted / private). Same rationale as the snapshot's + // bucket filter; this path is fresh-loaded so it needs its own pass. + const excludedDownloadIds = new Set<string>([ + ...excludedFromDownload.membersOnly, + ...excludedFromDownload.deleted, + ...excludedFromDownload.private, + ]); + const failedVideoIds = rawFailedVideoIds.filter( + (id) => !excludedDownloadIds.has(id), + ); const availability = normalizeAvailability(snapshot.availability); const stages = computeStageStatuses({ snapshot, @@ -156,6 +171,7 @@ export default async function ChannelDetailPage({ defaultQueueKey={platformDefaultQueueKey} existingQueues={existingQueues} undownloadedIds={undownloadedIds} + excludedFromDownload={excludedFromDownload} noTranscriptIds={buckets.noTranscript} missingShard={downloadMissingShard} /> @@ -168,6 +184,7 @@ export default async function ChannelDetailPage({ downloadedNoTranscriptIds={buckets.downloadedNoTranscript} defaultConcurrency={paths.parallelTranscribeLimit} defaultQueueKey={TRANSCRIPTION_QUEUE} + downloadDefaultQueueKey={platformDefaultQueueKey} missingShard={transcribeMissingShard} /> ), @@ -189,6 +206,8 @@ export default async function ChannelDetailPage({ totals={snapshot.totals} availability={availability} availabilityShard={availabilityShard} + existingQueues={existingQueues} + downloadDefaultQueueKey={platformDefaultQueueKey} /> ), danger: ( diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -1,7 +1,11 @@ "use server"; import { revalidatePath } from "next/cache"; -import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; +import { + HANDLING_VALUES, + type ChannelConfig, + type ChannelHandling, +} from "yt-dlp-transcript-common/lib/channelConfig"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { detectPlatform, @@ -25,7 +29,8 @@ async function runPipelineAction( | "download-from-playlist" | "sync" | "download-missing" - | "download-missing-subs", + | "download-missing-subs" + | "retry-bucket", kind: string, queueKey?: string, options?: { @@ -33,6 +38,8 @@ async function runPipelineAction( abortOnError?: boolean; shardTotal?: number; shardIndex?: number; + bucketIds?: ReadonlyArray<string>; + handlingOverride?: ChannelHandling; }, ): Promise<StreamActionResult> { const paths = getPaths(); @@ -61,6 +68,8 @@ async function runPipelineAction( abortOnError: options?.abortOnError, shardTotal: options?.shardTotal, shardIndex: options?.shardIndex, + bucketIds: options?.bucketIds, + handlingOverride: options?.handlingOverride, }); revalidatePath(`/channels/${slug}`); revalidatePath("/channels"); @@ -126,3 +135,32 @@ export async function downloadMissingSubsAction( { abortOnError }, ); } + +export async function retryBucketAction( + slug: string, + bucketIds: string[], + queueKey?: string, + abortOnError?: boolean, + handlingOverride?: string, +): Promise<StreamActionResult> { + if (!Array.isArray(bucketIds) || bucketIds.length === 0) { + return { ok: false, error: "No video IDs supplied for retry." }; + } + let handling: ChannelHandling | undefined; + if (handlingOverride) { + if (!(HANDLING_VALUES as ReadonlyArray<string>).includes(handlingOverride)) { + return { + ok: false, + error: `Invalid handling override: ${handlingOverride}`, + }; + } + handling = handlingOverride as ChannelHandling; + } + return runPipelineAction( + slug, + "retry-bucket", + "retry-bucket", + queueKey, + { bucketIds, handlingOverride: handling, abortOnError }, + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -16,8 +16,8 @@ import { TextFilePreview } from "./TextFilePreview"; import { deleteVideoDirAction, deleteVideoFileAction, + downloadVideoPipelineAction, markVideoUntranscribableAction, - redownloadVideoAction, transcodeAudioAction, transcribeOneAction, whisperVideoAction, @@ -35,7 +35,6 @@ type Props = { videoId: string; files: VideoFile[]; handling: ChannelHandling; - audioFormat: AudioFormat; defaultQueueKey: string; existingQueues: string[]; downloadOutcome: DownloadOutcomeRecord | null; @@ -72,6 +71,17 @@ function isAudioFile(name: string): boolean { return AUDIO_EXTS.has(fileExt(name)); } +// Files we can transcode FROM: any media file ffmpeg can demux to audio. +// Includes video containers so a kept source (e.g. audio.mp4 when +// keepSourceVideo=true) can be re-extracted to audio.<targetFormat> to +// overwrite a truncated or corrupt prior extraction. +function isTranscodeSource(name: string): boolean { + if (!name.startsWith("audio.")) return false; + if (name.startsWith("audio.tmp-")) return false; + const ext = fileExt(name); + return AUDIO_EXTS.has(ext) || VIDEO_EXTS.has(ext); +} + function audioExt(name: string): string { const dot = name.lastIndexOf("."); return dot >= 0 ? name.slice(dot + 1) : ""; @@ -96,24 +106,17 @@ function formatSize(bytes: number): string { return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`; } -function parseExtraArgs(raw: string): string[] { - return raw - .split(/\s+/) - .map((s) => s.trim()) - .filter(Boolean); -} - export function VideoPanel({ slug, videoId, files, handling, - audioFormat, defaultQueueKey, existingQueues, downloadOutcome, }: Props) { const audioFiles = files.filter((f) => isAudioFile(f.name)); + const transcodeSources = files.filter((f) => isTranscodeSource(f.name)); const hasTranscriptJson = files.some((f) => f.name === "transcript.json"); const hasYtVtt = files.some((f) => f.name === "transcript.en.vtt"); const hasTranscript = hasTranscriptJson || hasYtVtt; @@ -121,17 +124,17 @@ export function VideoPanel({ const downloadSummary = noAudio ? handling === "youtube" - ? "No audio on disk — download to enable whisper." + ? "No audio on disk — run the pipeline (VTT first, whisper if needed)." : "Audio not yet downloaded." : `${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} on disk.`; const transcribeSummary = hasTranscript ? "Transcript present." : noAudio - ? "No audio yet — download first or use Whisper transcribe (combo)." + ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript." : "Audio ready, no transcript."; const transcodeSummary = - audioFiles.length > 0 - ? `Convert ${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} to another format.` + transcodeSources.length > 0 + ? `Convert ${transcodeSources.length} source file${transcodeSources.length === 1 ? "" : "s"} to another format.` : "Nothing to transcode."; return ( @@ -148,14 +151,13 @@ export function VideoPanel({ slug={slug} videoId={videoId} handling={handling} - defaultFormat={audioFormat} defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} noAudio={noAudio} /> </PipelineStageCard> - {audioFiles.length > 0 && ( + {transcodeSources.length > 0 && ( <PipelineStageCard id="transcode" title="Transcode" @@ -164,7 +166,7 @@ export function VideoPanel({ tone="neutral" > <div className="flex flex-col gap-4"> - {audioFiles.map((f) => ( + {transcodeSources.map((f) => ( <PerFileTranscodeRow key={f.name} slug={slug} @@ -185,14 +187,6 @@ export function VideoPanel({ tone={hasTranscript ? "ok" : noAudio ? "neutral" : "attention"} > <div className="flex flex-col gap-4"> - {!hasTranscript && ( - <WhisperVideoSection - slug={slug} - videoId={videoId} - existingQueues={existingQueues} - noAudio={noAudio} - /> - )} {audioFiles.map((f) => ( <PerFileTranscribeRow key={f.name} @@ -241,11 +235,12 @@ export function VideoPanel({ ); } +type DownloadMode = "pipeline" | "whisper"; + function RedownloadSection({ slug, videoId, handling, - defaultFormat, defaultQueueKey, existingQueues, noAudio, @@ -253,62 +248,62 @@ function RedownloadSection({ slug: string; videoId: string; handling: ChannelHandling; - defaultFormat: AudioFormat; defaultQueueKey: string; existingQueues: string[]; noAudio: boolean; }) { - const [format, setFormat] = useState<AudioFormat>(defaultFormat); - const [extraArgsRaw, setExtraArgsRaw] = useState(""); + const [mode, setMode] = useState<DownloadMode>("pipeline"); const [queueKey, setQueueKey] = useState(defaultQueueKey); - const desc = noAudio - ? handling === "youtube" - ? "YouTube channels normally only fetch auto-subs. Use this when the video has no auto-subs and you want to transcribe it manually." - : "Fetch the audio for this video using yt-dlp." - : "Re-fetch the audio for this video. The existing audio file is kept on disk; yt-dlp may overwrite it with the new one if names collide."; - const actionLabel = `${noAudio ? "Download" : "Redownload"} audio for ${videoId}`; + + const desc = (() => { + if (mode === "whisper") { + return noAudio + ? "Download the audio and run whisper-cli, in one job. Skips any VTT attempt." + : "Run whisper-cli on this video's audio. Skips the download phase since audio is already on disk."; + } + if (handling === "youtube") { + return noAudio + ? "Try the VTT download first; fall back to audio + whisper if no subs are available." + : "Re-run the pipeline: try VTT, then fall back to audio + whisper if needed."; + } + return noAudio + ? "Download audio (and run any channel-configured audio checks)." + : "Re-run the download pipeline for this video."; + })(); + + const verb = mode === "whisper" + ? "Audio + Whisper" + : noAudio + ? "Run download pipeline" + : "Re-run download pipeline"; + const runningLabel = mode === "whisper" + ? "Transcribing…" + : "Running pipeline…"; + const actionLabel = `${verb} for ${videoId}`; + const trigger = mode === "whisper" + ? () => whisperVideoAction(slug, videoId, queueKey) + : () => downloadVideoPipelineAction(slug, videoId, queueKey); + return ( <div className="flex flex-col gap-3"> <p className="text-sm text-zinc-500">{desc}</p> - <div className="flex flex-col sm:flex-row sm:flex-wrap sm:items-center gap-3"> - <label className="flex items-center gap-2 text-sm"> - Format - <select - value={format} - onChange={(e) => setFormat(e.target.value as AudioFormat)} - aria-label="redownload audio format" - className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" - > - {AUDIO_FORMAT_VALUES.map((f) => ( - <option key={f} value={f}> - {f} - </option> - ))} - </select> - </label> - <label className="flex flex-col sm:flex-row sm:items-center gap-2 text-sm flex-1 sm:min-w-[16rem]"> - <span>Extra yt-dlp args</span> - <input - type="text" - value={extraArgsRaw} - onChange={(e) => setExtraArgsRaw(e.target.value)} - aria-label="redownload extra yt-dlp args" - placeholder="--cookies-from-browser firefox --quiet" - className="flex-1 rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono min-w-0" - /> - </label> - </div> + <label className="flex items-center gap-2 text-sm"> + Mode + <select + value={mode} + onChange={(e) => setMode(e.target.value as DownloadMode)} + aria-label="download mode" + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + <option value="pipeline">Designated pipeline</option> + <option value="whisper">Audio + Whisper (skip pipeline)</option> + </select> + </label> <StreamActionLog - trigger={() => - redownloadVideoAction(slug, videoId, { - audioFormat: format, - extraArgs: parseExtraArgs(extraArgsRaw), - queueKey, - }) - } + trigger={trigger} cancelAction={cancelJobAction} - buttonLabel={`${noAudio ? "Download" : "Redownload"} audio (${format})`} - runningLabel={noAudio ? "Downloading…" : "Redownloading…"} + buttonLabel={verb} + runningLabel={runningLabel} label={actionLabel} extraControls={ <QueueControl @@ -324,45 +319,6 @@ function RedownloadSection({ ); } -function WhisperVideoSection({ - slug, - videoId, - existingQueues, - noAudio, -}: { - slug: string; - videoId: string; - existingQueues: string[]; - noAudio: boolean; -}) { - const [queueKey, setQueueKey] = useState(""); - const desc = noAudio - ? "Download the audio and run whisper-cli, in one job. Useful when a YouTube video has no auto-subs." - : "Run whisper-cli on this video's audio. Skips the download phase since audio is already on disk."; - const actionLabel = `Whisper transcribe ${videoId}`; - return ( - <div className="flex flex-col gap-2"> - <Heading title="Whisper transcribe" desc={desc} /> - <StreamActionLog - trigger={() => whisperVideoAction(slug, videoId, queueKey)} - cancelAction={cancelJobAction} - buttonLabel="Whisper transcribe" - runningLabel="Transcribing…" - label={actionLabel} - extraControls={ - <QueueControl - value={queueKey} - onChange={setQueueKey} - defaultQueueKey="" - existingQueues={existingQueues} - actionLabel={actionLabel} - /> - } - /> - </div> - ); -} - function PerFileTranscribeRow({ slug, videoId, diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -167,7 +167,6 @@ export default async function VideoDetailPage({ videoId={id} files={dirData.files} handling={config.handling} - audioFormat={config.audioFormat ?? "mp3"} defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} downloadOutcome={downloadOutcome} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -19,6 +19,8 @@ import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/f import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode"; import { transcribeOneVideo } from "yt-dlp-transcript-common/controller/transcribeOne"; import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos"; +import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { runManagedFunction, @@ -111,19 +113,11 @@ export async function transcribeOneAction( }); } -export async function redownloadVideoAction( +export async function downloadVideoPipelineAction( slug: string, videoId: string, - options: { - audioFormat?: AudioFormat; - extraArgs?: string[]; - queueKey?: string; - } = {}, + queueKey?: string, ): Promise<StreamActionResult> { - const { audioFormat, extraArgs, queueKey } = options; - if (audioFormat && !AUDIO_FORMAT_VALUES.includes(audioFormat)) { - return { ok: false, error: `Unsupported audio format: ${audioFormat}` }; - } const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); @@ -135,23 +129,24 @@ export async function redownloadVideoAction( "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", }; } + const settings = getSettings(); return runManagedFunction({ - kind: "download-one-audio", + kind: "download-one-pipeline", queueKey: videoQueueKey(r.config, queueKey), paths, channelSlug: slug, videoId, fn: async (onLog, signal) => { - await runYtdlp({ + await downloadOneManaged({ channelSlug: slug, - mode: "download-one-audio", channelConfig: r.config, paths, + videoUrl: url, onLog, signal, - singleVideoUrl: url, - audioFormatOverride: audioFormat, - extraYtdlpArgs: extraArgs, + globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined, + inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, + appendArchive: true, }); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -29,6 +29,8 @@ export async function saveSettingsAction( const sleepRaw = String( formData.get("sleepBetweenDownloadsSeconds") ?? "", ).trim(); + const inlineTranscribeOnFallback = + formData.get("inlineTranscribeOnFallback") === "on"; const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? ""); const transcribeArgs = transcribeArgsRaw .split("\n") @@ -86,6 +88,7 @@ export async function saveSettingsAction( transcribeArgs, cookiesFromBrowser, sleepBetweenDownloadsSeconds: sleepParsed, + inlineTranscribeOnFallback, }; await writeSettings(next); revalidatePath("/settings"); diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -100,6 +100,26 @@ export function SettingsForm({ initial }: Props) { type="number" hint="Pause inserted between per-video yt-dlp invocations in managed batch downloads (download-from-playlist, download-missing). 0 disables. Default 10s. Each channel can override this in its Advanced settings." /> + <label className="flex items-start gap-2 text-sm"> + <input + type="checkbox" + name="inlineTranscribeOnFallback" + defaultChecked={initial.inlineTranscribeOnFallback} + className="mt-1" + /> + <span className="flex flex-col gap-1"> + <span className="font-medium"> + Transcribe immediately after no-subs fallback + </span> + <span className="text-xs text-zinc-500"> + When the managed downloader falls back to audio+whisper for a + video without captions, run whisper inline instead of leaving + the audio for the next &ldquo;Transcribe missing&rdquo; pass. + Off by default so batch downloads finish faster and whisper + can run in parallel. + </span> + </span> + </label> <div className="flex items-center gap-3"> <button type="submit" diff --git a/editor/e2e/exclude-from-counts.spec.ts b/editor/e2e/exclude-from-counts.spec.ts @@ -0,0 +1,102 @@ +import { mkdir, writeFile } from "node:fs/promises"; +import { dirname } from "node:path"; +import { test, expect } from "@playwright/test"; +import { resetData, resolvePath } from "./helpers"; + +const CHANNEL = "test-transcribe"; +const FIXTURE = "one-transcribe-channel-with-audio"; + +type AvailabilityStatus = + | "public" + | "unlisted" + | "private" + | "members_only" + | "needs_auth" + | "deleted" + | "error"; + +async function writeAvailability(id: string, status: AvailabilityStatus) { + const path = resolvePath( + `test-transcripts/channels/${CHANNEL}/data/${id}/availability.json`, + ); + await mkdir(dirname(path), { recursive: true }); + await writeFile( + path, + JSON.stringify({ + checkedAt: new Date().toISOString(), + availability: status, + webpageUrl: `https://www.youtube.com/watch?v=${id}`, + }), + ); +} + +test("members-only video is hidden from the Missing metadata bucket", async ({ + page, +}) => { + // Fixture has vidA, vidB, vidC — each with audio.m4a but no metadata.info.json. + // Without filtering, all three would show as "Missing metadata.info.json (3)". + await resetData(FIXTURE); + await writeAvailability("vidA", "members_only"); + + await page.goto(`/channels/${CHANNEL}`); + + await expect( + page.getByRole("heading", { name: /Missing metadata\.info\.json \(2\)/ }), + ).toBeVisible(); + + const list = page.getByLabel("missing metadata list"); + await expect(list.getByLabel("missing metadata vidA")).toHaveCount(0); + await expect(list.getByLabel("missing metadata vidB")).toBeVisible(); + await expect(list.getByLabel("missing metadata vidC")).toBeVisible(); +}); + +test("members-only video is hidden from the Failed transcriptions list", async ({ + page, +}) => { + await resetData(FIXTURE); + await writeAvailability("vidA", "members_only"); + await writeFile( + resolvePath(`test-transcripts/channels/${CHANNEL}/failed-transcriptions`), + "vidA\nvidB\n", + ); + + await page.goto(`/channels/${CHANNEL}`); + + await expect( + page.getByRole("heading", { name: /Failed transcriptions \(1\)/ }), + ).toBeVisible(); + + const list = page.getByLabel("failed transcriptions list"); + await expect(list.getByLabel("failed transcription vidA")).toHaveCount(0); + await expect(list.getByLabel("failed transcription vidB")).toBeVisible(); +}); + +test("stage badge counts exclude members-only videos", async ({ page }) => { + await resetData(FIXTURE); + // vidA: members_only — should drop out of both buckets. + // vidB: stays in both buckets (no availability sidecar). + await writeAvailability("vidA", "members_only"); + await writeFile( + resolvePath(`test-transcripts/channels/${CHANNEL}/failed-transcriptions`), + "vidA\nvidB\n", + ); + + await page.goto(`/channels/${CHANNEL}`); + + // Diagnostics pending = noMetadata (2: vidB, vidC) + missingFromArchive + duplicateDirs. + // The fixture has no archive or duplicates, so the count is just the + // filtered noMetadata count. Both the side-rail and the stage card surface + // this label; assert every instance reads "2". + const diagPending = page.getByLabel("Diagnostics pending count"); + await expect(diagPending).not.toHaveCount(0); + for (const el of await diagPending.all()) { + await expect(el).toHaveText("2"); + } + + // Transcribe failed = filtered failedVideoIds count (1: vidB). + const transcribeFailed = page.getByLabel("Transcribe failed count"); + await expect(transcribeFailed).not.toHaveCount(0); + for (const el of await transcribeFailed.all()) { + await expect(el).toHaveText(/^1( failed)?$/); + } +}); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -57,7 +57,7 @@ async function ensureDir(p) { await mkdir(p, { recursive: true }); } -async function writeMetadata(videoDir, id) { +async function writeMetadata(videoDir, id, opts = {}) { const meta = { id, title: `Synthetic ${id}`, @@ -75,6 +75,16 @@ async function writeMetadata(videoDir, id) { extractor_key: "Youtube", webpage_url: `https://www.youtube.com/watch?v=${id}`, }; + // For the no-subs-fallback tests we need yt-dlp's metadata to reflect + // whether the video advertises caption tracks. The default (no opts) + // omits both keys so the rest of the existing suite keeps its behavior. + if (opts.livechatOnly) { + meta.subtitles = { live_chat: [{ ext: "json", url: "fake://chat" }] }; + meta.automatic_captions = {}; + } else if (opts.noCaptions) { + meta.subtitles = {}; + meta.automatic_captions = {}; + } await writeFile( path.join(videoDir, "metadata.info.json"), JSON.stringify(meta), @@ -305,10 +315,29 @@ async function modeYoutubeSingleUrlManaged(url) { `youtube-single:${url}\n`, ); process.stdout.write(`[fake-ytdlp] managed single-URL ${id}\n`); + + // URL sentinels for no-subs-fallback tests. The sentinels live in the + // video id itself so they round-trip through urlIdYouTube and the + // managed downloader's URL re-extraction. + const lower = url.toLowerCase(); + const livechatOnly = lower.includes("livechatonly"); + const noCaptions = lower.includes("nocaptions"); + if (!existsSync(path.join(videoDir, "metadata.info.json"))) { - await writeMetadata(videoDir, id); + await writeMetadata(videoDir, id, { livechatOnly, noCaptions }); + } + if (livechatOnly) { + // yt-dlp writes live_chat as JSON-lines (one continuation per line). + // Content shape doesn't matter for these tests; just produce a file. + await writeFile( + path.join(videoDir, "transcript.live_chat.json"), + '{"clientId":"fake","action":{"addChatItemAction":{}}}\n', + ); + } else if (noCaptions) { + // Simulate "metadata-only" success — no transcript file at all. + } else { + await writeTranscript(videoDir); } - await writeTranscript(videoDir); process.stdout.write(`DLOM_ARCHIVE youtube ${id}\n`); process.stdout.write(`[fake-ytdlp] single-url fetched ${id}\n`); } @@ -401,10 +430,24 @@ async function main() { return; } - // Single-URL invocation (download-one-audio): no -a, no playlist flags, - // last positional is the URL. + // Single-URL invocation (download-one-audio or no-subs fallback): no + // -a, no playlist flags. The fallback path passes --load-info-json + // instead of a positional URL — derive the id from the info.json path + // (our layout is `data/<id>/metadata.info.json`), so the rest of the + // mode stays URL-shaped. if (audioFmt) { - const url = lastNonFlag(); + const infoJsonPath = arg("--load-info-json"); + let url; + if (infoJsonPath) { + const id = path.basename(path.dirname(infoJsonPath)); + url = `https://www.youtube.com/watch?v=${id}`; + await appendFile( + "fake-ytdlp.invocations", + `load-info-json:${infoJsonPath}\n`, + ); + } else { + url = lastNonFlag(); + } if (!url) { process.stderr.write(`[fake-ytdlp] single-url mode missing URL\n`); process.exit(2); diff --git a/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/config.json b/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/config.json @@ -0,0 +1,5 @@ +{ + "handling": "youtube", + "name": "Test Live Chat", + "url": "https://www.youtube.com/@example/videos" +} diff --git a/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/playlist b/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/playlist @@ -0,0 +1,3 @@ +https://www.youtube.com/watch?v=livechatonly1 +https://www.youtube.com/watch?v=nocaptions001 +https://www.youtube.com/watch?v=normalvideo1 diff --git a/editor/e2e/job-stream-cancel.spec.ts b/editor/e2e/job-stream-cancel.spec.ts @@ -0,0 +1,74 @@ +// Regression: when the consumer disconnects from a streaming server action +// (e.g. user navigates away mid-job), the producer kept calling +// controller.enqueue on the closed stream and uncaughtException leaked out +// of the RSC encoder with "Controller is already closed". streamCommand.ts +// now nullifies the controller and aborts the producer on cancel; this spec +// guards against the regression. + +import { test, expect } from "@playwright/test"; +import { resetData } from "./helpers"; + +const baseUrl = "http://localhost:3011"; + +async function readUncaughtCount(): Promise<{ + uncaught: number; + unhandled: number; + messages: string[]; +}> { + const res = await fetch(`${baseUrl}/api/test/uncaught-count`); + return res.json() as Promise<{ + uncaught: number; + unhandled: number; + messages: string[]; + }>; +} + +async function clearUncaughtCount(): Promise<void> { + await fetch(`${baseUrl}/api/test/uncaught-count`, { method: "DELETE" }); +} + +test("disconnecting from a running job stream does not crash with 'Controller is already closed'", async ({ + page, +}) => { + test.setTimeout(60_000); + await resetData("slow-pipeline-channel"); + await clearUncaughtCount(); + + // Start the slow sync — fake-ytdlp sleeps 30s before producing output, so + // the producer is alive and pushing into the stream for the duration. + await page.goto("/channels/slow-channel"); + await page.getByRole("button", { name: "Sync" }).click(); + await expect(page.getByLabel("Sync output")).toContainText("test-slow", { + timeout: 15_000, + }); + + // Simulate consumer disconnect by navigating to a different page. The + // RSC response stream attached to the previous page is torn down, which + // cancels our underlying ReadableStream source. The job itself keeps + // running on the server (writes to disk) — that's intentional, since + // another client could reconnect. + await page.goto("/jobs"); + + // Cancel the still-running job via the jobs page so the test doesn't + // wait the full 30s for fake-ytdlp to wake up. The actual regression + // signal is the uncaught-counter check after this — which catches the + // "Controller is already closed" error that the prior implementation + // produced whenever the source kept enqueueing post-disconnect. + const jobRow = page + .getByRole("row") + .filter({ hasText: "slow-channel" }) + .first(); + await expect(jobRow).toContainText("running", { timeout: 10_000 }); + await jobRow.getByRole("button", { name: /^Cancel$/ }).click(); + await expect(jobRow).toContainText("cancelled", { timeout: 15_000 }); + + // Settle so any post-cancel async cleanup has a chance to fire its error + // path before we check the counter. + await page.waitForTimeout(1_000); + + const counts = await readUncaughtCount(); + expect( + counts.messages.filter((m) => m.includes("Controller is already closed")), + ).toEqual([]); + expect(counts.uncaught).toBe(0); +}); diff --git a/editor/e2e/no-subs-fallback.spec.ts b/editor/e2e/no-subs-fallback.spec.ts @@ -0,0 +1,156 @@ +import { readFile } from "node:fs/promises"; +import { fileURLToPath } from "node:url"; +import path from "node:path"; +import { test, expect } from "@playwright/test"; +import { pathExists, readJson, resetData, writeSettings } from "./helpers"; + +const CHANNEL = "test-livechat"; +const ROOT = `test-transcripts/channels/${CHANNEL}`; +const here = path.dirname(fileURLToPath(import.meta.url)); + +type DownloadOutcome = { + status: string; + attempts: Array<{ kind: string; handling: string }>; + fellBackToTranscribe?: boolean; +}; + +async function defaultSettings(): Promise<Record<string, unknown>> { + const raw = await readFile( + path.join(here, "fixtures", "test-settings.default.json"), + "utf8", + ); + return JSON.parse(raw) as Record<string, unknown>; +} + +test("default off: live_chat-only video downloads audio but defers whisper", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("livechat-fallback-channel"); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + "No subs available for livechatonly1; falling back to audio download + whisper.", + { timeout: 60_000 }, + ); + await expect(log).toContainText( + "Audio downloaded for livechatonly1; skipping inline whisper", + { timeout: 60_000 }, + ); + await expect(log).toContainText("Managed download complete", { + timeout: 60_000, + }); + + expect( + await pathExists( + `${ROOT}/data/livechatonly1/transcript.live_chat.json`, + ), + ).toBe(true); + expect(await pathExists(`${ROOT}/data/livechatonly1/audio.mp3`)).toBe(true); + expect( + await pathExists(`${ROOT}/data/livechatonly1/transcript.json`), + ).toBe(false); + + const outcome = await readJson<DownloadOutcome>( + `${ROOT}/data/livechatonly1/download-outcome.json`, + ); + expect(outcome.status).toBe("ok"); + expect(outcome.fellBackToTranscribe).toBe(true); + expect( + outcome.attempts.some( + (a) => a.kind === "no-subs-fallback" && a.handling === "transcribe", + ), + ).toBe(true); + + // The fallback should reuse the metadata.info.json the primary attempt + // wrote, not re-query the extractor. + const invocations = await readFile( + path.join(here, "..", ROOT, "fake-ytdlp.invocations"), + "utf8", + ); + expect(invocations).toMatch( + /load-info-json:.*data\/livechatonly1\/metadata\.info\.json/, + ); +}); + +test("default off: no-captions video downloads audio but defers whisper", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("livechat-fallback-channel"); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + "No subs available for nocaptions001; falling back to audio download + whisper.", + { timeout: 60_000 }, + ); + await expect(log).toContainText( + "Audio downloaded for nocaptions001; skipping inline whisper", + { timeout: 60_000 }, + ); + expect(await pathExists(`${ROOT}/data/nocaptions001/audio.mp3`)).toBe(true); + expect( + await pathExists(`${ROOT}/data/nocaptions001/transcript.json`), + ).toBe(false); + + const outcome = await readJson<DownloadOutcome>( + `${ROOT}/data/nocaptions001/download-outcome.json`, + ); + expect(outcome.status).toBe("ok"); + expect(outcome.fellBackToTranscribe).toBe(true); +}); + +test("video with real auto-subs does NOT trigger fallback (regression)", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("livechat-fallback-channel"); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText("Managed download complete", { + timeout: 60_000, + }); + await expect(log).not.toContainText("No subs available for normalvideo1"); + expect( + await pathExists(`${ROOT}/data/normalvideo1/transcript.en.vtt`), + ).toBe(true); + expect(await pathExists(`${ROOT}/data/normalvideo1/audio.mp3`)).toBe(false); +}); + +test("inlineTranscribeOnFallback=true runs whisper inline after the fallback", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("livechat-fallback-channel"); + await writeSettings({ + ...(await defaultSettings()), + inlineTranscribeOnFallback: true, + }); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Download videos" }).click(); + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText( + "No subs available for livechatonly1; falling back to audio download + whisper.", + { timeout: 60_000 }, + ); + await expect(log).not.toContainText( + "Audio downloaded for livechatonly1; skipping inline whisper", + ); + await expect(log).toContainText("Managed download complete", { + timeout: 60_000, + }); + + expect(await pathExists(`${ROOT}/data/livechatonly1/audio.mp3`)).toBe(true); + expect( + await pathExists(`${ROOT}/data/livechatonly1/transcript.json`), + ).toBe(true); + + const outcome = await readJson<DownloadOutcome>( + `${ROOT}/data/livechatonly1/download-outcome.json`, + ); + expect(outcome.status).toBe("ok-auto-transcribed"); + expect(outcome.fellBackToTranscribe).toBe(true); +}); diff --git a/editor/e2e/retry-bucket.spec.ts b/editor/e2e/retry-bucket.spec.ts @@ -0,0 +1,110 @@ +import { mkdir, writeFile } from "node:fs/promises"; +import { dirname } from "node:path"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath } from "./helpers"; + +const CHANNEL = "availability-test"; +const FIXTURE = "availability-baseline"; + +type AvailabilityStatus = + | "public" + | "unlisted" + | "private" + | "members_only" + | "needs_auth" + | "deleted" + | "error"; + +async function writeAvailability( + id: string, + status: AvailabilityStatus, + opts: { error?: string } = {}, +) { + const path = resolvePath( + `test-transcripts/channels/${CHANNEL}/data/${id}/availability.json`, + ); + await mkdir(dirname(path), { recursive: true }); + const record = { + checkedAt: new Date().toISOString(), + availability: status, + webpageUrl: `https://www.youtube.com/watch?v=${id}`, + ...(opts.error ? { error: opts.error } : {}), + }; + await writeFile(path, JSON.stringify(record)); +} + +test("retry control renders for needs_auth and is absent on excluded buckets", async ({ + page, +}) => { + await resetData(FIXTURE); + // Populate buckets directly via sidecars so the test doesn't depend on + // running availability checks first. + await writeAvailability("vidneedsauth1", "needs_auth"); + await writeAvailability("viddeleted1", "deleted"); + await writeAvailability("vidprivate1", "private"); + await writeAvailability("vidmembers1", "members_only"); + + await page.goto(`/channels/${CHANNEL}`); + + const needsAuth = page.getByLabel("retry needs auth bucket"); + await expect(needsAuth).toBeVisible(); + await expect( + needsAuth.getByRole("button", { name: /^Retry \(1\)$/ }), + ).toBeVisible(); + + // Excluded-from-download statuses must NOT show a retry control even + // though their bucket cards render. + await expect(page.getByLabel("availability deleted list")).toBeVisible(); + await expect(page.getByLabel("retry deleted bucket")).toHaveCount(0); + await expect(page.getByLabel("retry private bucket")).toHaveCount(0); + await expect(page.getByLabel("retry members-only bucket")).toHaveCount(0); +}); + +test("retry control renders for the error bucket", async ({ page }) => { + await resetData(FIXTURE); + // Reuse an existing fixture video for the synthetic error state — the + // test only cares about UI wiring, not the cause of the error. + await writeAvailability("vidneedsauth1", "error", { error: "transient" }); + + await page.goto(`/channels/${CHANNEL}`); + + const errorBucket = page.getByLabel("retry error bucket"); + await expect(errorBucket).toBeVisible(); + await expect( + errorBucket.getByRole("button", { name: /^Retry \(1\)$/ }), + ).toBeVisible(); +}); + +test("clicking retry on needs_auth re-downloads the listed video", async ({ + page, +}) => { + test.setTimeout(60_000); + await resetData(FIXTURE); + await writeAvailability("vidneedsauth1", "needs_auth"); + // retry-bucket reads the channel playlist file to know which URL to fetch. + await writeFile( + resolvePath(`test-transcripts/channels/${CHANNEL}/playlist`), + "https://www.youtube.com/watch?v=vidneedsauth1\n", + ); + + await page.goto(`/channels/${CHANNEL}`); + + await page + .getByLabel("retry needs auth bucket") + .getByRole("button", { name: /^Retry \(1\)$/ }) + .click(); + + const log = page.getByLabel("Retry needs auth output"); + await expect(log).toContainText("Retry allowlist: kept 1 of 1", { + timeout: 30_000, + }); + await expect(log).toContainText("Managed download complete", { + timeout: 30_000, + }); + + expect( + await pathExists( + `test-transcripts/channels/${CHANNEL}/data/vidneedsauth1/transcript.en.vtt`, + ), + ).toBe(true); +}); diff --git a/editor/e2e/undownloaded.spec.ts b/editor/e2e/undownloaded.spec.ts @@ -45,11 +45,9 @@ test("clicking an undownloaded entry lands on a usable video page", async ({ "fake00000002", ); await expect( - page.getByRole("button", { name: /^Download audio/ }), - ).toBeVisible(); - await expect( - page.getByRole("button", { name: "Whisper transcribe" }), + page.getByRole("button", { name: /^Run download pipeline$/ }), ).toBeVisible(); + await expect(page.getByLabel("download mode")).toBeVisible(); }); test("undownloaded list shrinks after a one-click whisper completes", async ({ @@ -57,9 +55,10 @@ test("undownloaded list shrinks after a one-click whisper completes", async ({ }) => { await resetData("youtube-with-playlist"); await page.goto("/channels/test-youtube/videos/fake00000002"); - await page.getByRole("button", { name: "Whisper transcribe" }).click(); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); await expect( - page.getByLabel("Whisper transcribe fake00000002 output"), + page.getByLabel("Audio + Whisper for fake00000002 output"), ).toContainText("Transcribe fake00000002 done", { timeout: 30_000 }); expect( diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts @@ -106,7 +106,7 @@ test("transcode keeps source and writes audio.<target>", async ({ page }) => { ).toBe(true); }); -test("download audio for a YouTube video with no auto-subs", async ({ +test("audio + whisper for a YouTube video with no auto-subs", async ({ page, }) => { await resetData("one-youtube-channel-with-data"); @@ -128,10 +128,11 @@ test("download audio for a YouTube video with no auto-subs", async ({ await expect(page.getByRole("heading", { level: 1 })).toContainText( "No-subs video", ); - await page.getByRole("button", { name: /^Download audio/ }).click(); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); await expect( - page.getByLabel("Download audio for test1234567 output"), - ).toContainText("download complete", { timeout: 30_000 }); + page.getByLabel("Audio + Whisper for test1234567 output"), + ).toContainText("Transcribe test1234567 done", { timeout: 30_000 }); expect( await pathExists( "test-transcripts/channels/test-youtube/data/test1234567/audio.mp3", @@ -362,7 +363,7 @@ test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => { ); // Transcode summary too. await expect(page.getByLabel("Transcode stage summary")).toContainText( - "Convert 1 audio file to another format.", + "Convert 1 source file to another format.", ); // Per-file transcode row exists for audio.mp3 only — not for the text files. await expect( @@ -376,6 +377,42 @@ test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => { ).toHaveCount(0); }); +test("Transcode card lists a kept source video (audio.mp4) so it can be re-extracted", async ({ + page, +}) => { + // Repro: a truncated audio.mp3 next to an intact audio.mp4 kept source. + // Pre-fix, audio.mp4 was filtered out of audioFiles, so no Transcode row + // appeared and the user could not overwrite the truncated mp3 from the UI. + await resetData("one-transcribe-channel-with-audio"); + const dir = resolvePath( + "test-transcripts/channels/test-transcribe/data/vidA", + ); + await writeFile(`${dir}/audio.mp4`, "fake source container"); + await writeFile(`${dir}/audio.mp3`, "truncated extraction"); + + await page.goto("/channels/test-transcribe/videos/vidA"); + + // Both source files appear as Transcode rows. + await expect( + page.getByRole("heading", { name: "Transcode audio.mp4" }), + ).toBeVisible(); + await expect( + page.getByRole("heading", { name: "Transcode audio.mp3" }), + ).toBeVisible(); + // Summary counts the m4a fixture file plus the mp4 and mp3 we added. + await expect(page.getByLabel("Transcode stage summary")).toContainText( + "Convert 3 source files to another format.", + ); + + // The Transcribe card still only shows real audio (no audio.mp4 row). + await expect( + page.getByRole("heading", { name: "Transcribe audio.mp4" }), + ).toHaveCount(0); + await expect( + page.getByRole("heading", { name: /^Transcribe audio\.m4a$/ }), + ).toBeVisible(); +}); + test("existing transcript text files are inspectable from the file list", async ({ page, }) => { @@ -391,35 +428,3 @@ test("existing transcript text files are inspectable from the file list", async ); }); -test("Redownload writes audio in the chosen format", async ({ page }) => { - await resetData("one-youtube-channel-with-data"); - // Stage a video dir whose name matches the URL id so fake-ytdlp lands its - // output (data/<id>/audio.<fmt>) back into the dir we navigated to. Include - // a pre-existing audio.m4a so the panel is in "Redownload" mode. - const dir = resolvePath( - "test-transcripts/channels/test-youtube/data/test1234567", - ); - await mkdir(dir, { recursive: true }); - await writeFile( - `${dir}/metadata.info.json`, - JSON.stringify({ - id: "test1234567", - title: "Redownload target", - webpage_url: "https://www.youtube.com/watch?v=test1234567", - }), - ); - await writeFile(`${dir}/audio.m4a`, "stale audio"); - await page.goto("/channels/test-youtube/videos/test1234567"); - await page.getByLabel("redownload audio format").selectOption("mp3"); - await page - .getByRole("button", { name: /^Redownload audio \(mp3\)$/ }) - .click(); - await expect( - page.getByLabel("Redownload audio for test1234567 output"), - ).toContainText("download complete", { timeout: 30_000 }); - expect( - await pathExists( - "test-transcripts/channels/test-youtube/data/test1234567/audio.mp3", - ), - ).toBe(true); -}); diff --git a/editor/e2e/whisper-video.spec.ts b/editor/e2e/whisper-video.spec.ts @@ -7,7 +7,7 @@ test("one-click whisper downloads audio then transcribes a YouTube video", async }) => { await resetData("youtube-with-playlist"); // fake00000001 is the seeded "downloaded" entry. Drop its auto-subs so the - // page treats it as untranscribed and the Whisper button appears. + // page treats it as untranscribed. await rm( resolvePath( "test-transcripts/channels/test-youtube/data/fake00000001/transcript.en.vtt", @@ -15,9 +15,10 @@ test("one-click whisper downloads audio then transcribes a YouTube video", async ); await page.goto("/channels/test-youtube/videos/fake00000001"); - await page.getByRole("button", { name: "Whisper transcribe" }).click(); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); - const log = page.getByLabel("Whisper transcribe fake00000001 output"); + const log = page.getByLabel("Audio + Whisper for fake00000001 output"); await expect(log).toContainText("downloading", { timeout: 30_000 }); await expect(log).toContainText("Transcribe fake00000001 done", { timeout: 30_000, @@ -41,9 +42,10 @@ test("one-click whisper skips download when audio is already on disk", async ({ await resetData("one-transcribe-channel-with-audio"); await page.goto("/channels/test-transcribe/videos/vidA"); - await page.getByRole("button", { name: "Whisper transcribe" }).click(); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); - const log = page.getByLabel("Whisper transcribe vidA output"); + const log = page.getByLabel("Audio + Whisper for vidA output"); await expect(log).toContainText("skipping download", { timeout: 30_000 }); await expect(log).toContainText("Transcribe vidA done", { timeout: 30_000 }); @@ -61,13 +63,3 @@ test("one-click whisper skips download when audio is already on disk", async ({ const invocations = await readFile(invocationsPath, "utf8").catch(() => ""); expect(invocations).not.toContain("download-one:"); }); - -test("Whisper button is hidden when a transcript already exists", async ({ - page, -}) => { - await resetData("one-youtube-channel-with-data"); - await page.goto("/channels/test-youtube/videos/20240101_test1234567"); - await expect( - page.getByRole("button", { name: "Whisper transcribe" }), - ).toHaveCount(0); -}); diff --git a/editor/e2e/whisper.spec.ts b/editor/e2e/whisper.spec.ts @@ -19,6 +19,35 @@ test("transcribes every audio file with no transcript", async ({ page }) => { } }); +test("Transcribe missing skips VTT-only videos instead of failing them", async ({ + page, +}) => { + await resetData("youtube-with-playlist"); + await page.goto("/channels/test-youtube"); + await page.getByRole("button", { name: "Transcribe missing" }).click(); + const log = page.getByLabel("Transcribe missing output"); + await expect(log).toContainText( + "Whisper batch: 0 succeeded, 0 failed, 1 skipped, 0 attempted.", + { timeout: 30_000 }, + ); + await expect(log).toContainText( + "Transcription for fake00000001 already exists", + ); + + const failPath = + "test-transcripts/channels/test-youtube/failed-transcriptions"; + if (await pathExists(failPath)) { + const raw = await readFile(resolvePath(failPath), "utf8"); + expect(raw).not.toContain("fake00000001"); + } + + expect( + await pathExists( + "test-transcripts/channels/test-youtube/data/fake00000001/transcript.json", + ), + ).toBe(false); +}); + test("verify reports nothing missing once transcribed", async ({ page }) => { await resetData("one-transcribe-channel-with-audio"); await page.goto("/channels/test-transcribe");