commit fb6f6cb96eeea02f1100d3e1ed45b3ea2a528d50
parent af80ae1efde681097533f9e5ad1dc641dfbb04d3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 18 May 2026 09:29:26 -0400
retry pipelines
Diffstat:
32 files changed, 1485 insertions(+), 274 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -11,9 +11,13 @@ import {
} from "../lib/videoStatus";
import {
AVAILABILITY_VALUES,
+ EXCLUDED_FROM_DOWNLOAD,
type Availability,
} from "../lib/availability";
-import { loadAvailability } from "../lib/availability-server";
+import {
+ loadAvailability,
+ resolveEffectiveAvailability,
+} from "../lib/availability-server";
import type { Paths } from "../lib/paths";
import { extractVideoId, isRumbleUrl } from "../ytdlp/runYtdlp";
import { loadFailedTranscriptions } from "./failedTranscriptions";
@@ -24,6 +28,12 @@ export type AvailabilitySnapshot = {
unchecked: string[];
};
+export type ExcludedFromDownload = {
+ membersOnly: string[];
+ deleted: string[];
+ private: string[];
+};
+
export type ChannelSnapshot = {
generatedAt: string;
totals: {
@@ -43,9 +53,24 @@ export type ChannelSnapshot = {
duplicateDirs: string[];
};
undownloadedIds: string[];
+ excludedFromDownload?: ExcludedFromDownload;
availability?: AvailabilitySnapshot;
};
+export function emptyExcludedFromDownload(): ExcludedFromDownload {
+ return { membersOnly: [], deleted: [], private: [] };
+}
+
+export function normalizeExcludedFromDownload(
+ raw: Partial<ExcludedFromDownload> | undefined,
+): ExcludedFromDownload {
+ return {
+ membersOnly: raw?.membersOnly ?? [],
+ deleted: raw?.deleted ?? [],
+ private: raw?.private ?? [],
+ };
+}
+
function emptyAvailability(): AvailabilitySnapshot {
const byStatus = {} as Record<Availability, string[]>;
for (const v of AVAILABILITY_VALUES) byStatus[v] = [];
@@ -145,7 +170,14 @@ export async function generateChannelSnapshot(
}
}
const availability = await loadAvailability(dir);
- return { id, files, webpageUrl, availability };
+ const effectiveAvailability = await resolveEffectiveAvailability(dir);
+ return {
+ id,
+ files,
+ webpageUrl,
+ availability,
+ effectiveAvailability,
+ };
}),
),
);
@@ -160,6 +192,22 @@ export async function generateChannelSnapshot(
}
}
+ // Computed up-front so the buckets below can suppress IDs that can't be
+ // acted on (members_only / deleted / private). Surfacing them in
+ // "Missing metadata" or "Failed transcriptions" creates noise the user
+ // cannot resolve.
+ const excludedById = new Map<string, Availability>();
+ for (const v of perVideo) {
+ if (
+ v.effectiveAvailability &&
+ (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes(
+ v.effectiveAvailability,
+ )
+ ) {
+ excludedById.set(v.id, v.effectiveAvailability);
+ }
+ }
+
const noTranscript: string[] = [];
const downloadedNoTranscript: string[] = [];
const untranscoded: string[] = [];
@@ -171,7 +219,7 @@ export async function generateChannelSnapshot(
for (const { id, files } of perVideo) {
if (isVideoTranscribed(files)) transcribed++;
if (isVideoDownloaded(files)) downloaded++;
- if (!files.hasMeta) noMetadata.push(id);
+ if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id);
if (
targetAudioFile &&
files.audioFiles.length > 0 &&
@@ -213,6 +261,16 @@ export async function generateChannelSnapshot(
if (!filesById.has(id)) missingFromArchive.push(id);
}
+ const excludedFromDownload = emptyExcludedFromDownload();
+ for (const [id, cls] of excludedById) {
+ if (cls === "members_only") excludedFromDownload.membersOnly.push(id);
+ else if (cls === "deleted") excludedFromDownload.deleted.push(id);
+ else if (cls === "private") excludedFromDownload.private.push(id);
+ }
+ excludedFromDownload.membersOnly.sort();
+ excludedFromDownload.deleted.sort();
+ excludedFromDownload.private.sort();
+
const undownloadedIds: string[] = [];
for (const url of urls) {
const slugId = extractVideoId(url);
@@ -223,6 +281,7 @@ export async function generateChannelSnapshot(
: slugId;
const f = filesById.get(dirId);
if (f && videoHasAnyArtifact(f)) continue;
+ if (excludedById.has(dirId)) continue;
undownloadedIds.push(dirId);
}
@@ -240,11 +299,12 @@ export async function generateChannelSnapshot(
multipleAudioFormats: multipleAudioFormats.sort(),
untranscribable: untranscribable.sort(),
noMetadata: noMetadata.sort(),
- failedListed,
+ failedListed: failedListed.filter((id) => !excludedById.has(id)),
missingFromArchive: missingFromArchive.sort(),
duplicateDirs: [],
},
undownloadedIds,
+ excludedFromDownload,
availability,
};
diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts
@@ -3,6 +3,7 @@ import fs from "fs-extra";
import pLimit from "p-limit";
import type { Paths } from "../lib/paths";
import type { AudioFormat } from "../lib/channelConfig";
+import { VTT_FILENAME, WHISPER_FILENAME } from "../lib/videoStatus";
import { transcribeOneVideo } from "./transcribeOne";
import { resolveShardItems } from "./shard";
@@ -90,7 +91,16 @@ export async function runWhisperBatch({
skipped++;
return;
}
- if (await pathExists(path.join(videoPath, "transcript.json"))) {
+ // A transcript already exists if either whisper has run
+ // (transcript.json) OR yt-dlp wrote auto/manual English subs
+ // (transcript.en.vtt). Without the VTT branch, videos on a
+ // youtube-handling channel with only VTT auto-subs end up
+ // attempted -> fail with "no audio file found" -> get added to
+ // failed-transcriptions on every run.
+ if (
+ (await pathExists(path.join(videoPath, WHISPER_FILENAME))) ||
+ (await pathExists(path.join(videoPath, VTT_FILENAME)))
+ ) {
log(`Transcription for ${videoDir} already exists`);
skipped++;
return;
diff --git a/common/jobs/streamCommand.ts b/common/jobs/streamCommand.ts
@@ -58,6 +58,51 @@ function makeJob(
return { id, logPath, record };
}
+// Safety contract shared by both stream creators. The consumer (RSC encoder)
+// can close the stream at any time when the client disconnects; once closed,
+// any further enqueue/close on the controller throws ERR_INVALID_STATE. The
+// throw escapes our local try/catch because it surfaces inside the RSC
+// encoder's own pipeline, not at our call site. Tracking a `closed` flag and
+// gating all controller ops on it ensures we stop producing the moment the
+// downstream goes away. Returns helpers plus a `markClosed()` for the
+// cancel() / finally() paths.
+function makeSafeController() {
+ let controller: ReadableStreamDefaultController<string> | null = null;
+ let closed = false;
+ const setController = (c: ReadableStreamDefaultController<string>) => {
+ controller = c;
+ };
+ const safeEnqueue = (text: string) => {
+ if (closed) return;
+ try {
+ controller?.enqueue(text);
+ } catch (err) {
+ if (
+ (err as NodeJS.ErrnoException | undefined)?.code === "ERR_INVALID_STATE"
+ ) {
+ closed = true;
+ }
+ }
+ };
+ const safeClose = () => {
+ if (closed) return;
+ closed = true;
+ try {
+ controller?.close();
+ } catch {}
+ };
+ const markClosed = () => {
+ closed = true;
+ };
+ return { setController, safeEnqueue, safeClose, markClosed };
+}
+
+// Best-effort error listener for the log WriteStream: writes can fail (disk
+// full, permissions, etc.) and would otherwise become uncaughtExceptions.
+function ignoreFileStreamErrors(stream: WriteStream): void {
+ stream.on("error", () => {});
+}
+
export async function runManagedCommand(
opts: RunManagedCommandOpts,
): Promise<StreamActionResult> {
@@ -71,27 +116,32 @@ export async function runManagedCommand(
opts.videoId,
);
- let controller: ReadableStreamDefaultController<string> | null = null;
+ const safe = makeSafeController();
let fileStream: WriteStream | null = null;
let cancelledBeforeStart = false;
const stream = new ReadableStream<string>({
start(c) {
- controller = c;
+ safe.setController(c);
},
cancel() {
- // Client disconnected. Job continues to write to its log file.
+ // Consumer disconnected (page navigation, tab close, RSC response
+ // tear-down). Stop pushing into the controller — but keep the child
+ // and the on-disk log going so the job runs to completion and can be
+ // observed by another reconnecting client via /api/jobs/<id>/log.
+ // Explicit user cancels go through `registry.cancel()` (which kills
+ // the child), not through this source-cancel hook.
+ safe.markClosed();
},
});
const start = () => {
if (cancelledBeforeStart) {
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
return;
}
fileStream = createWriteStream(logPath);
+ ignoreFileStreamErrors(fileStream);
const child = execa(opts.command, opts.args, {
cwd: opts.cwd,
env: opts.env,
@@ -103,9 +153,7 @@ export async function runManagedCommand(
child.all?.on("data", (chunk: Buffer) => {
fileStream!.write(chunk);
- try {
- controller?.enqueue(chunk.toString("utf8"));
- } catch {}
+ safe.safeEnqueue(chunk.toString("utf8"));
});
child.all?.on("error", () => {});
@@ -123,24 +171,18 @@ export async function runManagedCommand(
registry.finalize(id, "cancelled");
return;
}
- try {
- controller?.enqueue(`\n[error] ${(err as Error).message}\n`);
- } catch {}
+ safe.safeEnqueue(`\n[error] ${(err as Error).message}\n`);
registry.finalize(id, "failed");
})
.finally(() => {
fileStream?.end();
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
});
};
const onCancel = () => {
cancelledBeforeStart = true;
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
};
registry.enqueue(record, { start, onCancel });
@@ -160,34 +202,38 @@ export async function runManagedFunction(
opts.videoId,
);
- let controller: ReadableStreamDefaultController<string> | null = null;
+ const safe = makeSafeController();
let fileStream: WriteStream | null = null;
let cancelledBeforeStart = false;
const stream = new ReadableStream<string>({
start(c) {
- controller = c;
+ safe.setController(c);
+ },
+ cancel() {
+ // Consumer disconnected (page navigation, tab close, RSC response
+ // tear-down). Stop pushing into the controller — but let the function
+ // run to completion, writing to disk; reconnecting clients can poll
+ // /api/jobs/<id>/log. Explicit user cancels go through
+ // `registry.cancel()` which aborts this controller.
+ safe.markClosed();
},
- cancel() {},
});
const start = () => {
if (cancelledBeforeStart) {
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
return;
}
fileStream = createWriteStream(logPath);
+ ignoreFileStreamErrors(fileStream);
const abort = new AbortController();
record.abortController = abort;
const onLog = (line: string) => {
const text = line.endsWith("\n") ? line : `${line}\n`;
fileStream!.write(text);
- try {
- controller?.enqueue(text);
- } catch {}
+ safe.safeEnqueue(text);
};
opts
@@ -209,17 +255,13 @@ export async function runManagedFunction(
})
.finally(() => {
fileStream?.end();
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
});
};
const onCancel = () => {
cancelledBeforeStart = true;
- try {
- controller?.close();
- } catch {}
+ safe.safeClose();
};
registry.enqueue(record, { start, onCancel });
diff --git a/common/lib/availability-server.ts b/common/lib/availability-server.ts
@@ -3,8 +3,13 @@ import { readFile, rename, writeFile } from "node:fs/promises";
import {
AVAILABILITY_FILENAME,
AVAILABILITY_VALUES,
+ type Availability,
type AvailabilityRecord,
} from "./availability";
+import {
+ DOWNLOAD_OUTCOME_FILENAME,
+ type DownloadOutcomeRecord,
+} from "./downloadOutcome";
export function availabilityPath(videoDir: string): string {
return path.join(videoDir, AVAILABILITY_FILENAME);
@@ -38,3 +43,31 @@ export async function writeAvailability(
await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
await rename(tmp, file);
}
+
+// Resolve a video's availability class, preferring an explicit availability.json
+// (the authoritative record written by the availability-check or metadata-backfill
+// passes) and falling back to the last attempt's availabilityClass in
+// download-outcome.json. This second source matters for videos that have
+// never successfully downloaded — their availability never reaches
+// availability.json otherwise.
+export async function resolveEffectiveAvailability(
+ videoDir: string,
+): Promise<Availability | null> {
+ const explicit = await loadAvailability(videoDir);
+ if (explicit) return explicit.availability;
+ try {
+ const raw = await readFile(
+ path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME),
+ "utf8",
+ );
+ const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>;
+ const attempts = parsed.attempts;
+ if (Array.isArray(attempts) && attempts.length > 0) {
+ const last = attempts[attempts.length - 1];
+ if (last?.availabilityClass) return last.availabilityClass;
+ }
+ } catch {
+ // No outcome file or unreadable → unknown
+ }
+ return null;
+}
diff --git a/common/lib/availability.ts b/common/lib/availability.ts
@@ -21,6 +21,16 @@ export const AVAILABILITY_VALUES: Availability[] = [
"error",
];
+// Availability classes treated as permanently blocking: videos in these
+// states are filtered out of undownloadedIds and skipped by the managed
+// download flows. needs_auth is intentionally excluded (auth-retry may
+// recover it); error is excluded too (often transient: rate limits, network).
+export const EXCLUDED_FROM_DOWNLOAD: ReadonlyArray<Availability> = [
+ "members_only",
+ "deleted",
+ "private",
+];
+
export type AvailabilityRecord = {
checkedAt: string;
availability: Availability;
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -19,6 +19,11 @@ export type SiteSettings = {
// within one invocation, so without this the managed loop hammers the
// source IP back-to-back. 0 disables. Per-channel override available.
sleepBetweenDownloadsSeconds: number;
+ // When true, the no-subs fallback in the managed downloader runs whisper
+ // inline immediately after the audio download succeeds. When false
+ // (default), audio is left for the next "Transcribe missing" pass so a
+ // batch download finishes faster and whisper can parallelize.
+ inlineTranscribeOnFallback: boolean;
};
export const SLEEP_BETWEEN_DOWNLOADS_MAX_SECONDS = 600;
@@ -61,6 +66,7 @@ function defaults(): SiteSettings {
transcribeModel: paths.whisperModel,
cookiesFromBrowser: "",
sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS,
+ inlineTranscribeOnFallback: false,
};
}
@@ -101,6 +107,9 @@ export function getSettings(): SiteSettings {
merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds(
merged.sleepBetweenDownloadsSeconds,
);
+ if (typeof merged.inlineTranscribeOnFallback !== "boolean") {
+ merged.inlineTranscribeOnFallback = false;
+ }
return merged;
}
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -37,6 +37,11 @@ export type ManagedDownloadOpts = {
// When false, suppress appending to the channel archive file. Mirrors
// `ignoreArchive` from the playlist-level callers.
appendArchive?: boolean;
+ // When true, run whisper inline immediately after the no-subs fallback's
+ // audio download succeeds. When false (default), leave the audio for the
+ // next "Transcribe missing" pass. Resolved from the site setting by the
+ // playlist-level caller.
+ inlineTranscribeOnFallback?: boolean;
};
const OUTPUT_ARGS: string[] = [
@@ -199,11 +204,16 @@ async function resolveVideoIdFromUrl(
async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> {
const entries = await readdir(videoDir).catch(() => [] as string[]);
- return entries.some(
- (e) =>
- e === "transcript.json" ||
- /^transcript\.[^.]+\.(?:vtt|json|json3|srv1|srv2|srv3)$/.test(e),
- );
+ return entries.some((e) => {
+ if (e === "transcript.json") return true;
+ const m = e.match(
+ /^transcript\.([^.]+)\.(?:vtt|json|json3|srv1|srv2|srv3)$/,
+ );
+ if (!m) return false;
+ // live_chat is YouTube's chat replay, not a transcript of speech — don't
+ // let its presence suppress the no-subs fallback to audio + whisper.
+ return m[1] !== "live_chat";
+ });
}
async function metadataReportsNoCaptions(
@@ -229,12 +239,13 @@ async function metadataReportsNoCaptions(
}
const subs = parsed.subtitles ?? {};
const auto = parsed.automatic_captions ?? {};
- return (
- typeof subs === "object" &&
- typeof auto === "object" &&
- Object.keys(subs).length === 0 &&
- Object.keys(auto).length === 0
- );
+ if (typeof subs !== "object" || typeof auto !== "object") return false;
+ // live_chat is YouTube's chat replay, not a caption track. yt-dlp places
+ // it under `subtitles`, so a video whose only listed track is live_chat
+ // should be treated as having no captions for fallback purposes.
+ const realSubs = Object.keys(subs).filter((k) => k !== "live_chat");
+ const realAuto = Object.keys(auto).filter((k) => k !== "live_chat");
+ return realSubs.length === 0 && realAuto.length === 0;
}
function trimError(stderrTail: string): string | undefined {
@@ -429,6 +440,10 @@ export async function downloadOneManaged(
status === "ok-with-cookies"
? opts.globalCookiesFromBrowser
: undefined;
+ // Feed yt-dlp the metadata it already wrote during the primary
+ // attempt instead of re-querying the extractor — saves a network
+ // round-trip per video, which adds up across batch runs and helps
+ // dodge rate limits we'd otherwise burn on info we already have.
const fallbackArgs = [
"--ignore-config",
"--restrict-filenames",
@@ -437,8 +452,8 @@ export async function downloadOneManaged(
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
...channelConfigArgs(fallbackConfig, fallbackCookieOverride),
- "--",
- opts.videoUrl,
+ "--load-info-json",
+ path.join(videoDir, "metadata.info.json"),
];
const fallbackRes = await runOneYtdlp(opts, channelDir, fallbackArgs);
const fallbackAvail = attemptSucceeded(fallbackRes.exitCode)
@@ -458,24 +473,33 @@ export async function downloadOneManaged(
if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine;
if (attemptSucceeded(fallbackRes.exitCode)) {
- // Inline whisper: matches whisperVideoAction's shape.
- try {
- await transcribeOneVideo({
- paths: opts.paths,
- videoDir,
- videoId,
- audioFilename: `audio.${audioFmt}`,
- onLog: opts.onLog,
- signal: opts.signal,
- });
- fellBackToTranscribe = true;
- status = "ok-auto-transcribed";
- } catch (err) {
+ fellBackToTranscribe = true;
+ if (opts.inlineTranscribeOnFallback) {
+ // Inline whisper: matches whisperVideoAction's shape.
+ try {
+ await transcribeOneVideo({
+ paths: opts.paths,
+ videoDir,
+ videoId,
+ audioFilename: `audio.${audioFmt}`,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ status = "ok-auto-transcribed";
+ } catch (err) {
+ opts.onLog(
+ `Whisper failed after no-subs fallback: ${(err as Error).message}\n`,
+ );
+ status = "failed";
+ lastSucceeded = false;
+ }
+ } else {
opts.onLog(
- `Whisper failed after no-subs fallback: ${(err as Error).message}\n`,
+ `Audio downloaded for ${videoId}; skipping inline whisper (inlineTranscribeOnFallback=off). Run "Transcribe missing" to transcribe.\n`,
);
- status = "failed";
- lastSucceeded = false;
+ // status remains whatever the primary attempt set; the
+ // fellBackToTranscribe flag distinguishes from a normal "ok"
+ // so the UI and downstream jobs can tell what happened.
}
} else {
status = "failed";
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -18,6 +18,11 @@ import {
} from "../lib/channelConfig";
import { getSettings } from "../lib/settings";
import type { Paths } from "../lib/paths";
+import {
+ EXCLUDED_FROM_DOWNLOAD,
+ type Availability,
+} from "../lib/availability";
+import { resolveEffectiveAvailability } from "../lib/availability-server";
import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability";
import { resolveShardItems } from "../controller/shard";
import { downloadOneManaged } from "./downloadOneManaged";
@@ -28,6 +33,7 @@ export type YtdlpMode =
| "download-missing"
| "download-missing-subs"
| "download-one-audio"
+ | "retry-bucket"
| "sync";
export type RunYtdlpOpts = {
@@ -54,6 +60,14 @@ export type RunYtdlpOpts = {
audioFormatOverride?: AudioFormat;
// download-one-audio only: appended after configArgs, before the URL.
extraYtdlpArgs?: string[];
+ // retry-bucket only: video IDs to limit the run to (matched against the
+ // saved playlist by extracted ID). IDs not present in the playlist are
+ // reported and skipped.
+ bucketIds?: ReadonlyArray<string>;
+ // retry-bucket only: override channelConfig.handling for this run without
+ // mutating the channel config on disk. Useful for retrying old "youtube"
+ // videos as "transcribe".
+ handlingOverride?: ChannelHandling;
};
export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> {
@@ -76,12 +90,27 @@ export async function runYtdlp(opts: RunYtdlpOpts): Promise<void> {
case "download-one-audio":
await downloadOneAudio(opts);
return;
+ case "retry-bucket":
+ await retryBucket(opts);
+ return;
case "sync":
await sync(opts);
return;
}
}
+async function retryBucket(opts: RunYtdlpOpts): Promise<void> {
+ if (!opts.bucketIds || opts.bucketIds.length === 0) {
+ opts.onLog("retry-bucket: no video IDs supplied. Nothing to do.\n");
+ return;
+ }
+ await downloadPlaylistManaged(opts, {
+ prefilter: "destination-exists",
+ idAllowList: opts.bucketIds,
+ handlingOverride: opts.handlingOverride,
+ });
+}
+
function channelRoot(opts: RunYtdlpOpts): string {
return path.join(opts.paths.channelsDir, opts.channelSlug);
}
@@ -203,8 +232,20 @@ type PrefilterMode = "archive" | "destination-exists";
async function downloadPlaylistManaged(
opts: RunYtdlpOpts,
- { prefilter }: { prefilter: PrefilterMode },
+ cfg: {
+ prefilter: PrefilterMode;
+ idAllowList?: ReadonlyArray<string>;
+ handlingOverride?: ChannelHandling;
+ },
): Promise<void> {
+ const { prefilter, idAllowList, handlingOverride } = cfg;
+ const effectiveHandling: ChannelHandling =
+ handlingOverride ?? opts.channelConfig.handling;
+ const effectiveChannelConfig: ChannelConfig =
+ handlingOverride && handlingOverride !== opts.channelConfig.handling
+ ? { ...opts.channelConfig, handling: handlingOverride }
+ : opts.channelConfig;
+
const root = channelRoot(opts);
const dataDir = path.join(root, "data");
const playlistPath = path.join(root, "playlist");
@@ -218,7 +259,7 @@ async function downloadPlaylistManaged(
`No saved playlist at ${playlistPath}. Run "Store playlist" first.`,
);
}
- const urls = playlistText
+ let urls = playlistText
.split("\n")
.map((s) => s.trim())
.filter(Boolean);
@@ -227,6 +268,38 @@ async function downloadPlaylistManaged(
? await buildRumbleSlugIndex(dataDir)
: null;
+ if (idAllowList) {
+ const allow = new Set(idAllowList);
+ const before = urls.length;
+ const matched: string[] = [];
+ const matchedIds = new Set<string>();
+ for (const url of urls) {
+ const slugId = extractVideoId(url);
+ if (!slugId) continue;
+ const dirId =
+ rumbleIndex && isRumbleUrl(url)
+ ? (rumbleIndex.get(slugId) ?? slugId)
+ : slugId;
+ if (allow.has(dirId)) {
+ matched.push(url);
+ matchedIds.add(dirId);
+ }
+ }
+ const missing = [...allow].filter((id) => !matchedIds.has(id));
+ opts.onLog(
+ `Retry allowlist: kept ${matched.length} of ${before} playlist URLs (${missing.length} requested IDs not in playlist).\n`,
+ );
+ if (missing.length > 0) {
+ opts.onLog(` Missing: ${missing.join(", ")}\n`);
+ }
+ urls = matched;
+ }
+ if (handlingOverride) {
+ opts.onLog(
+ `Handling override: using ${handlingOverride} for this run (channel config unchanged).\n`,
+ );
+ }
+
const tofetch: string[] = [];
if (prefilter === "archive") {
// Skip URLs whose ID is already recorded in the channel's archive file.
@@ -264,7 +337,7 @@ async function downloadPlaylistManaged(
rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
if (
dirId &&
- (await destinationExists(dataDir, dirId, opts.channelConfig.handling))
+ (await destinationExists(dataDir, dirId, effectiveHandling))
) {
alreadyComplete++;
continue;
@@ -278,6 +351,45 @@ async function downloadPlaylistManaged(
);
}
+ // Exclude videos whose effective availability marks them as permanently
+ // unavailable (members_only, deleted, private). Applies to both prefilter
+ // modes: failed download attempts don't write to the archive, so the
+ // archive-prefilter path would also keep retrying them.
+ const excludedCounts = { members_only: 0, deleted: 0, private: 0 };
+ const filteredTofetch: string[] = [];
+ for (const url of tofetch) {
+ const slugId = extractVideoId(url);
+ const dirId = slugId
+ ? rumbleIndex && isRumbleUrl(url)
+ ? (rumbleIndex.get(slugId) ?? slugId)
+ : slugId
+ : null;
+ if (dirId) {
+ const cls = await resolveEffectiveAvailability(
+ path.join(dataDir, dirId),
+ );
+ if (
+ cls &&
+ (EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes(cls)
+ ) {
+ excludedCounts[cls as keyof typeof excludedCounts]++;
+ continue;
+ }
+ }
+ filteredTofetch.push(url);
+ }
+ const totalExcluded =
+ excludedCounts.members_only +
+ excludedCounts.deleted +
+ excludedCounts.private;
+ if (totalExcluded > 0) {
+ opts.onLog(
+ `Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`,
+ );
+ }
+ tofetch.length = 0;
+ tofetch.push(...filteredTofetch);
+
// Sharding is only meaningful for the missing-files mode (matches prior
// behavior where `download-missing` accepted shardTotal / shardIndex).
const items =
@@ -303,6 +415,7 @@ async function downloadPlaylistManaged(
const settings = getSettings();
const globalCookies = settings.cookiesFromBrowser;
+ const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback;
const sleepSeconds =
opts.channelConfig.sleepBetweenDownloadsSeconds ??
settings.sleepBetweenDownloadsSeconds;
@@ -329,13 +442,14 @@ async function downloadPlaylistManaged(
if (firstFailure && abortOnError) return;
const outcome = await downloadOneManaged({
channelSlug: opts.channelSlug,
- channelConfig: opts.channelConfig,
+ channelConfig: effectiveChannelConfig,
paths: opts.paths,
videoUrl: url,
onLog: opts.onLog,
signal: opts.signal,
globalCookiesFromBrowser: globalCookies || undefined,
appendArchive: !opts.ignoreArchive,
+ inlineTranscribeOnFallback,
});
if (outcome.status === "failed") {
failedCount++;
diff --git a/editor/app/api/test/uncaught-count/route.ts b/editor/app/api/test/uncaught-count/route.ts
@@ -0,0 +1,72 @@
+import { NextResponse } from "next/server";
+
+export const dynamic = "force-dynamic";
+
+// E2E test harness only. Counts uncaughtException + unhandledRejection
+// events seen by the dev-server's Node process so a spec can assert that
+// "Controller is already closed" no longer leaks out of the streaming job
+// pipeline when the consumer disconnects.
+//
+// The process-level handler is installed once on first module load. It's
+// gated by the test-fixture TRANSCRIPTS_DIR path so production-shaped
+// installs never see it.
+
+type CountState = {
+ uncaught: number;
+ unhandled: number;
+ messages: string[];
+};
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttUncaughtCount__: CountState | undefined;
+}
+
+function isTestEnv(): boolean {
+ const dir = process.env.TRANSCRIPTS_DIR ?? "";
+ return dir.endsWith("test-transcripts");
+}
+
+function getState(): CountState {
+ if (!globalThis.__yttUncaughtCount__) {
+ globalThis.__yttUncaughtCount__ = {
+ uncaught: 0,
+ unhandled: 0,
+ messages: [],
+ };
+ if (isTestEnv()) {
+ process.on("uncaughtException", (err) => {
+ const s = globalThis.__yttUncaughtCount__!;
+ s.uncaught++;
+ s.messages.push(`uncaught: ${(err as Error)?.message ?? String(err)}`);
+ if (s.messages.length > 50) s.messages.shift();
+ });
+ process.on("unhandledRejection", (reason) => {
+ const s = globalThis.__yttUncaughtCount__!;
+ s.unhandled++;
+ s.messages.push(
+ `unhandled: ${(reason as Error)?.message ?? String(reason)}`,
+ );
+ if (s.messages.length > 50) s.messages.shift();
+ });
+ }
+ }
+ return globalThis.__yttUncaughtCount__;
+}
+
+export async function GET() {
+ const state = getState();
+ return NextResponse.json({
+ uncaught: state.uncaught,
+ unhandled: state.unhandled,
+ messages: state.messages.slice(),
+ });
+}
+
+export async function DELETE() {
+ const state = getState();
+ state.uncaught = 0;
+ state.unhandled = 0;
+ state.messages.length = 0;
+ return NextResponse.json({ ok: true });
+}
diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx
@@ -0,0 +1,89 @@
+"use client";
+
+import { useState } from "react";
+import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog";
+import { HANDLING_VALUES } from "yt-dlp-transcript-common/lib/channelConfig";
+import { QueueControl } from "../../../components/QueueControl";
+import { cancelJobAction } from "../../../jobs/actions";
+import { retryBucketAction } from "../pipelineActions";
+
+type Props = {
+ slug: string;
+ ids: string[];
+ actionLabel: string;
+ defaultQueueKey: string;
+ existingQueues: string[];
+};
+
+export function RetryBucketControl({
+ slug,
+ ids,
+ actionLabel,
+ defaultQueueKey,
+ existingQueues,
+}: Props) {
+ const [queue, setQueue] = useState(defaultQueueKey);
+ const [handlingOverride, setHandlingOverride] = useState("");
+ const [abortOnError, setAbortOnError] = useState(false);
+
+ if (ids.length === 0) return null;
+
+ return (
+ <div
+ className="flex flex-col gap-2"
+ aria-label={`retry ${actionLabel} bucket`}
+ >
+ <StreamActionLog
+ trigger={() =>
+ retryBucketAction(
+ slug,
+ ids,
+ queue,
+ abortOnError,
+ handlingOverride || undefined,
+ )
+ }
+ cancelAction={cancelJobAction}
+ buttonLabel={`Retry (${ids.length})`}
+ runningLabel="Retrying…"
+ label={`Retry ${actionLabel}`}
+ extraControls={
+ <>
+ <label className="flex items-center gap-1 text-xs text-zinc-500">
+ <span>Retry as</span>
+ <select
+ value={handlingOverride}
+ onChange={(e) => setHandlingOverride(e.target.value)}
+ aria-label={`handling override for retry ${actionLabel}`}
+ className="font-mono px-2 py-1 rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 text-zinc-900 dark:text-zinc-100"
+ >
+ <option value="">channel default</option>
+ {HANDLING_VALUES.map((h) => (
+ <option key={h} value={h}>
+ {h}
+ </option>
+ ))}
+ </select>
+ </label>
+ <QueueControl
+ value={queue}
+ onChange={setQueue}
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ actionLabel={`Retry ${actionLabel}`}
+ />
+ <label className="flex items-center gap-1 text-xs text-zinc-500">
+ <input
+ type="checkbox"
+ checked={abortOnError}
+ onChange={(e) => setAbortOnError(e.target.checked)}
+ aria-label={`abort on error for retry ${actionLabel}`}
+ />
+ Abort on error
+ </label>
+ </>
+ }
+ />
+ </div>
+ );
+}
diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx
@@ -17,6 +17,7 @@ import {
parseShardField,
type ShardConfigSummary,
} from "../../../../components/ShardControl";
+import { RetryBucketControl } from "../RetryBucketControl";
import { VideoIdList } from "../VideoIdList";
function parseConcurrency(s: string): number | undefined {
@@ -32,6 +33,7 @@ type DiagnosticBucket = {
label: string;
description: string;
ariaLabel: string;
+ retry?: boolean;
};
type Props = {
@@ -43,6 +45,8 @@ type Props = {
totals: { videos: number; transcribed: number; downloaded: number };
availability: AvailabilitySnapshot;
availabilityShard: ShardConfigSummary | null;
+ existingQueues: string[];
+ downloadDefaultQueueKey: string;
};
export function DiagnosticsStage({
@@ -54,6 +58,8 @@ export function DiagnosticsStage({
totals,
availability,
availabilityShard,
+ existingQueues,
+ downloadDefaultQueueKey,
}: Props) {
const buckets: DiagnosticBucket[] = [
{
@@ -62,6 +68,7 @@ export function DiagnosticsStage({
description:
"Video directory exists without yt-dlp metadata; index will skip it.",
ariaLabel: "missing metadata",
+ retry: true,
},
{
ids: missingFromArchiveIds,
@@ -69,6 +76,7 @@ export function DiagnosticsStage({
description:
"Video ID is in the archive but its data directory is missing.",
ariaLabel: "archived without dir",
+ retry: true,
},
{
ids: duplicateDirIds,
@@ -127,7 +135,12 @@ export function DiagnosticsStage({
desc="Probe yt-dlp to detect videos that have been deleted, made private, or restricted by the uploader. Results write to channels/<slug>/data/<id>/availability.json."
/>
<AvailabilityButtons slug={slug} shard={availabilityShard} />
- <AvailabilitySummary slug={slug} availability={availability} />
+ <AvailabilitySummary
+ slug={slug}
+ availability={availability}
+ existingQueues={existingQueues}
+ downloadDefaultQueueKey={downloadDefaultQueueKey}
+ />
</div>
<div className="flex flex-col gap-2" aria-label="channel health">
<div>
@@ -166,6 +179,15 @@ export function DiagnosticsStage({
emptyMessage="None"
itemAriaLabel={(id) => `${b.ariaLabel} ${id}`}
/>
+ {b.retry ? (
+ <RetryBucketControl
+ slug={slug}
+ ids={b.ids}
+ actionLabel={b.ariaLabel}
+ defaultQueueKey={downloadDefaultQueueKey}
+ existingQueues={existingQueues}
+ />
+ ) : null}
</div>
))}
</div>
@@ -390,12 +412,24 @@ const AVAILABILITY_LIST_STATUSES: Availability[] = [
"error",
];
+// Statuses where a retry-download is meaningful. needs_auth and error are the
+// only availability buckets not in EXCLUDED_FROM_DOWNLOAD (see availability.ts);
+// the others would be silent no-ops because the downloader filters them.
+const AVAILABILITY_RETRY_STATUSES: ReadonlyArray<Availability> = [
+ "needs_auth",
+ "error",
+];
+
function AvailabilitySummary({
slug,
availability,
+ existingQueues,
+ downloadDefaultQueueKey,
}: {
slug: string;
availability: AvailabilitySnapshot;
+ existingQueues: string[];
+ downloadDefaultQueueKey: string;
}) {
const total =
AVAILABILITY_VALUES.reduce(
@@ -462,6 +496,15 @@ function AvailabilitySummary({
emptyMessage="None"
itemAriaLabel={(id) => `availability ${v} ${id}`}
/>
+ {AVAILABILITY_RETRY_STATUSES.includes(v) && (
+ <RetryBucketControl
+ slug={slug}
+ ids={availability.byStatus[v]}
+ actionLabel={AVAILABILITY_LABELS[v].toLowerCase()}
+ defaultQueueKey={downloadDefaultQueueKey}
+ existingQueues={existingQueues}
+ />
+ )}
</div>
))}
</div>
diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx
@@ -2,6 +2,7 @@
import { useState } from "react";
import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog";
+import type { ExcludedFromDownload } from "yt-dlp-transcript-common/controller/channelSnapshot";
import { QueueControl } from "../../../../components/QueueControl";
import {
ShardControl,
@@ -14,6 +15,7 @@ import {
downloadMissingAction,
downloadMissingSubsAction,
} from "../../pipelineActions";
+import { RetryBucketControl } from "../RetryBucketControl";
import { VideoIdList } from "../VideoIdList";
type Props = {
@@ -22,6 +24,7 @@ type Props = {
defaultQueueKey: string;
existingQueues: string[];
undownloadedIds: string[];
+ excludedFromDownload: ExcludedFromDownload;
noTranscriptIds: string[];
missingShard: ShardConfigSummary | null;
};
@@ -32,6 +35,7 @@ export function DownloadStage({
defaultQueueKey,
existingQueues,
undownloadedIds,
+ excludedFromDownload,
noTranscriptIds,
missingShard,
}: Props) {
@@ -59,6 +63,10 @@ export function DownloadStage({
return (
<div className="flex flex-col gap-6">
<UndownloadedList slug={slug} ids={undownloadedIds} />
+ <ExcludedFromDownloadCard
+ slug={slug}
+ excluded={excludedFromDownload}
+ />
<div className="flex flex-col gap-2">
<Heading
title="Download from playlist"
@@ -194,7 +202,102 @@ export function DownloadStage({
}
/>
</div>
- <NoTranscriptList slug={slug} ids={noTranscriptIds} />
+ <NoTranscriptList
+ slug={slug}
+ ids={noTranscriptIds}
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ />
+ </div>
+ );
+}
+
+function ExcludedFromDownloadCard({
+ slug,
+ excluded,
+}: {
+ slug: string;
+ excluded: ExcludedFromDownload;
+}) {
+ const total =
+ excluded.membersOnly.length +
+ excluded.deleted.length +
+ excluded.private.length;
+ if (total === 0) return null;
+ type Section = {
+ key: "membersOnly" | "deleted" | "private";
+ label: string;
+ ariaPrefix: string;
+ ids: string[];
+ };
+ const allSections: Section[] = [
+ {
+ key: "membersOnly",
+ label: "Members-only",
+ ariaPrefix: "members only",
+ ids: excluded.membersOnly,
+ },
+ {
+ key: "deleted",
+ label: "Deleted",
+ ariaPrefix: "deleted",
+ ids: excluded.deleted,
+ },
+ {
+ key: "private",
+ label: "Private",
+ ariaPrefix: "private",
+ ids: excluded.private,
+ },
+ ];
+ const sections = allSections.filter((s) => s.ids.length > 0);
+ return (
+ <div
+ aria-label="excluded from download"
+ className="flex flex-col gap-2 rounded border border-zinc-200 dark:border-zinc-800 p-3"
+ >
+ <div>
+ <h4 className="text-sm font-semibold">
+ Excluded from download ({total})
+ </h4>
+ <p className="text-xs text-zinc-500">
+ Permanently-unavailable videos (members-only, deleted, or private)
+ are skipped by the download flows. To retry one after it becomes
+ public again, re-check its availability in Diagnostics.
+ </p>
+ </div>
+ <div className="flex flex-wrap gap-x-3 gap-y-1 text-xs text-zinc-600 dark:text-zinc-400">
+ {sections.map((s) => (
+ <span key={s.key} aria-label={`excluded ${s.ariaPrefix} count`}>
+ <span className="font-medium text-zinc-800 dark:text-zinc-200">
+ {s.label}
+ </span>{" "}
+ {s.ids.length}
+ </span>
+ ))}
+ </div>
+ <div className="grid grid-cols-1 lg:grid-cols-2 gap-3">
+ {sections.map((s) => (
+ <details
+ key={s.key}
+ className="rounded border border-zinc-200 dark:border-zinc-800 p-2"
+ >
+ <summary className="cursor-pointer text-xs font-semibold">
+ {s.label} ({s.ids.length})
+ </summary>
+ <div className="mt-2">
+ <VideoIdList
+ slug={slug}
+ ids={s.ids}
+ ariaLabel={`excluded ${s.ariaPrefix} list`}
+ emptyAriaLabel={`excluded ${s.ariaPrefix} empty`}
+ emptyMessage="None"
+ itemAriaLabel={(id) => `excluded ${s.ariaPrefix} ${id}`}
+ />
+ </div>
+ </details>
+ ))}
+ </div>
</div>
);
}
@@ -217,7 +320,17 @@ function UndownloadedList({ slug, ids }: { slug: string; ids: string[] }) {
);
}
-function NoTranscriptList({ slug, ids }: { slug: string; ids: string[] }) {
+function NoTranscriptList({
+ slug,
+ ids,
+ defaultQueueKey,
+ existingQueues,
+}: {
+ slug: string;
+ ids: string[];
+ defaultQueueKey: string;
+ existingQueues: string[];
+}) {
if (ids.length === 0) return null;
return (
<div className="flex flex-col gap-2 rounded border border-zinc-200 dark:border-zinc-800 p-3">
@@ -237,6 +350,13 @@ function NoTranscriptList({ slug, ids }: { slug: string; ids: string[] }) {
emptyMessage="None"
itemAriaLabel={(id) => `no transcript or download ${id}`}
/>
+ <RetryBucketControl
+ slug={slug}
+ ids={ids}
+ actionLabel="missing transcript and no download"
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ />
</div>
);
}
diff --git a/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx b/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx
@@ -18,6 +18,7 @@ import {
clearFailedTranscriptionsAction,
transcribeMissingAction,
} from "../../whisperActions";
+import { RetryBucketControl } from "../RetryBucketControl";
import { VideoIdList } from "../VideoIdList";
type AudioFormatChoice = AudioFormat | "any";
@@ -37,6 +38,11 @@ type Props = {
downloadedNoTranscriptIds: string[];
defaultConcurrency: number;
defaultQueueKey: string;
+ // Default queue used when the user clicks Retry on the
+ // downloadedNoTranscript bucket. The retry triggers a download flow, so
+ // it should run on the platform's download queue (not the transcription
+ // queue) to avoid hammering the source.
+ downloadDefaultQueueKey: string;
missingShard: ShardConfigSummary | null;
};
@@ -47,6 +53,7 @@ export function TranscribeStage({
downloadedNoTranscriptIds,
defaultConcurrency,
defaultQueueKey,
+ downloadDefaultQueueKey,
missingShard,
}: Props) {
const channelQueueKey = `channel:${slug}`;
@@ -133,6 +140,8 @@ export function TranscribeStage({
<DownloadedNoTranscriptList
slug={slug}
ids={downloadedNoTranscriptIds}
+ downloadDefaultQueueKey={downloadDefaultQueueKey}
+ existingQueues={existingQueues}
/>
</div>
<FailedTranscriptionsSection
@@ -224,9 +233,13 @@ function FailedTranscriptionsSection({
function DownloadedNoTranscriptList({
slug,
ids,
+ downloadDefaultQueueKey,
+ existingQueues,
}: {
slug: string;
ids: string[];
+ downloadDefaultQueueKey: string;
+ existingQueues: string[];
}) {
if (ids.length === 0) return null;
return (
@@ -236,7 +249,9 @@ function DownloadedNoTranscriptList({
Downloaded but not transcribed ({ids.length})
</h4>
<p className="text-xs text-zinc-500">
- Audio is on disk but no .vtt or transcript.json yet.
+ Audio is on disk but no .vtt or transcript.json yet. The retry runs
+ a download pass — pick “youtube” handling to fetch a
+ missing VTT, or leave default to verify nothing else is needed.
</p>
</div>
<VideoIdList
@@ -247,6 +262,13 @@ function DownloadedNoTranscriptList({
emptyMessage="None"
itemAriaLabel={(id) => `downloaded without transcript ${id}`}
/>
+ <RetryBucketControl
+ slug={slug}
+ ids={ids}
+ actionLabel="downloaded but not transcribed"
+ defaultQueueKey={downloadDefaultQueueKey}
+ existingQueues={existingQueues}
+ />
</div>
);
}
diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts
@@ -1,5 +1,8 @@
import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig";
-import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot";
+import {
+ normalizeExcludedFromDownload,
+ type ChannelSnapshot,
+} from "yt-dlp-transcript-common/controller/channelSnapshot";
import type { JobRecord } from "yt-dlp-transcript-common/jobs/registry";
export type SnapshotBuckets = ChannelSnapshot["buckets"];
@@ -95,6 +98,17 @@ export function computeStageStatuses(
const buckets = normalizeBuckets(snapshot.buckets);
const undownloadedIds = snapshot.undownloadedIds ?? [];
+ const excludedFromDownload = normalizeExcludedFromDownload(
+ snapshot.excludedFromDownload,
+ );
+ const excludedDownloadIds = new Set<string>([
+ ...excludedFromDownload.membersOnly,
+ ...excludedFromDownload.deleted,
+ ...excludedFromDownload.private,
+ ]);
+ const actionableNoTranscript = buckets.noTranscript.filter(
+ (id) => !excludedDownloadIds.has(id),
+ );
const runningByStage = new Set<StageId>();
for (const job of runningJobs) {
@@ -106,7 +120,8 @@ export function computeStageStatuses(
const transcodeApplies =
config.handling === "transcribe" && !!config.audioFormat;
- const downloadPending = undownloadedIds.length + buckets.noTranscript.length;
+ const downloadPending =
+ undownloadedIds.length + actionableNoTranscript.length;
const transcodePending = transcodeApplies ? buckets.untranscoded.length : 0;
const transcodeFailed = transcodeApplies ? failedTranscodingIds.length : 0;
const transcribePending = buckets.downloadedNoTranscript.length;
@@ -157,10 +172,10 @@ export function computeStageStatuses(
if (undownloadedIds.length > 0) {
downloadParts.push(pluralize(undownloadedIds.length, "undownloaded"));
}
- if (buckets.noTranscript.length > 0) {
+ if (actionableNoTranscript.length > 0) {
downloadParts.push(
pluralize(
- buckets.noTranscript.length,
+ actionableNoTranscript.length,
"dir missing transcript & audio",
"dirs missing transcript & audio",
),
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -5,6 +5,7 @@ import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"
import {
generateChannelSnapshot,
normalizeAvailability,
+ normalizeExcludedFromDownload,
readChannelSnapshot,
} from "yt-dlp-transcript-common/controller/channelSnapshot";
import { loadFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions";
@@ -85,7 +86,7 @@ export default async function ChannelDetailPage({
// The retry-failures panel is interactive (a click mutates the file), so
// read it fresh on every render — the snapshot bucket only reflects state
// at refresh time.
- const failedVideoIds = await loadFailedTranscriptions(paths, slug);
+ const rawFailedVideoIds = await loadFailedTranscriptions(paths, slug);
const failedTranscodingIds = await loadFailedTranscodings(paths, slug);
const summarize = (c: ShardConfig | null) =>
@@ -106,6 +107,20 @@ export default async function ChannelDetailPage({
const buckets = normalizeBuckets(snapshot.buckets);
const undownloadedIds = snapshot.undownloadedIds ?? [];
+ const excludedFromDownload = normalizeExcludedFromDownload(
+ snapshot.excludedFromDownload,
+ );
+ // Hide failed-transcription entries whose video can no longer be acted on
+ // (members_only / deleted / private). Same rationale as the snapshot's
+ // bucket filter; this path is fresh-loaded so it needs its own pass.
+ const excludedDownloadIds = new Set<string>([
+ ...excludedFromDownload.membersOnly,
+ ...excludedFromDownload.deleted,
+ ...excludedFromDownload.private,
+ ]);
+ const failedVideoIds = rawFailedVideoIds.filter(
+ (id) => !excludedDownloadIds.has(id),
+ );
const availability = normalizeAvailability(snapshot.availability);
const stages = computeStageStatuses({
snapshot,
@@ -156,6 +171,7 @@ export default async function ChannelDetailPage({
defaultQueueKey={platformDefaultQueueKey}
existingQueues={existingQueues}
undownloadedIds={undownloadedIds}
+ excludedFromDownload={excludedFromDownload}
noTranscriptIds={buckets.noTranscript}
missingShard={downloadMissingShard}
/>
@@ -168,6 +184,7 @@ export default async function ChannelDetailPage({
downloadedNoTranscriptIds={buckets.downloadedNoTranscript}
defaultConcurrency={paths.parallelTranscribeLimit}
defaultQueueKey={TRANSCRIPTION_QUEUE}
+ downloadDefaultQueueKey={platformDefaultQueueKey}
missingShard={transcribeMissingShard}
/>
),
@@ -189,6 +206,8 @@ export default async function ChannelDetailPage({
totals={snapshot.totals}
availability={availability}
availabilityShard={availabilityShard}
+ existingQueues={existingQueues}
+ downloadDefaultQueueKey={platformDefaultQueueKey}
/>
),
danger: (
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -1,7 +1,11 @@
"use server";
import { revalidatePath } from "next/cache";
-import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig";
+import {
+ HANDLING_VALUES,
+ type ChannelConfig,
+ type ChannelHandling,
+} from "yt-dlp-transcript-common/lib/channelConfig";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
detectPlatform,
@@ -25,7 +29,8 @@ async function runPipelineAction(
| "download-from-playlist"
| "sync"
| "download-missing"
- | "download-missing-subs",
+ | "download-missing-subs"
+ | "retry-bucket",
kind: string,
queueKey?: string,
options?: {
@@ -33,6 +38,8 @@ async function runPipelineAction(
abortOnError?: boolean;
shardTotal?: number;
shardIndex?: number;
+ bucketIds?: ReadonlyArray<string>;
+ handlingOverride?: ChannelHandling;
},
): Promise<StreamActionResult> {
const paths = getPaths();
@@ -61,6 +68,8 @@ async function runPipelineAction(
abortOnError: options?.abortOnError,
shardTotal: options?.shardTotal,
shardIndex: options?.shardIndex,
+ bucketIds: options?.bucketIds,
+ handlingOverride: options?.handlingOverride,
});
revalidatePath(`/channels/${slug}`);
revalidatePath("/channels");
@@ -126,3 +135,32 @@ export async function downloadMissingSubsAction(
{ abortOnError },
);
}
+
+export async function retryBucketAction(
+ slug: string,
+ bucketIds: string[],
+ queueKey?: string,
+ abortOnError?: boolean,
+ handlingOverride?: string,
+): Promise<StreamActionResult> {
+ if (!Array.isArray(bucketIds) || bucketIds.length === 0) {
+ return { ok: false, error: "No video IDs supplied for retry." };
+ }
+ let handling: ChannelHandling | undefined;
+ if (handlingOverride) {
+ if (!(HANDLING_VALUES as ReadonlyArray<string>).includes(handlingOverride)) {
+ return {
+ ok: false,
+ error: `Invalid handling override: ${handlingOverride}`,
+ };
+ }
+ handling = handlingOverride as ChannelHandling;
+ }
+ return runPipelineAction(
+ slug,
+ "retry-bucket",
+ "retry-bucket",
+ queueKey,
+ { bucketIds, handlingOverride: handling, abortOnError },
+ );
+}
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -16,8 +16,8 @@ import { TextFilePreview } from "./TextFilePreview";
import {
deleteVideoDirAction,
deleteVideoFileAction,
+ downloadVideoPipelineAction,
markVideoUntranscribableAction,
- redownloadVideoAction,
transcodeAudioAction,
transcribeOneAction,
whisperVideoAction,
@@ -35,7 +35,6 @@ type Props = {
videoId: string;
files: VideoFile[];
handling: ChannelHandling;
- audioFormat: AudioFormat;
defaultQueueKey: string;
existingQueues: string[];
downloadOutcome: DownloadOutcomeRecord | null;
@@ -72,6 +71,17 @@ function isAudioFile(name: string): boolean {
return AUDIO_EXTS.has(fileExt(name));
}
+// Files we can transcode FROM: any media file ffmpeg can demux to audio.
+// Includes video containers so a kept source (e.g. audio.mp4 when
+// keepSourceVideo=true) can be re-extracted to audio.<targetFormat> to
+// overwrite a truncated or corrupt prior extraction.
+function isTranscodeSource(name: string): boolean {
+ if (!name.startsWith("audio.")) return false;
+ if (name.startsWith("audio.tmp-")) return false;
+ const ext = fileExt(name);
+ return AUDIO_EXTS.has(ext) || VIDEO_EXTS.has(ext);
+}
+
function audioExt(name: string): string {
const dot = name.lastIndexOf(".");
return dot >= 0 ? name.slice(dot + 1) : "";
@@ -96,24 +106,17 @@ function formatSize(bytes: number): string {
return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`;
}
-function parseExtraArgs(raw: string): string[] {
- return raw
- .split(/\s+/)
- .map((s) => s.trim())
- .filter(Boolean);
-}
-
export function VideoPanel({
slug,
videoId,
files,
handling,
- audioFormat,
defaultQueueKey,
existingQueues,
downloadOutcome,
}: Props) {
const audioFiles = files.filter((f) => isAudioFile(f.name));
+ const transcodeSources = files.filter((f) => isTranscodeSource(f.name));
const hasTranscriptJson = files.some((f) => f.name === "transcript.json");
const hasYtVtt = files.some((f) => f.name === "transcript.en.vtt");
const hasTranscript = hasTranscriptJson || hasYtVtt;
@@ -121,17 +124,17 @@ export function VideoPanel({
const downloadSummary = noAudio
? handling === "youtube"
- ? "No audio on disk — download to enable whisper."
+ ? "No audio on disk — run the pipeline (VTT first, whisper if needed)."
: "Audio not yet downloaded."
: `${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} on disk.`;
const transcribeSummary = hasTranscript
? "Transcript present."
: noAudio
- ? "No audio yet — download first or use Whisper transcribe (combo)."
+ ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript."
: "Audio ready, no transcript.";
const transcodeSummary =
- audioFiles.length > 0
- ? `Convert ${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} to another format.`
+ transcodeSources.length > 0
+ ? `Convert ${transcodeSources.length} source file${transcodeSources.length === 1 ? "" : "s"} to another format.`
: "Nothing to transcode.";
return (
@@ -148,14 +151,13 @@ export function VideoPanel({
slug={slug}
videoId={videoId}
handling={handling}
- defaultFormat={audioFormat}
defaultQueueKey={defaultQueueKey}
existingQueues={existingQueues}
noAudio={noAudio}
/>
</PipelineStageCard>
- {audioFiles.length > 0 && (
+ {transcodeSources.length > 0 && (
<PipelineStageCard
id="transcode"
title="Transcode"
@@ -164,7 +166,7 @@ export function VideoPanel({
tone="neutral"
>
<div className="flex flex-col gap-4">
- {audioFiles.map((f) => (
+ {transcodeSources.map((f) => (
<PerFileTranscodeRow
key={f.name}
slug={slug}
@@ -185,14 +187,6 @@ export function VideoPanel({
tone={hasTranscript ? "ok" : noAudio ? "neutral" : "attention"}
>
<div className="flex flex-col gap-4">
- {!hasTranscript && (
- <WhisperVideoSection
- slug={slug}
- videoId={videoId}
- existingQueues={existingQueues}
- noAudio={noAudio}
- />
- )}
{audioFiles.map((f) => (
<PerFileTranscribeRow
key={f.name}
@@ -241,11 +235,12 @@ export function VideoPanel({
);
}
+type DownloadMode = "pipeline" | "whisper";
+
function RedownloadSection({
slug,
videoId,
handling,
- defaultFormat,
defaultQueueKey,
existingQueues,
noAudio,
@@ -253,62 +248,62 @@ function RedownloadSection({
slug: string;
videoId: string;
handling: ChannelHandling;
- defaultFormat: AudioFormat;
defaultQueueKey: string;
existingQueues: string[];
noAudio: boolean;
}) {
- const [format, setFormat] = useState<AudioFormat>(defaultFormat);
- const [extraArgsRaw, setExtraArgsRaw] = useState("");
+ const [mode, setMode] = useState<DownloadMode>("pipeline");
const [queueKey, setQueueKey] = useState(defaultQueueKey);
- const desc = noAudio
- ? handling === "youtube"
- ? "YouTube channels normally only fetch auto-subs. Use this when the video has no auto-subs and you want to transcribe it manually."
- : "Fetch the audio for this video using yt-dlp."
- : "Re-fetch the audio for this video. The existing audio file is kept on disk; yt-dlp may overwrite it with the new one if names collide.";
- const actionLabel = `${noAudio ? "Download" : "Redownload"} audio for ${videoId}`;
+
+ const desc = (() => {
+ if (mode === "whisper") {
+ return noAudio
+ ? "Download the audio and run whisper-cli, in one job. Skips any VTT attempt."
+ : "Run whisper-cli on this video's audio. Skips the download phase since audio is already on disk.";
+ }
+ if (handling === "youtube") {
+ return noAudio
+ ? "Try the VTT download first; fall back to audio + whisper if no subs are available."
+ : "Re-run the pipeline: try VTT, then fall back to audio + whisper if needed.";
+ }
+ return noAudio
+ ? "Download audio (and run any channel-configured audio checks)."
+ : "Re-run the download pipeline for this video.";
+ })();
+
+ const verb = mode === "whisper"
+ ? "Audio + Whisper"
+ : noAudio
+ ? "Run download pipeline"
+ : "Re-run download pipeline";
+ const runningLabel = mode === "whisper"
+ ? "Transcribing…"
+ : "Running pipeline…";
+ const actionLabel = `${verb} for ${videoId}`;
+ const trigger = mode === "whisper"
+ ? () => whisperVideoAction(slug, videoId, queueKey)
+ : () => downloadVideoPipelineAction(slug, videoId, queueKey);
+
return (
<div className="flex flex-col gap-3">
<p className="text-sm text-zinc-500">{desc}</p>
- <div className="flex flex-col sm:flex-row sm:flex-wrap sm:items-center gap-3">
- <label className="flex items-center gap-2 text-sm">
- Format
- <select
- value={format}
- onChange={(e) => setFormat(e.target.value as AudioFormat)}
- aria-label="redownload audio format"
- className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm"
- >
- {AUDIO_FORMAT_VALUES.map((f) => (
- <option key={f} value={f}>
- {f}
- </option>
- ))}
- </select>
- </label>
- <label className="flex flex-col sm:flex-row sm:items-center gap-2 text-sm flex-1 sm:min-w-[16rem]">
- <span>Extra yt-dlp args</span>
- <input
- type="text"
- value={extraArgsRaw}
- onChange={(e) => setExtraArgsRaw(e.target.value)}
- aria-label="redownload extra yt-dlp args"
- placeholder="--cookies-from-browser firefox --quiet"
- className="flex-1 rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono min-w-0"
- />
- </label>
- </div>
+ <label className="flex items-center gap-2 text-sm">
+ Mode
+ <select
+ value={mode}
+ onChange={(e) => setMode(e.target.value as DownloadMode)}
+ aria-label="download mode"
+ className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm"
+ >
+ <option value="pipeline">Designated pipeline</option>
+ <option value="whisper">Audio + Whisper (skip pipeline)</option>
+ </select>
+ </label>
<StreamActionLog
- trigger={() =>
- redownloadVideoAction(slug, videoId, {
- audioFormat: format,
- extraArgs: parseExtraArgs(extraArgsRaw),
- queueKey,
- })
- }
+ trigger={trigger}
cancelAction={cancelJobAction}
- buttonLabel={`${noAudio ? "Download" : "Redownload"} audio (${format})`}
- runningLabel={noAudio ? "Downloading…" : "Redownloading…"}
+ buttonLabel={verb}
+ runningLabel={runningLabel}
label={actionLabel}
extraControls={
<QueueControl
@@ -324,45 +319,6 @@ function RedownloadSection({
);
}
-function WhisperVideoSection({
- slug,
- videoId,
- existingQueues,
- noAudio,
-}: {
- slug: string;
- videoId: string;
- existingQueues: string[];
- noAudio: boolean;
-}) {
- const [queueKey, setQueueKey] = useState("");
- const desc = noAudio
- ? "Download the audio and run whisper-cli, in one job. Useful when a YouTube video has no auto-subs."
- : "Run whisper-cli on this video's audio. Skips the download phase since audio is already on disk.";
- const actionLabel = `Whisper transcribe ${videoId}`;
- return (
- <div className="flex flex-col gap-2">
- <Heading title="Whisper transcribe" desc={desc} />
- <StreamActionLog
- trigger={() => whisperVideoAction(slug, videoId, queueKey)}
- cancelAction={cancelJobAction}
- buttonLabel="Whisper transcribe"
- runningLabel="Transcribing…"
- label={actionLabel}
- extraControls={
- <QueueControl
- value={queueKey}
- onChange={setQueueKey}
- defaultQueueKey=""
- existingQueues={existingQueues}
- actionLabel={actionLabel}
- />
- }
- />
- </div>
- );
-}
-
function PerFileTranscribeRow({
slug,
videoId,
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -167,7 +167,6 @@ export default async function VideoDetailPage({
videoId={id}
files={dirData.files}
handling={config.handling}
- audioFormat={config.audioFormat ?? "mp3"}
defaultQueueKey={defaultQueueKey}
existingQueues={existingQueues}
downloadOutcome={downloadOutcome}
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -19,6 +19,8 @@ import { pruneFailedTranscriptions } from "yt-dlp-transcript-common/controller/f
import { transcodeAudio } from "yt-dlp-transcript-common/controller/transcode";
import { transcribeOneVideo } from "yt-dlp-transcript-common/controller/transcribeOne";
import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import {
runManagedFunction,
@@ -111,19 +113,11 @@ export async function transcribeOneAction(
});
}
-export async function redownloadVideoAction(
+export async function downloadVideoPipelineAction(
slug: string,
videoId: string,
- options: {
- audioFormat?: AudioFormat;
- extraArgs?: string[];
- queueKey?: string;
- } = {},
+ queueKey?: string,
): Promise<StreamActionResult> {
- const { audioFormat, extraArgs, queueKey } = options;
- if (audioFormat && !AUDIO_FORMAT_VALUES.includes(audioFormat)) {
- return { ok: false, error: `Unsupported audio format: ${audioFormat}` };
- }
const r = await loadConfigOrError(slug);
if (!r.ok) return r;
const paths = getPaths();
@@ -135,23 +129,24 @@ export async function redownloadVideoAction(
"Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.",
};
}
+ const settings = getSettings();
return runManagedFunction({
- kind: "download-one-audio",
+ kind: "download-one-pipeline",
queueKey: videoQueueKey(r.config, queueKey),
paths,
channelSlug: slug,
videoId,
fn: async (onLog, signal) => {
- await runYtdlp({
+ await downloadOneManaged({
channelSlug: slug,
- mode: "download-one-audio",
channelConfig: r.config,
paths,
+ videoUrl: url,
onLog,
signal,
- singleVideoUrl: url,
- audioFormatOverride: audioFormat,
- extraYtdlpArgs: extraArgs,
+ globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
+ appendArchive: true,
});
revalidatePath(`/channels/${slug}/videos/${videoId}`);
revalidatePath(`/channels/${slug}`);
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -29,6 +29,8 @@ export async function saveSettingsAction(
const sleepRaw = String(
formData.get("sleepBetweenDownloadsSeconds") ?? "",
).trim();
+ const inlineTranscribeOnFallback =
+ formData.get("inlineTranscribeOnFallback") === "on";
const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? "");
const transcribeArgs = transcribeArgsRaw
.split("\n")
@@ -86,6 +88,7 @@ export async function saveSettingsAction(
transcribeArgs,
cookiesFromBrowser,
sleepBetweenDownloadsSeconds: sleepParsed,
+ inlineTranscribeOnFallback,
};
await writeSettings(next);
revalidatePath("/settings");
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -100,6 +100,26 @@ export function SettingsForm({ initial }: Props) {
type="number"
hint="Pause inserted between per-video yt-dlp invocations in managed batch downloads (download-from-playlist, download-missing). 0 disables. Default 10s. Each channel can override this in its Advanced settings."
/>
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ name="inlineTranscribeOnFallback"
+ defaultChecked={initial.inlineTranscribeOnFallback}
+ className="mt-1"
+ />
+ <span className="flex flex-col gap-1">
+ <span className="font-medium">
+ Transcribe immediately after no-subs fallback
+ </span>
+ <span className="text-xs text-zinc-500">
+ When the managed downloader falls back to audio+whisper for a
+ video without captions, run whisper inline instead of leaving
+ the audio for the next “Transcribe missing” pass.
+ Off by default so batch downloads finish faster and whisper
+ can run in parallel.
+ </span>
+ </span>
+ </label>
<div className="flex items-center gap-3">
<button
type="submit"
diff --git a/editor/e2e/exclude-from-counts.spec.ts b/editor/e2e/exclude-from-counts.spec.ts
@@ -0,0 +1,102 @@
+import { mkdir, writeFile } from "node:fs/promises";
+import { dirname } from "node:path";
+import { test, expect } from "@playwright/test";
+import { resetData, resolvePath } from "./helpers";
+
+const CHANNEL = "test-transcribe";
+const FIXTURE = "one-transcribe-channel-with-audio";
+
+type AvailabilityStatus =
+ | "public"
+ | "unlisted"
+ | "private"
+ | "members_only"
+ | "needs_auth"
+ | "deleted"
+ | "error";
+
+async function writeAvailability(id: string, status: AvailabilityStatus) {
+ const path = resolvePath(
+ `test-transcripts/channels/${CHANNEL}/data/${id}/availability.json`,
+ );
+ await mkdir(dirname(path), { recursive: true });
+ await writeFile(
+ path,
+ JSON.stringify({
+ checkedAt: new Date().toISOString(),
+ availability: status,
+ webpageUrl: `https://www.youtube.com/watch?v=${id}`,
+ }),
+ );
+}
+
+test("members-only video is hidden from the Missing metadata bucket", async ({
+ page,
+}) => {
+ // Fixture has vidA, vidB, vidC — each with audio.m4a but no metadata.info.json.
+ // Without filtering, all three would show as "Missing metadata.info.json (3)".
+ await resetData(FIXTURE);
+ await writeAvailability("vidA", "members_only");
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ await expect(
+ page.getByRole("heading", { name: /Missing metadata\.info\.json \(2\)/ }),
+ ).toBeVisible();
+
+ const list = page.getByLabel("missing metadata list");
+ await expect(list.getByLabel("missing metadata vidA")).toHaveCount(0);
+ await expect(list.getByLabel("missing metadata vidB")).toBeVisible();
+ await expect(list.getByLabel("missing metadata vidC")).toBeVisible();
+});
+
+test("members-only video is hidden from the Failed transcriptions list", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await writeAvailability("vidA", "members_only");
+ await writeFile(
+ resolvePath(`test-transcripts/channels/${CHANNEL}/failed-transcriptions`),
+ "vidA\nvidB\n",
+ );
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ await expect(
+ page.getByRole("heading", { name: /Failed transcriptions \(1\)/ }),
+ ).toBeVisible();
+
+ const list = page.getByLabel("failed transcriptions list");
+ await expect(list.getByLabel("failed transcription vidA")).toHaveCount(0);
+ await expect(list.getByLabel("failed transcription vidB")).toBeVisible();
+});
+
+test("stage badge counts exclude members-only videos", async ({ page }) => {
+ await resetData(FIXTURE);
+ // vidA: members_only — should drop out of both buckets.
+ // vidB: stays in both buckets (no availability sidecar).
+ await writeAvailability("vidA", "members_only");
+ await writeFile(
+ resolvePath(`test-transcripts/channels/${CHANNEL}/failed-transcriptions`),
+ "vidA\nvidB\n",
+ );
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ // Diagnostics pending = noMetadata (2: vidB, vidC) + missingFromArchive + duplicateDirs.
+ // The fixture has no archive or duplicates, so the count is just the
+ // filtered noMetadata count. Both the side-rail and the stage card surface
+ // this label; assert every instance reads "2".
+ const diagPending = page.getByLabel("Diagnostics pending count");
+ await expect(diagPending).not.toHaveCount(0);
+ for (const el of await diagPending.all()) {
+ await expect(el).toHaveText("2");
+ }
+
+ // Transcribe failed = filtered failedVideoIds count (1: vidB).
+ const transcribeFailed = page.getByLabel("Transcribe failed count");
+ await expect(transcribeFailed).not.toHaveCount(0);
+ for (const el of await transcribeFailed.all()) {
+ await expect(el).toHaveText(/^1( failed)?$/);
+ }
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -57,7 +57,7 @@ async function ensureDir(p) {
await mkdir(p, { recursive: true });
}
-async function writeMetadata(videoDir, id) {
+async function writeMetadata(videoDir, id, opts = {}) {
const meta = {
id,
title: `Synthetic ${id}`,
@@ -75,6 +75,16 @@ async function writeMetadata(videoDir, id) {
extractor_key: "Youtube",
webpage_url: `https://www.youtube.com/watch?v=${id}`,
};
+ // For the no-subs-fallback tests we need yt-dlp's metadata to reflect
+ // whether the video advertises caption tracks. The default (no opts)
+ // omits both keys so the rest of the existing suite keeps its behavior.
+ if (opts.livechatOnly) {
+ meta.subtitles = { live_chat: [{ ext: "json", url: "fake://chat" }] };
+ meta.automatic_captions = {};
+ } else if (opts.noCaptions) {
+ meta.subtitles = {};
+ meta.automatic_captions = {};
+ }
await writeFile(
path.join(videoDir, "metadata.info.json"),
JSON.stringify(meta),
@@ -305,10 +315,29 @@ async function modeYoutubeSingleUrlManaged(url) {
`youtube-single:${url}\n`,
);
process.stdout.write(`[fake-ytdlp] managed single-URL ${id}\n`);
+
+ // URL sentinels for no-subs-fallback tests. The sentinels live in the
+ // video id itself so they round-trip through urlIdYouTube and the
+ // managed downloader's URL re-extraction.
+ const lower = url.toLowerCase();
+ const livechatOnly = lower.includes("livechatonly");
+ const noCaptions = lower.includes("nocaptions");
+
if (!existsSync(path.join(videoDir, "metadata.info.json"))) {
- await writeMetadata(videoDir, id);
+ await writeMetadata(videoDir, id, { livechatOnly, noCaptions });
+ }
+ if (livechatOnly) {
+ // yt-dlp writes live_chat as JSON-lines (one continuation per line).
+ // Content shape doesn't matter for these tests; just produce a file.
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"clientId":"fake","action":{"addChatItemAction":{}}}\n',
+ );
+ } else if (noCaptions) {
+ // Simulate "metadata-only" success — no transcript file at all.
+ } else {
+ await writeTranscript(videoDir);
}
- await writeTranscript(videoDir);
process.stdout.write(`DLOM_ARCHIVE youtube ${id}\n`);
process.stdout.write(`[fake-ytdlp] single-url fetched ${id}\n`);
}
@@ -401,10 +430,24 @@ async function main() {
return;
}
- // Single-URL invocation (download-one-audio): no -a, no playlist flags,
- // last positional is the URL.
+ // Single-URL invocation (download-one-audio or no-subs fallback): no
+ // -a, no playlist flags. The fallback path passes --load-info-json
+ // instead of a positional URL — derive the id from the info.json path
+ // (our layout is `data/<id>/metadata.info.json`), so the rest of the
+ // mode stays URL-shaped.
if (audioFmt) {
- const url = lastNonFlag();
+ const infoJsonPath = arg("--load-info-json");
+ let url;
+ if (infoJsonPath) {
+ const id = path.basename(path.dirname(infoJsonPath));
+ url = `https://www.youtube.com/watch?v=${id}`;
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `load-info-json:${infoJsonPath}\n`,
+ );
+ } else {
+ url = lastNonFlag();
+ }
if (!url) {
process.stderr.write(`[fake-ytdlp] single-url mode missing URL\n`);
process.exit(2);
diff --git a/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/config.json b/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/config.json
@@ -0,0 +1,5 @@
+{
+ "handling": "youtube",
+ "name": "Test Live Chat",
+ "url": "https://www.youtube.com/@example/videos"
+}
diff --git a/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/playlist b/editor/e2e/fixtures/test-transcripts/livechat-fallback-channel/channels/test-livechat/playlist
@@ -0,0 +1,3 @@
+https://www.youtube.com/watch?v=livechatonly1
+https://www.youtube.com/watch?v=nocaptions001
+https://www.youtube.com/watch?v=normalvideo1
diff --git a/editor/e2e/job-stream-cancel.spec.ts b/editor/e2e/job-stream-cancel.spec.ts
@@ -0,0 +1,74 @@
+// Regression: when the consumer disconnects from a streaming server action
+// (e.g. user navigates away mid-job), the producer kept calling
+// controller.enqueue on the closed stream and uncaughtException leaked out
+// of the RSC encoder with "Controller is already closed". streamCommand.ts
+// now nullifies the controller and aborts the producer on cancel; this spec
+// guards against the regression.
+
+import { test, expect } from "@playwright/test";
+import { resetData } from "./helpers";
+
+const baseUrl = "http://localhost:3011";
+
+async function readUncaughtCount(): Promise<{
+ uncaught: number;
+ unhandled: number;
+ messages: string[];
+}> {
+ const res = await fetch(`${baseUrl}/api/test/uncaught-count`);
+ return res.json() as Promise<{
+ uncaught: number;
+ unhandled: number;
+ messages: string[];
+ }>;
+}
+
+async function clearUncaughtCount(): Promise<void> {
+ await fetch(`${baseUrl}/api/test/uncaught-count`, { method: "DELETE" });
+}
+
+test("disconnecting from a running job stream does not crash with 'Controller is already closed'", async ({
+ page,
+}) => {
+ test.setTimeout(60_000);
+ await resetData("slow-pipeline-channel");
+ await clearUncaughtCount();
+
+ // Start the slow sync — fake-ytdlp sleeps 30s before producing output, so
+ // the producer is alive and pushing into the stream for the duration.
+ await page.goto("/channels/slow-channel");
+ await page.getByRole("button", { name: "Sync" }).click();
+ await expect(page.getByLabel("Sync output")).toContainText("test-slow", {
+ timeout: 15_000,
+ });
+
+ // Simulate consumer disconnect by navigating to a different page. The
+ // RSC response stream attached to the previous page is torn down, which
+ // cancels our underlying ReadableStream source. The job itself keeps
+ // running on the server (writes to disk) — that's intentional, since
+ // another client could reconnect.
+ await page.goto("/jobs");
+
+ // Cancel the still-running job via the jobs page so the test doesn't
+ // wait the full 30s for fake-ytdlp to wake up. The actual regression
+ // signal is the uncaught-counter check after this — which catches the
+ // "Controller is already closed" error that the prior implementation
+ // produced whenever the source kept enqueueing post-disconnect.
+ const jobRow = page
+ .getByRole("row")
+ .filter({ hasText: "slow-channel" })
+ .first();
+ await expect(jobRow).toContainText("running", { timeout: 10_000 });
+ await jobRow.getByRole("button", { name: /^Cancel$/ }).click();
+ await expect(jobRow).toContainText("cancelled", { timeout: 15_000 });
+
+ // Settle so any post-cancel async cleanup has a chance to fire its error
+ // path before we check the counter.
+ await page.waitForTimeout(1_000);
+
+ const counts = await readUncaughtCount();
+ expect(
+ counts.messages.filter((m) => m.includes("Controller is already closed")),
+ ).toEqual([]);
+ expect(counts.uncaught).toBe(0);
+});
diff --git a/editor/e2e/no-subs-fallback.spec.ts b/editor/e2e/no-subs-fallback.spec.ts
@@ -0,0 +1,156 @@
+import { readFile } from "node:fs/promises";
+import { fileURLToPath } from "node:url";
+import path from "node:path";
+import { test, expect } from "@playwright/test";
+import { pathExists, readJson, resetData, writeSettings } from "./helpers";
+
+const CHANNEL = "test-livechat";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+const here = path.dirname(fileURLToPath(import.meta.url));
+
+type DownloadOutcome = {
+ status: string;
+ attempts: Array<{ kind: string; handling: string }>;
+ fellBackToTranscribe?: boolean;
+};
+
+async function defaultSettings(): Promise<Record<string, unknown>> {
+ const raw = await readFile(
+ path.join(here, "fixtures", "test-settings.default.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as Record<string, unknown>;
+}
+
+test("default off: live_chat-only video downloads audio but defers whisper", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("livechat-fallback-channel");
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ "No subs available for livechatonly1; falling back to audio download + whisper.",
+ { timeout: 60_000 },
+ );
+ await expect(log).toContainText(
+ "Audio downloaded for livechatonly1; skipping inline whisper",
+ { timeout: 60_000 },
+ );
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+
+ expect(
+ await pathExists(
+ `${ROOT}/data/livechatonly1/transcript.live_chat.json`,
+ ),
+ ).toBe(true);
+ expect(await pathExists(`${ROOT}/data/livechatonly1/audio.mp3`)).toBe(true);
+ expect(
+ await pathExists(`${ROOT}/data/livechatonly1/transcript.json`),
+ ).toBe(false);
+
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/livechatonly1/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("ok");
+ expect(outcome.fellBackToTranscribe).toBe(true);
+ expect(
+ outcome.attempts.some(
+ (a) => a.kind === "no-subs-fallback" && a.handling === "transcribe",
+ ),
+ ).toBe(true);
+
+ // The fallback should reuse the metadata.info.json the primary attempt
+ // wrote, not re-query the extractor.
+ const invocations = await readFile(
+ path.join(here, "..", ROOT, "fake-ytdlp.invocations"),
+ "utf8",
+ );
+ expect(invocations).toMatch(
+ /load-info-json:.*data\/livechatonly1\/metadata\.info\.json/,
+ );
+});
+
+test("default off: no-captions video downloads audio but defers whisper", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("livechat-fallback-channel");
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ "No subs available for nocaptions001; falling back to audio download + whisper.",
+ { timeout: 60_000 },
+ );
+ await expect(log).toContainText(
+ "Audio downloaded for nocaptions001; skipping inline whisper",
+ { timeout: 60_000 },
+ );
+ expect(await pathExists(`${ROOT}/data/nocaptions001/audio.mp3`)).toBe(true);
+ expect(
+ await pathExists(`${ROOT}/data/nocaptions001/transcript.json`),
+ ).toBe(false);
+
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/nocaptions001/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("ok");
+ expect(outcome.fellBackToTranscribe).toBe(true);
+});
+
+test("video with real auto-subs does NOT trigger fallback (regression)", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("livechat-fallback-channel");
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+ await expect(log).not.toContainText("No subs available for normalvideo1");
+ expect(
+ await pathExists(`${ROOT}/data/normalvideo1/transcript.en.vtt`),
+ ).toBe(true);
+ expect(await pathExists(`${ROOT}/data/normalvideo1/audio.mp3`)).toBe(false);
+});
+
+test("inlineTranscribeOnFallback=true runs whisper inline after the fallback", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("livechat-fallback-channel");
+ await writeSettings({
+ ...(await defaultSettings()),
+ inlineTranscribeOnFallback: true,
+ });
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ "No subs available for livechatonly1; falling back to audio download + whisper.",
+ { timeout: 60_000 },
+ );
+ await expect(log).not.toContainText(
+ "Audio downloaded for livechatonly1; skipping inline whisper",
+ );
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+
+ expect(await pathExists(`${ROOT}/data/livechatonly1/audio.mp3`)).toBe(true);
+ expect(
+ await pathExists(`${ROOT}/data/livechatonly1/transcript.json`),
+ ).toBe(true);
+
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/livechatonly1/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("ok-auto-transcribed");
+ expect(outcome.fellBackToTranscribe).toBe(true);
+});
diff --git a/editor/e2e/retry-bucket.spec.ts b/editor/e2e/retry-bucket.spec.ts
@@ -0,0 +1,110 @@
+import { mkdir, writeFile } from "node:fs/promises";
+import { dirname } from "node:path";
+import { test, expect } from "@playwright/test";
+import { pathExists, resetData, resolvePath } from "./helpers";
+
+const CHANNEL = "availability-test";
+const FIXTURE = "availability-baseline";
+
+type AvailabilityStatus =
+ | "public"
+ | "unlisted"
+ | "private"
+ | "members_only"
+ | "needs_auth"
+ | "deleted"
+ | "error";
+
+async function writeAvailability(
+ id: string,
+ status: AvailabilityStatus,
+ opts: { error?: string } = {},
+) {
+ const path = resolvePath(
+ `test-transcripts/channels/${CHANNEL}/data/${id}/availability.json`,
+ );
+ await mkdir(dirname(path), { recursive: true });
+ const record = {
+ checkedAt: new Date().toISOString(),
+ availability: status,
+ webpageUrl: `https://www.youtube.com/watch?v=${id}`,
+ ...(opts.error ? { error: opts.error } : {}),
+ };
+ await writeFile(path, JSON.stringify(record));
+}
+
+test("retry control renders for needs_auth and is absent on excluded buckets", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ // Populate buckets directly via sidecars so the test doesn't depend on
+ // running availability checks first.
+ await writeAvailability("vidneedsauth1", "needs_auth");
+ await writeAvailability("viddeleted1", "deleted");
+ await writeAvailability("vidprivate1", "private");
+ await writeAvailability("vidmembers1", "members_only");
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ const needsAuth = page.getByLabel("retry needs auth bucket");
+ await expect(needsAuth).toBeVisible();
+ await expect(
+ needsAuth.getByRole("button", { name: /^Retry \(1\)$/ }),
+ ).toBeVisible();
+
+ // Excluded-from-download statuses must NOT show a retry control even
+ // though their bucket cards render.
+ await expect(page.getByLabel("availability deleted list")).toBeVisible();
+ await expect(page.getByLabel("retry deleted bucket")).toHaveCount(0);
+ await expect(page.getByLabel("retry private bucket")).toHaveCount(0);
+ await expect(page.getByLabel("retry members-only bucket")).toHaveCount(0);
+});
+
+test("retry control renders for the error bucket", async ({ page }) => {
+ await resetData(FIXTURE);
+ // Reuse an existing fixture video for the synthetic error state — the
+ // test only cares about UI wiring, not the cause of the error.
+ await writeAvailability("vidneedsauth1", "error", { error: "transient" });
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ const errorBucket = page.getByLabel("retry error bucket");
+ await expect(errorBucket).toBeVisible();
+ await expect(
+ errorBucket.getByRole("button", { name: /^Retry \(1\)$/ }),
+ ).toBeVisible();
+});
+
+test("clicking retry on needs_auth re-downloads the listed video", async ({
+ page,
+}) => {
+ test.setTimeout(60_000);
+ await resetData(FIXTURE);
+ await writeAvailability("vidneedsauth1", "needs_auth");
+ // retry-bucket reads the channel playlist file to know which URL to fetch.
+ await writeFile(
+ resolvePath(`test-transcripts/channels/${CHANNEL}/playlist`),
+ "https://www.youtube.com/watch?v=vidneedsauth1\n",
+ );
+
+ await page.goto(`/channels/${CHANNEL}`);
+
+ await page
+ .getByLabel("retry needs auth bucket")
+ .getByRole("button", { name: /^Retry \(1\)$/ })
+ .click();
+
+ const log = page.getByLabel("Retry needs auth output");
+ await expect(log).toContainText("Retry allowlist: kept 1 of 1", {
+ timeout: 30_000,
+ });
+ await expect(log).toContainText("Managed download complete", {
+ timeout: 30_000,
+ });
+
+ expect(
+ await pathExists(
+ `test-transcripts/channels/${CHANNEL}/data/vidneedsauth1/transcript.en.vtt`,
+ ),
+ ).toBe(true);
+});
diff --git a/editor/e2e/undownloaded.spec.ts b/editor/e2e/undownloaded.spec.ts
@@ -45,11 +45,9 @@ test("clicking an undownloaded entry lands on a usable video page", async ({
"fake00000002",
);
await expect(
- page.getByRole("button", { name: /^Download audio/ }),
- ).toBeVisible();
- await expect(
- page.getByRole("button", { name: "Whisper transcribe" }),
+ page.getByRole("button", { name: /^Run download pipeline$/ }),
).toBeVisible();
+ await expect(page.getByLabel("download mode")).toBeVisible();
});
test("undownloaded list shrinks after a one-click whisper completes", async ({
@@ -57,9 +55,10 @@ test("undownloaded list shrinks after a one-click whisper completes", async ({
}) => {
await resetData("youtube-with-playlist");
await page.goto("/channels/test-youtube/videos/fake00000002");
- await page.getByRole("button", { name: "Whisper transcribe" }).click();
+ await page.getByLabel("download mode").selectOption("whisper");
+ await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click();
await expect(
- page.getByLabel("Whisper transcribe fake00000002 output"),
+ page.getByLabel("Audio + Whisper for fake00000002 output"),
).toContainText("Transcribe fake00000002 done", { timeout: 30_000 });
expect(
diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts
@@ -106,7 +106,7 @@ test("transcode keeps source and writes audio.<target>", async ({ page }) => {
).toBe(true);
});
-test("download audio for a YouTube video with no auto-subs", async ({
+test("audio + whisper for a YouTube video with no auto-subs", async ({
page,
}) => {
await resetData("one-youtube-channel-with-data");
@@ -128,10 +128,11 @@ test("download audio for a YouTube video with no auto-subs", async ({
await expect(page.getByRole("heading", { level: 1 })).toContainText(
"No-subs video",
);
- await page.getByRole("button", { name: /^Download audio/ }).click();
+ await page.getByLabel("download mode").selectOption("whisper");
+ await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click();
await expect(
- page.getByLabel("Download audio for test1234567 output"),
- ).toContainText("download complete", { timeout: 30_000 });
+ page.getByLabel("Audio + Whisper for test1234567 output"),
+ ).toContainText("Transcribe test1234567 done", { timeout: 30_000 });
expect(
await pathExists(
"test-transcripts/channels/test-youtube/data/test1234567/audio.mp3",
@@ -362,7 +363,7 @@ test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => {
);
// Transcode summary too.
await expect(page.getByLabel("Transcode stage summary")).toContainText(
- "Convert 1 audio file to another format.",
+ "Convert 1 source file to another format.",
);
// Per-file transcode row exists for audio.mp3 only — not for the text files.
await expect(
@@ -376,6 +377,42 @@ test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => {
).toHaveCount(0);
});
+test("Transcode card lists a kept source video (audio.mp4) so it can be re-extracted", async ({
+ page,
+}) => {
+ // Repro: a truncated audio.mp3 next to an intact audio.mp4 kept source.
+ // Pre-fix, audio.mp4 was filtered out of audioFiles, so no Transcode row
+ // appeared and the user could not overwrite the truncated mp3 from the UI.
+ await resetData("one-transcribe-channel-with-audio");
+ const dir = resolvePath(
+ "test-transcripts/channels/test-transcribe/data/vidA",
+ );
+ await writeFile(`${dir}/audio.mp4`, "fake source container");
+ await writeFile(`${dir}/audio.mp3`, "truncated extraction");
+
+ await page.goto("/channels/test-transcribe/videos/vidA");
+
+ // Both source files appear as Transcode rows.
+ await expect(
+ page.getByRole("heading", { name: "Transcode audio.mp4" }),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("heading", { name: "Transcode audio.mp3" }),
+ ).toBeVisible();
+ // Summary counts the m4a fixture file plus the mp4 and mp3 we added.
+ await expect(page.getByLabel("Transcode stage summary")).toContainText(
+ "Convert 3 source files to another format.",
+ );
+
+ // The Transcribe card still only shows real audio (no audio.mp4 row).
+ await expect(
+ page.getByRole("heading", { name: "Transcribe audio.mp4" }),
+ ).toHaveCount(0);
+ await expect(
+ page.getByRole("heading", { name: /^Transcribe audio\.m4a$/ }),
+ ).toBeVisible();
+});
+
test("existing transcript text files are inspectable from the file list", async ({
page,
}) => {
@@ -391,35 +428,3 @@ test("existing transcript text files are inspectable from the file list", async
);
});
-test("Redownload writes audio in the chosen format", async ({ page }) => {
- await resetData("one-youtube-channel-with-data");
- // Stage a video dir whose name matches the URL id so fake-ytdlp lands its
- // output (data/<id>/audio.<fmt>) back into the dir we navigated to. Include
- // a pre-existing audio.m4a so the panel is in "Redownload" mode.
- const dir = resolvePath(
- "test-transcripts/channels/test-youtube/data/test1234567",
- );
- await mkdir(dir, { recursive: true });
- await writeFile(
- `${dir}/metadata.info.json`,
- JSON.stringify({
- id: "test1234567",
- title: "Redownload target",
- webpage_url: "https://www.youtube.com/watch?v=test1234567",
- }),
- );
- await writeFile(`${dir}/audio.m4a`, "stale audio");
- await page.goto("/channels/test-youtube/videos/test1234567");
- await page.getByLabel("redownload audio format").selectOption("mp3");
- await page
- .getByRole("button", { name: /^Redownload audio \(mp3\)$/ })
- .click();
- await expect(
- page.getByLabel("Redownload audio for test1234567 output"),
- ).toContainText("download complete", { timeout: 30_000 });
- expect(
- await pathExists(
- "test-transcripts/channels/test-youtube/data/test1234567/audio.mp3",
- ),
- ).toBe(true);
-});
diff --git a/editor/e2e/whisper-video.spec.ts b/editor/e2e/whisper-video.spec.ts
@@ -7,7 +7,7 @@ test("one-click whisper downloads audio then transcribes a YouTube video", async
}) => {
await resetData("youtube-with-playlist");
// fake00000001 is the seeded "downloaded" entry. Drop its auto-subs so the
- // page treats it as untranscribed and the Whisper button appears.
+ // page treats it as untranscribed.
await rm(
resolvePath(
"test-transcripts/channels/test-youtube/data/fake00000001/transcript.en.vtt",
@@ -15,9 +15,10 @@ test("one-click whisper downloads audio then transcribes a YouTube video", async
);
await page.goto("/channels/test-youtube/videos/fake00000001");
- await page.getByRole("button", { name: "Whisper transcribe" }).click();
+ await page.getByLabel("download mode").selectOption("whisper");
+ await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click();
- const log = page.getByLabel("Whisper transcribe fake00000001 output");
+ const log = page.getByLabel("Audio + Whisper for fake00000001 output");
await expect(log).toContainText("downloading", { timeout: 30_000 });
await expect(log).toContainText("Transcribe fake00000001 done", {
timeout: 30_000,
@@ -41,9 +42,10 @@ test("one-click whisper skips download when audio is already on disk", async ({
await resetData("one-transcribe-channel-with-audio");
await page.goto("/channels/test-transcribe/videos/vidA");
- await page.getByRole("button", { name: "Whisper transcribe" }).click();
+ await page.getByLabel("download mode").selectOption("whisper");
+ await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click();
- const log = page.getByLabel("Whisper transcribe vidA output");
+ const log = page.getByLabel("Audio + Whisper for vidA output");
await expect(log).toContainText("skipping download", { timeout: 30_000 });
await expect(log).toContainText("Transcribe vidA done", { timeout: 30_000 });
@@ -61,13 +63,3 @@ test("one-click whisper skips download when audio is already on disk", async ({
const invocations = await readFile(invocationsPath, "utf8").catch(() => "");
expect(invocations).not.toContain("download-one:");
});
-
-test("Whisper button is hidden when a transcript already exists", async ({
- page,
-}) => {
- await resetData("one-youtube-channel-with-data");
- await page.goto("/channels/test-youtube/videos/20240101_test1234567");
- await expect(
- page.getByRole("button", { name: "Whisper transcribe" }),
- ).toHaveCount(0);
-});
diff --git a/editor/e2e/whisper.spec.ts b/editor/e2e/whisper.spec.ts
@@ -19,6 +19,35 @@ test("transcribes every audio file with no transcript", async ({ page }) => {
}
});
+test("Transcribe missing skips VTT-only videos instead of failing them", async ({
+ page,
+}) => {
+ await resetData("youtube-with-playlist");
+ await page.goto("/channels/test-youtube");
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+ const log = page.getByLabel("Transcribe missing output");
+ await expect(log).toContainText(
+ "Whisper batch: 0 succeeded, 0 failed, 1 skipped, 0 attempted.",
+ { timeout: 30_000 },
+ );
+ await expect(log).toContainText(
+ "Transcription for fake00000001 already exists",
+ );
+
+ const failPath =
+ "test-transcripts/channels/test-youtube/failed-transcriptions";
+ if (await pathExists(failPath)) {
+ const raw = await readFile(resolvePath(failPath), "utf8");
+ expect(raw).not.toContain("fake00000001");
+ }
+
+ expect(
+ await pathExists(
+ "test-transcripts/channels/test-youtube/data/fake00000001/transcript.json",
+ ),
+ ).toBe(false);
+});
+
test("verify reports nothing missing once transcribed", async ({ page }) => {
await resetData("one-transcribe-channel-with-audio");
await page.goto("/channels/test-transcribe");