commit 813939fc4fce3ee4d69645ced7bdd6980224de5e
parent 7ce693ef86310088415fa61fa19288316c81fae6
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 15 May 2026 23:09:52 -0400
rework dl to be managed
Diffstat:
11 files changed, 956 insertions(+), 138 deletions(-)
diff --git a/common/lib/downloadOutcome-server.ts b/common/lib/downloadOutcome-server.ts
@@ -0,0 +1,48 @@
+import path from "node:path";
+import { readFile, rename, writeFile } from "node:fs/promises";
+import {
+ DOWNLOAD_OUTCOME_FILENAME,
+ DOWNLOAD_OUTCOME_STATUS_VALUES,
+ type DownloadOutcomeRecord,
+ type DownloadOutcomeStatus,
+} from "./downloadOutcome";
+
+export function downloadOutcomePath(videoDir: string): string {
+ return path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME);
+}
+
+export async function loadDownloadOutcome(
+ videoDir: string,
+): Promise<DownloadOutcomeRecord | null> {
+ try {
+ const raw = await readFile(downloadOutcomePath(videoDir), "utf8");
+ const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>;
+ if (
+ typeof parsed?.status === "string" &&
+ (DOWNLOAD_OUTCOME_STATUS_VALUES as string[]).includes(parsed.status) &&
+ typeof parsed.videoId === "string" &&
+ typeof parsed.startedAt === "string" &&
+ typeof parsed.finishedAt === "string" &&
+ Array.isArray(parsed.attempts)
+ ) {
+ // Cast: status is already narrowed to a member of the enum tuple.
+ return parsed as DownloadOutcomeRecord;
+ }
+ return null;
+ } catch {
+ return null;
+ }
+}
+
+export async function writeDownloadOutcome(
+ videoDir: string,
+ record: DownloadOutcomeRecord,
+): Promise<void> {
+ const file = downloadOutcomePath(videoDir);
+ const tmp = `${file}.tmp-${process.pid}`;
+ await writeFile(tmp, JSON.stringify(record, null, 2) + "\n");
+ await rename(tmp, file);
+}
+
+// Narrow re-export so callers don't have to import from both modules.
+export type { DownloadOutcomeStatus };
diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts
@@ -0,0 +1,45 @@
+// Client-safe types and constants for the per-video download outcome sidecar.
+// Mirrors availability.ts: server-only I/O lives in downloadOutcome-server.ts.
+
+import type { Availability } from "./availability";
+import type { ChannelHandling } from "./channelConfig";
+
+export type DownloadOutcomeStatus =
+ | "ok"
+ | "ok-with-cookies"
+ | "ok-auto-transcribed"
+ | "failed";
+
+export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus> = [
+ "ok",
+ "ok-with-cookies",
+ "ok-auto-transcribed",
+ "failed",
+];
+
+export type DownloadAttemptKind =
+ | "primary"
+ | "auth-retry"
+ | "no-subs-fallback";
+
+export type DownloadAttempt = {
+ n: 1 | 2 | 3;
+ kind: DownloadAttemptKind;
+ handling: ChannelHandling;
+ usedCookies: boolean;
+ ytdlpExitCode: number | null;
+ availabilityClass?: Availability;
+ error?: string;
+};
+
+export type DownloadOutcomeRecord = {
+ videoId: string;
+ webpageUrl?: string;
+ status: DownloadOutcomeStatus;
+ startedAt: string;
+ finishedAt: string;
+ attempts: DownloadAttempt[];
+ fellBackToTranscribe?: boolean;
+};
+
+export const DOWNLOAD_OUTCOME_FILENAME = "download-outcome.json";
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -10,6 +10,10 @@ export type SiteSettings = {
transcribeBin: string;
transcribeArgs: string[];
transcribeModel: string;
+ // Browser spec (e.g. "firefox", "chrome:Default") passed to
+ // `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the
+ // managed per-video download flow. Empty string = disabled.
+ cookiesFromBrowser: string;
};
export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
@@ -47,6 +51,7 @@ function defaults(): SiteSettings {
transcribeBin: paths.whisperBin,
transcribeArgs: [...DEFAULT_TRANSCRIBE_ARGS],
transcribeModel: paths.whisperModel,
+ cookiesFromBrowser: "",
};
}
@@ -69,6 +74,9 @@ export function getSettings(): SiteSettings {
if (typeof merged.transcribeModel !== "string") {
merged.transcribeModel = defaults().transcribeModel;
}
+ if (typeof merged.cookiesFromBrowser !== "string") {
+ merged.cookiesFromBrowser = "";
+ }
return merged;
}
@@ -115,6 +123,10 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
...next,
maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes),
transcribeArgs: next.transcribeArgs.map((a) => String(a)),
+ cookiesFromBrowser:
+ typeof next.cookiesFromBrowser === "string"
+ ? next.cookiesFromBrowser.trim()
+ : "",
};
const tmp = `${file}.tmp-${process.pid}`;
await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n");
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -0,0 +1,449 @@
+import path from "node:path";
+import { appendFile, mkdir, readdir, readFile } from "node:fs/promises";
+import { execa } from "execa";
+import {
+ parseUnavailableFromStderr,
+ type Availability,
+} from "../lib/availability";
+import {
+ type AudioFormat,
+ type ChannelConfig,
+} from "../lib/channelConfig";
+import {
+ type DownloadAttempt,
+ type DownloadOutcomeRecord,
+ type DownloadOutcomeStatus,
+} from "../lib/downloadOutcome";
+import { writeDownloadOutcome } from "../lib/downloadOutcome-server";
+import type { Paths } from "../lib/paths";
+import { transcribeOneVideo } from "../controller/transcribeOne";
+import { extractVideoId, isRumbleUrl } from "./runYtdlp";
+
+const STDERR_TAIL_BYTES = 64 * 1024;
+const ARCHIVE_MARKER = "DLOM_ARCHIVE";
+
+export type ManagedDownloadOpts = {
+ channelSlug: string;
+ channelConfig: ChannelConfig;
+ paths: Paths;
+ videoUrl: string;
+ onLog: (s: string) => void;
+ signal: AbortSignal;
+ // Resolved global setting; only used when the primary attempt fails with an
+ // auth/age error AND the channel doesn't already have its own cookies set.
+ globalCookiesFromBrowser?: string;
+ // When false, suppress appending to the channel archive file. Mirrors
+ // `ignoreArchive` from the playlist-level callers.
+ appendArchive?: boolean;
+};
+
+const OUTPUT_ARGS: string[] = [
+ "-o",
+ "data/%(id)s/audio.%(ext)s",
+ "-o",
+ "subtitle:data/%(id)s/transcript",
+ "-o",
+ "infojson:data/%(id)s/metadata",
+ "--no-write-playlist-metafiles",
+];
+
+function youtubeHandlingArgs(config: ChannelConfig): string[] {
+ return [
+ "--write-auto-subs",
+ "--write-subs",
+ "--sub-langs",
+ config.subLangs ?? "en.*,live_chat",
+ "--write-info-json",
+ "--skip-download",
+ "-t",
+ "sleep",
+ ];
+}
+
+function transcribeHandlingArgs(config: ChannelConfig): string[] {
+ const fmt: AudioFormat = config.audioFormat ?? "mp3";
+ const args = [
+ "--write-info-json",
+ "-f",
+ "bestaudio/worst",
+ "-x",
+ "--audio-format",
+ fmt,
+ ];
+ if (config.keepSourceVideo) args.push("-k");
+ return args;
+}
+
+function channelConfigArgs(
+ config: ChannelConfig,
+ overrideCookies?: string,
+): string[] {
+ const args: string[] = [];
+ const cookies = overrideCookies ?? config.cookiesFromBrowser;
+ if (cookies) args.push("--cookies-from-browser", cookies);
+ if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
+ return args;
+}
+
+type AttemptOutcome = {
+ exitCode: number | null;
+ stderrTail: string;
+ archiveLine: string | null;
+};
+
+async function runOneYtdlp(
+ opts: ManagedDownloadOpts,
+ cwd: string,
+ args: string[],
+): Promise<AttemptOutcome> {
+ opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`);
+ const child = execa(opts.paths.ytdlpBin, args, {
+ cwd,
+ cancelSignal: opts.signal,
+ all: false,
+ buffer: false,
+ reject: false,
+ });
+
+ let stderrTail = "";
+ child.stderr?.on("data", (c: Buffer) => {
+ const chunk = c.toString("utf8");
+ opts.onLog(chunk);
+ stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_BYTES);
+ });
+
+ let stdoutBuf = "";
+ let archiveLine: string | null = null;
+ child.stdout?.on("data", (c: Buffer) => {
+ const chunk = c.toString("utf8");
+ opts.onLog(chunk);
+ stdoutBuf += chunk;
+ // Pull whole lines out of the buffer; keep the trailing partial line.
+ let nl: number;
+ while ((nl = stdoutBuf.indexOf("\n")) !== -1) {
+ const line = stdoutBuf.slice(0, nl).trim();
+ stdoutBuf = stdoutBuf.slice(nl + 1);
+ if (line.startsWith(`${ARCHIVE_MARKER} `)) {
+ archiveLine = line.slice(ARCHIVE_MARKER.length + 1).trim();
+ }
+ }
+ });
+
+ const result = await child;
+ // Flush any final partial line.
+ const trailing = stdoutBuf.trim();
+ if (trailing.startsWith(`${ARCHIVE_MARKER} `)) {
+ archiveLine = trailing.slice(ARCHIVE_MARKER.length + 1).trim();
+ }
+ return {
+ exitCode: result.exitCode ?? null,
+ stderrTail,
+ archiveLine,
+ };
+}
+
+function attemptSucceeded(exitCode: number | null): boolean {
+ // yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop).
+ return exitCode === 0 || exitCode === 101;
+}
+
+const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([
+ "needs_auth",
+ "members_only",
+ "private",
+]);
+
+async function resolveVideoIdFromUrl(
+ url: string,
+ channelDir: string,
+): Promise<string | null> {
+ // Most platforms: the URL slug IS the id used in data/<id>/. Rumble is the
+ // exception — its archive id (and on-disk dir name) is yt-dlp's internal
+ // RumbleEmbed id, not the URL slug. Build a minimal lookup using
+ // metadata.info.json files we just wrote.
+ const slug = extractVideoId(url);
+ if (!slug) return null;
+ if (!isRumbleUrl(url)) return slug;
+ const dataDir = path.join(channelDir, "data");
+ const dirs = await readdir(dataDir).catch(() => [] as string[]);
+ for (const dir of dirs) {
+ try {
+ const raw = await readFile(
+ path.join(dataDir, dir, "metadata.info.json"),
+ "utf8",
+ );
+ const parsed = JSON.parse(raw) as { webpage_url?: unknown };
+ if (typeof parsed.webpage_url === "string") {
+ const dirSlug = extractVideoId(parsed.webpage_url);
+ if (dirSlug === slug) return dir;
+ }
+ } catch {
+ continue;
+ }
+ }
+ return slug;
+}
+
+async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ return entries.some(
+ (e) =>
+ e === "transcript.json" ||
+ /^transcript\.[^.]+\.(?:vtt|json|json3|srv1|srv2|srv3)$/.test(e),
+ );
+}
+
+async function metadataReportsNoCaptions(
+ videoDir: string,
+): Promise<boolean> {
+ let raw: string;
+ try {
+ raw = await readFile(
+ path.join(videoDir, "metadata.info.json"),
+ "utf8",
+ );
+ } catch {
+ return false;
+ }
+ let parsed: {
+ subtitles?: Record<string, unknown>;
+ automatic_captions?: Record<string, unknown>;
+ };
+ try {
+ parsed = JSON.parse(raw);
+ } catch {
+ return false;
+ }
+ const subs = parsed.subtitles ?? {};
+ const auto = parsed.automatic_captions ?? {};
+ return (
+ typeof subs === "object" &&
+ typeof auto === "object" &&
+ Object.keys(subs).length === 0 &&
+ Object.keys(auto).length === 0
+ );
+}
+
+function trimError(stderrTail: string): string | undefined {
+ const trimmed = stderrTail.trim().split("\n").slice(-3).join("\n");
+ return trimmed || undefined;
+}
+
+async function appendArchiveLine(
+ channelDir: string,
+ line: string,
+): Promise<void> {
+ await mkdir(channelDir, { recursive: true });
+ await appendFile(path.join(channelDir, "archive"), line + "\n");
+}
+
+export async function downloadOneManaged(
+ opts: ManagedDownloadOpts,
+): Promise<DownloadOutcomeRecord> {
+ const startedAt = new Date().toISOString();
+ const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug);
+ await mkdir(channelDir, { recursive: true });
+
+ const attempts: DownloadAttempt[] = [];
+ let status: DownloadOutcomeStatus = "failed";
+ let fellBackToTranscribe = false;
+ let lastArchiveLine: string | null = null;
+
+ // ---------- Attempt 1: primary ----------
+ const primaryArgs = [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...OUTPUT_ARGS,
+ ...(opts.channelConfig.handling === "youtube"
+ ? youtubeHandlingArgs(opts.channelConfig)
+ : transcribeHandlingArgs(opts.channelConfig)),
+ "--print",
+ `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
+ ...channelConfigArgs(opts.channelConfig),
+ "--",
+ opts.videoUrl,
+ ];
+ const primaryRes = await runOneYtdlp(opts, channelDir, primaryArgs);
+ const primaryAvail = attemptSucceeded(primaryRes.exitCode)
+ ? undefined
+ : parseUnavailableFromStderr(primaryRes.stderrTail);
+ attempts.push({
+ n: 1,
+ kind: "primary",
+ handling: opts.channelConfig.handling,
+ usedCookies: Boolean(opts.channelConfig.cookiesFromBrowser),
+ ytdlpExitCode: primaryRes.exitCode,
+ availabilityClass: primaryAvail,
+ error: attemptSucceeded(primaryRes.exitCode)
+ ? undefined
+ : trimError(primaryRes.stderrTail),
+ });
+ if (primaryRes.archiveLine) lastArchiveLine = primaryRes.archiveLine;
+
+ let lastSucceeded = attemptSucceeded(primaryRes.exitCode);
+ if (lastSucceeded) status = "ok";
+
+ // ---------- Attempt 2: auth retry ----------
+ const shouldAuthRetry =
+ !lastSucceeded &&
+ primaryAvail !== undefined &&
+ AUTH_RETRY_CLASSES.has(primaryAvail) &&
+ Boolean(opts.globalCookiesFromBrowser) &&
+ !opts.channelConfig.cookiesFromBrowser;
+
+ if (shouldAuthRetry && !opts.signal.aborted) {
+ opts.onLog(
+ `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${opts.globalCookiesFromBrowser}\n`,
+ );
+ const retryArgs = [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...OUTPUT_ARGS,
+ ...(opts.channelConfig.handling === "youtube"
+ ? youtubeHandlingArgs(opts.channelConfig)
+ : transcribeHandlingArgs(opts.channelConfig)),
+ "--print",
+ `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
+ ...channelConfigArgs(opts.channelConfig, opts.globalCookiesFromBrowser),
+ "--",
+ opts.videoUrl,
+ ];
+ const retryRes = await runOneYtdlp(opts, channelDir, retryArgs);
+ const retryAvail = attemptSucceeded(retryRes.exitCode)
+ ? undefined
+ : parseUnavailableFromStderr(retryRes.stderrTail);
+ attempts.push({
+ n: 2,
+ kind: "auth-retry",
+ handling: opts.channelConfig.handling,
+ usedCookies: true,
+ ytdlpExitCode: retryRes.exitCode,
+ availabilityClass: retryAvail,
+ error: attemptSucceeded(retryRes.exitCode)
+ ? undefined
+ : trimError(retryRes.stderrTail),
+ });
+ if (retryRes.archiveLine) lastArchiveLine = retryRes.archiveLine;
+ if (attemptSucceeded(retryRes.exitCode)) {
+ lastSucceeded = true;
+ status = "ok-with-cookies";
+ }
+ }
+
+ // ---------- Attempt 3: no-subs fallback (youtube handling only) ----------
+ const videoId =
+ (await resolveVideoIdFromUrl(opts.videoUrl, channelDir)) ?? "unknown";
+ const videoDir = path.join(channelDir, "data", videoId);
+
+ if (
+ lastSucceeded &&
+ opts.channelConfig.handling === "youtube" &&
+ !opts.signal.aborted
+ ) {
+ const hasTranscript = await hasAnyTranscriptOnDisk(videoDir);
+ const noCaptions = await metadataReportsNoCaptions(videoDir);
+ if (!hasTranscript && noCaptions) {
+ opts.onLog(
+ `No subs available for ${videoId}; falling back to audio download + whisper.\n`,
+ );
+ const audioFmt: AudioFormat = opts.channelConfig.audioFormat ?? "mp3";
+ const fallbackConfig: ChannelConfig = {
+ ...opts.channelConfig,
+ handling: "transcribe",
+ };
+ // Use cookies if the auth retry succeeded with them; otherwise channel-level.
+ const fallbackCookieOverride =
+ status === "ok-with-cookies"
+ ? opts.globalCookiesFromBrowser
+ : undefined;
+ const fallbackArgs = [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...OUTPUT_ARGS,
+ ...transcribeHandlingArgs(fallbackConfig),
+ "--print",
+ `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
+ ...channelConfigArgs(fallbackConfig, fallbackCookieOverride),
+ "--",
+ opts.videoUrl,
+ ];
+ const fallbackRes = await runOneYtdlp(opts, channelDir, fallbackArgs);
+ const fallbackAvail = attemptSucceeded(fallbackRes.exitCode)
+ ? undefined
+ : parseUnavailableFromStderr(fallbackRes.stderrTail);
+ attempts.push({
+ n: 3,
+ kind: "no-subs-fallback",
+ handling: "transcribe",
+ usedCookies: Boolean(fallbackCookieOverride),
+ ytdlpExitCode: fallbackRes.exitCode,
+ availabilityClass: fallbackAvail,
+ error: attemptSucceeded(fallbackRes.exitCode)
+ ? undefined
+ : trimError(fallbackRes.stderrTail),
+ });
+ if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine;
+
+ if (attemptSucceeded(fallbackRes.exitCode)) {
+ // Inline whisper: matches whisperVideoAction's shape.
+ try {
+ await transcribeOneVideo({
+ paths: opts.paths,
+ videoDir,
+ videoId,
+ audioFilename: `audio.${audioFmt}`,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ });
+ fellBackToTranscribe = true;
+ status = "ok-auto-transcribed";
+ } catch (err) {
+ opts.onLog(
+ `Whisper failed after no-subs fallback: ${(err as Error).message}\n`,
+ );
+ status = "failed";
+ lastSucceeded = false;
+ }
+ } else {
+ status = "failed";
+ lastSucceeded = false;
+ }
+ }
+ }
+
+ // ---------- Archive append ----------
+ if (lastSucceeded && opts.appendArchive !== false && lastArchiveLine) {
+ try {
+ await appendArchiveLine(channelDir, lastArchiveLine);
+ } catch (err) {
+ opts.onLog(
+ `Archive append failed (${lastArchiveLine}): ${(err as Error).message}\n`,
+ );
+ }
+ }
+
+ // ---------- Sidecar ----------
+ const finishedAt = new Date().toISOString();
+ const record: DownloadOutcomeRecord = {
+ videoId,
+ webpageUrl: opts.videoUrl,
+ status,
+ startedAt,
+ finishedAt,
+ attempts,
+ ...(fellBackToTranscribe ? { fellBackToTranscribe: true } : {}),
+ };
+ // Only write the sidecar if we know which dir to put it in. If the very first
+ // attempt failed before metadata could be written, the data/<id> dir may not
+ // exist yet; in that case `mkdir -p` it so the sidecar lands somewhere.
+ try {
+ await mkdir(videoDir, { recursive: true });
+ await writeDownloadOutcome(videoDir, record);
+ } catch (err) {
+ opts.onLog(
+ `Failed to write download-outcome.json: ${(err as Error).message}\n`,
+ );
+ }
+
+ return record;
+}
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -8,6 +8,7 @@ import {
writeFile,
} from "node:fs/promises";
import { execa } from "execa";
+import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
import {
parseChannelConfig,
@@ -15,9 +16,11 @@ import {
type ChannelConfig,
type ChannelHandling,
} from "../lib/channelConfig";
+import { getSettings } from "../lib/settings";
import type { Paths } from "../lib/paths";
import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability";
import { resolveShardItems } from "../controller/shard";
+import { downloadOneManaged } from "./downloadOneManaged";
export type YtdlpMode =
| "store-playlist"
@@ -177,11 +180,23 @@ async function storePlaylist(opts: RunYtdlpOpts): Promise<void> {
}
async function downloadFromPlaylist(opts: RunYtdlpOpts): Promise<void> {
+ await downloadPlaylistManaged(opts, { prefilter: "archive" });
+}
+
+async function downloadMissing(opts: RunYtdlpOpts): Promise<void> {
+ await downloadPlaylistManaged(opts, { prefilter: "destination-exists" });
+}
+
+type PrefilterMode = "archive" | "destination-exists";
+
+async function downloadPlaylistManaged(
+ opts: RunYtdlpOpts,
+ { prefilter }: { prefilter: PrefilterMode },
+): Promise<void> {
const root = channelRoot(opts);
const dataDir = path.join(root, "data");
const playlistPath = path.join(root, "playlist");
const archivePath = path.join(root, "archive");
- const toFetchPath = path.join(root, "playlist.tofetch");
let playlistText: string;
try {
@@ -196,152 +211,136 @@ async function downloadFromPlaylist(opts: RunYtdlpOpts): Promise<void> {
.map((s) => s.trim())
.filter(Boolean);
- // App-side prefilter: skip URLs whose ID is already in the archive so
- // yt-dlp doesn't re-fetch metadata for entries we know are done. The
- // --break-on-existing flag stays as defense in depth.
- const archive = await readArchive(archivePath);
const rumbleIndex = urls.some(isRumbleUrl)
? await buildRumbleSlugIndex(dataDir)
: null;
+
const tofetch: string[] = [];
- let skipped = 0;
- for (const url of urls) {
- const archiveId = lookupArchiveId(url, rumbleIndex);
- if (archiveId && archive.ids.has(archiveId)) {
- skipped++;
- continue;
+ if (prefilter === "archive") {
+ // Skip URLs whose ID is already recorded in the channel's archive file.
+ const archive = await readArchive(archivePath);
+ let skipped = 0;
+ if (opts.ignoreArchive) {
+ tofetch.push(...urls);
+ } else {
+ for (const url of urls) {
+ const archiveId = lookupArchiveId(url, rumbleIndex);
+ if (archiveId && archive.ids.has(archiveId)) {
+ skipped++;
+ continue;
+ }
+ tofetch.push(url);
+ }
}
- tofetch.push(url);
+ opts.onLog(
+ `Prefilter: ${tofetch.length} new, ${skipped} already archived (of ${urls.length} total).\n`,
+ );
+ } else {
+ // Skip URLs whose destination file (transcript.en.vtt for youtube
+ // handling, audio.* or transcript.json for transcribe handling) is
+ // already on disk. Catches videos missing from a stale archive file.
+ let alreadyComplete = 0;
+ let unidentifiable = 0;
+ for (const url of urls) {
+ const slug = extractVideoId(url);
+ if (!slug) {
+ unidentifiable++;
+ tofetch.push(url);
+ continue;
+ }
+ const dirId =
+ rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
+ if (
+ dirId &&
+ (await destinationExists(dataDir, dirId, opts.channelConfig.handling))
+ ) {
+ alreadyComplete++;
+ continue;
+ }
+ tofetch.push(url);
+ }
+ opts.onLog(
+ `Prefilter: ${tofetch.length} missing destination files, ${alreadyComplete} already complete (of ${urls.length} total)${
+ unidentifiable ? `, ${unidentifiable} could not be identified` : ""
+ }.\n`,
+ );
}
- opts.onLog(
- `Prefilter: ${tofetch.length} new, ${skipped} already archived (of ${urls.length} total).\n`,
- );
- if (tofetch.length === 0) {
- await rm(toFetchPath, { force: true });
+ // Sharding is only meaningful for the missing-files mode (matches prior
+ // behavior where `download-missing` accepted shardTotal / shardIndex).
+ const items =
+ prefilter === "destination-exists"
+ ? (
+ await resolveShardItems({
+ paths: opts.paths,
+ slug: opts.channelSlug,
+ op: "download-missing",
+ fullItems: tofetch,
+ totalShards: opts.shardTotal,
+ shardIndex: opts.shardIndex,
+ onLog: (m) => opts.onLog(`${m}\n`),
+ })
+ ).items
+ : tofetch;
+
+ if (items.length === 0) {
opts.onLog("Nothing to fetch.\n");
await touchLastFullDownload(opts);
return;
}
- await writeFile(toFetchPath, tofetch.join("\n") + "\n");
-
- const args: string[] = [
- "--ignore-config",
- "--restrict-filenames",
- ...OUTPUT_ARGS,
- ...handlingArgs(opts.channelConfig),
- "--force-write-archive",
- "--download-archive",
- "archive",
- ...(opts.abortOnError === false ? [] : ["--abort-on-error"]),
- "-a",
- "playlist.tofetch",
- ...configArgs(opts.channelConfig),
- ];
- await runChildAndStream(opts, root, args);
-
- await rm(toFetchPath, { force: true });
- await touchLastFullDownload(opts);
- await safeBackfillAvailability(opts);
-}
-
-async function downloadMissing(opts: RunYtdlpOpts): Promise<void> {
- const root = channelRoot(opts);
- const dataDir = path.join(root, "data");
- const playlistPath = path.join(root, "playlist");
- const toFetchPath = path.join(root, "playlist.tofetch");
-
- let playlistText: string;
- try {
- playlistText = await readFile(playlistPath, "utf8");
- } catch {
- throw new Error(
- `No saved playlist at ${playlistPath}. Run "Store playlist" first.`,
+ const globalCookies = getSettings().cookiesFromBrowser;
+ if (opts.ignoreArchive) {
+ opts.onLog(
+ "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n",
);
}
- const urls = playlistText
- .split("\n")
- .map((s) => s.trim())
- .filter(Boolean);
- // Prefilter by destination-file existence rather than the archive: skip
- // URLs whose expected output (transcript.en.vtt for youtube handling,
- // audio.* or transcript.json for transcribe handling) is already on disk.
- // Catches videos missing from a stale or absent archive file, and avoids
- // re-downloading audio for videos whose transcript has already been
- // produced (audio may have been cleaned up post-transcription).
- const rumbleIndex = urls.some(isRumbleUrl)
- ? await buildRumbleSlugIndex(dataDir)
- : null;
- const tofetch: string[] = [];
- let alreadyComplete = 0;
- let unidentifiable = 0;
- for (const url of urls) {
- const slug = extractVideoId(url);
- if (!slug) {
- unidentifiable++;
- tofetch.push(url);
- continue;
- }
- const dirId =
- rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
- if (
- dirId &&
- (await destinationExists(dataDir, dirId, opts.channelConfig.handling))
- ) {
- alreadyComplete++;
- continue;
- }
- tofetch.push(url);
- }
- opts.onLog(
- `Prefilter: ${tofetch.length} missing destination files, ${alreadyComplete} already complete (of ${urls.length} total)${
- unidentifiable ? `, ${unidentifiable} could not be identified` : ""
- }.\n`,
+ // Serialize within a run: keeps logs readable and avoids hammering the
+ // source with parallel requests (which is what often triggers needs_auth
+ // in the first place). The per-video startup cost is the price of being
+ // able to retry independently.
+ const limit = pLimit(1);
+ let failedCount = 0;
+ const abortOnError = opts.abortOnError !== false;
+ let firstFailure: Error | null = null;
+
+ await Promise.all(
+ items.map((url) =>
+ limit(async () => {
+ if (opts.signal.aborted) return;
+ if (firstFailure && abortOnError) return;
+ const outcome = await downloadOneManaged({
+ channelSlug: opts.channelSlug,
+ channelConfig: opts.channelConfig,
+ paths: opts.paths,
+ videoUrl: url,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ globalCookiesFromBrowser: globalCookies || undefined,
+ appendArchive: !opts.ignoreArchive,
+ });
+ if (outcome.status === "failed") {
+ failedCount++;
+ if (abortOnError && !firstFailure) {
+ firstFailure = new Error(
+ `Managed download failed for ${url} (status=failed)`,
+ );
+ }
+ }
+ }),
+ ),
);
- const { items: shardedTofetch } = await resolveShardItems({
- paths: opts.paths,
- slug: opts.channelSlug,
- op: "download-missing",
- fullItems: tofetch,
- totalShards: opts.shardTotal,
- shardIndex: opts.shardIndex,
- onLog: (m) => opts.onLog(`${m}\n`),
- });
-
- if (shardedTofetch.length === 0) {
- await rm(toFetchPath, { force: true });
- opts.onLog("Nothing to fetch.\n");
- await touchLastFullDownload(opts);
- return;
+ if (firstFailure && abortOnError && !opts.signal.aborted) {
+ throw firstFailure;
}
-
- await writeFile(toFetchPath, shardedTofetch.join("\n") + "\n");
-
- if (opts.ignoreArchive) {
+ if (failedCount > 0) {
opts.onLog(
- "Ignoring archive: --download-archive omitted, videos will be fetched even if listed in the archive file.\n",
+ `Managed download complete with ${failedCount} failed video(s) (abort-on-error=${abortOnError}).\n`,
);
}
- const archiveArgs = opts.ignoreArchive
- ? []
- : ["--force-write-archive", "--download-archive", "archive"];
- const args: string[] = [
- "--ignore-config",
- "--restrict-filenames",
- ...OUTPUT_ARGS,
- ...handlingArgs(opts.channelConfig),
- ...archiveArgs,
- ...(opts.abortOnError === false ? [] : ["--abort-on-error"]),
- "-a",
- "playlist.tofetch",
- ...configArgs(opts.channelConfig),
- ];
- await runChildAndStream(opts, root, args);
- await rm(toFetchPath, { force: true });
await touchLastFullDownload(opts);
await safeBackfillAvailability(opts);
}
diff --git a/editor/app/channels/[slug]/videos/[id]/components/TextFilePreview.tsx b/editor/app/channels/[slug]/videos/[id]/components/TextFilePreview.tsx
@@ -0,0 +1,82 @@
+"use client";
+
+import { useEffect, useRef, useState } from "react";
+
+type Phase =
+ | { kind: "idle" }
+ | { kind: "loading" }
+ | { kind: "ready"; text: string }
+ | { kind: "error"; message: string };
+
+type Props = {
+ src: string;
+ filename: string;
+};
+
+export function TextFilePreview({ src, filename }: Props) {
+ const [phase, setPhase] = useState<Phase>({ kind: "idle" });
+ const abortRef = useRef<AbortController | null>(null);
+
+ useEffect(
+ () => () => {
+ abortRef.current?.abort();
+ },
+ [],
+ );
+
+ function handleToggle(e: React.SyntheticEvent<HTMLDetailsElement>) {
+ if (!e.currentTarget.open || phase.kind !== "idle") return;
+ const controller = new AbortController();
+ abortRef.current = controller;
+ setPhase({ kind: "loading" });
+ fetch(src, { signal: controller.signal })
+ .then(async (res) => {
+ if (!res.ok) throw new Error(`HTTP ${res.status}`);
+ return res.text();
+ })
+ .then((text) => {
+ if (controller.signal.aborted) return;
+ setPhase({ kind: "ready", text });
+ })
+ .catch((err: unknown) => {
+ if (controller.signal.aborted) return;
+ const message = err instanceof Error ? err.message : String(err);
+ setPhase({ kind: "error", message });
+ });
+ }
+
+ return (
+ <details
+ onToggle={handleToggle}
+ className="text-sm"
+ aria-label={`preview ${filename}`}
+ >
+ <summary className="cursor-pointer text-zinc-600 dark:text-zinc-400 hover:text-zinc-900 dark:hover:text-zinc-100">
+ Preview contents
+ </summary>
+ <div className="mt-2">
+ {phase.kind === "loading" && (
+ <p role="status" className="text-xs text-zinc-500">
+ Loading…
+ </p>
+ )}
+ {phase.kind === "error" && (
+ <p
+ role="alert"
+ className="text-xs text-red-600 dark:text-red-400"
+ >
+ Failed to load: {phase.message}
+ </p>
+ )}
+ {phase.kind === "ready" && (
+ <pre
+ aria-label={`preview of ${filename}`}
+ className="text-xs font-mono bg-zinc-100 dark:bg-zinc-900 border border-zinc-200 dark:border-zinc-800 rounded p-3 h-96 overflow-auto whitespace-pre-wrap"
+ >
+ {phase.text}
+ </pre>
+ )}
+ </div>
+ </details>
+ );
+}
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -7,10 +7,12 @@ import {
type AudioFormat,
type ChannelHandling,
} from "yt-dlp-transcript-common/lib/channelConfig";
+import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome";
import { QueueControl } from "../../../../../components/QueueControl";
import { cancelJobAction } from "../../../../../jobs/actions";
import { PipelineStageCard } from "../../../components/PipelineStageCard";
import { MediaPlayer } from "./MediaPlayer";
+import { TextFilePreview } from "./TextFilePreview";
import {
deleteVideoDirAction,
deleteVideoFileAction,
@@ -36,16 +38,18 @@ type Props = {
audioFormat: AudioFormat;
defaultQueueKey: string;
existingQueues: string[];
+ downloadOutcome: DownloadOutcomeRecord | null;
};
-function isAudioFile(name: string): boolean {
- return name.startsWith("audio.");
-}
-
-function audioExt(name: string): string {
- const dot = name.lastIndexOf(".");
- return dot >= 0 ? name.slice(dot + 1) : "";
-}
+const AUDIO_EXTS = new Set([
+ "mp3",
+ "m4a",
+ "aac",
+ "ogg",
+ "opus",
+ "wav",
+ "flac",
+]);
const VIDEO_EXTS = new Set([
"mp4",
@@ -57,14 +61,27 @@ const VIDEO_EXTS = new Set([
"avi",
]);
+const TEXT_EXTS = new Set(["vtt", "json", "txt", "srt", "log"]);
+
function fileExt(name: string): string {
const dot = name.lastIndexOf(".");
return dot >= 0 ? name.slice(dot + 1).toLowerCase() : "";
}
-function mediaKind(name: string): "audio" | "video" | null {
- if (isAudioFile(name)) return "audio";
- if (VIDEO_EXTS.has(fileExt(name))) return "video";
+function isAudioFile(name: string): boolean {
+ return AUDIO_EXTS.has(fileExt(name));
+}
+
+function audioExt(name: string): string {
+ const dot = name.lastIndexOf(".");
+ return dot >= 0 ? name.slice(dot + 1) : "";
+}
+
+function mediaKind(name: string): "audio" | "video" | "text" | null {
+ const ext = fileExt(name);
+ if (AUDIO_EXTS.has(ext)) return "audio";
+ if (VIDEO_EXTS.has(ext)) return "video";
+ if (TEXT_EXTS.has(ext)) return "text";
return null;
}
@@ -94,6 +111,7 @@ export function VideoPanel({
audioFormat,
defaultQueueKey,
existingQueues,
+ downloadOutcome,
}: Props) {
const audioFiles = files.filter((f) => isAudioFile(f.name));
const hasTranscriptJson = files.some((f) => f.name === "transcript.json");
@@ -118,6 +136,7 @@ export function VideoPanel({
return (
<div className="flex flex-col gap-6">
+ {downloadOutcome && <DownloadOutcomeBadge outcome={downloadOutcome} />}
<PipelineStageCard
id="download"
title={noAudio ? "Download audio" : "Redownload audio"}
@@ -492,13 +511,18 @@ function FilesList({
{formatSize(f.size)} · {new Date(f.mtime).toLocaleString()}
</span>
</div>
- {kind && (
+ {kind === "audio" || kind === "video" ? (
<MediaPlayer
src={mediaUrl(slug, videoId, f.name)}
kind={kind}
filename={f.name}
/>
- )}
+ ) : kind === "text" ? (
+ <TextFilePreview
+ src={mediaUrl(slug, videoId, f.name)}
+ filename={f.name}
+ />
+ ) : null}
<DeleteFileButton slug={slug} videoId={videoId} filename={f.name} />
</li>
);
@@ -640,6 +664,63 @@ function DeleteVideoDirSection({
);
}
+function DownloadOutcomeBadge({
+ outcome,
+}: {
+ outcome: DownloadOutcomeRecord;
+}) {
+ if (outcome.status === "ok") return null;
+ const tone =
+ outcome.status === "failed"
+ ? "border-red-300 bg-red-50 text-red-900 dark:border-red-900 dark:bg-red-950 dark:text-red-200"
+ : "border-amber-300 bg-amber-50 text-amber-900 dark:border-amber-900 dark:bg-amber-950 dark:text-amber-200";
+ const label = ((): string => {
+ switch (outcome.status) {
+ case "ok-with-cookies":
+ return "Retried with cookies";
+ case "ok-auto-transcribed":
+ return "Auto-transcribed (no subs available)";
+ case "failed": {
+ const lastAttempt = outcome.attempts[outcome.attempts.length - 1];
+ const cls = lastAttempt?.availabilityClass;
+ return cls ? `Download failed: ${cls}` : "Download failed";
+ }
+ default:
+ return outcome.status;
+ }
+ })();
+ const lastErr =
+ outcome.status === "failed"
+ ? outcome.attempts[outcome.attempts.length - 1]?.error
+ : undefined;
+ return (
+ <div
+ role="status"
+ aria-label="download outcome"
+ className={`rounded border px-3 py-2 text-sm ${tone}`}
+ >
+ <div className="flex flex-wrap items-baseline justify-between gap-2">
+ <span className="font-medium">{label}</span>
+ <span className="text-xs opacity-70">
+ {outcome.attempts.length} attempt
+ {outcome.attempts.length === 1 ? "" : "s"} ·{" "}
+ {new Date(outcome.finishedAt).toLocaleString()}
+ </span>
+ </div>
+ {lastErr && (
+ <details className="mt-1">
+ <summary className="cursor-pointer text-xs opacity-80">
+ error tail
+ </summary>
+ <pre className="mt-1 whitespace-pre-wrap text-xs font-mono opacity-90">
+ {lastErr}
+ </pre>
+ </details>
+ )}
+ </div>
+ );
+}
+
function Heading({ title, desc }: { title: string; desc: string }) {
return (
<div>
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -5,6 +5,7 @@ import path from "node:path";
import { readdir, readFile, stat } from "node:fs/promises";
import type { Dirent } from "node:fs";
import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
detectPlatform,
@@ -91,6 +92,9 @@ export default async function VideoDetailPage({
if (!config) notFound();
const dirData = await loadVideoDir(slug, id);
const meta = await loadMeta(slug, id);
+ const downloadOutcome = await loadDownloadOutcome(
+ path.join(getPaths().channelsDir, slug, "data", id),
+ );
const existingQueues = getRegistry().activeQueueNames();
const defaultQueueKey = platformQueueKey(
@@ -146,6 +150,7 @@ export default async function VideoDetailPage({
audioFormat={config.audioFormat ?? "mp3"}
defaultQueueKey={defaultQueueKey}
existingQueues={existingQueues}
+ downloadOutcome={downloadOutcome}
/>
</div>
);
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -22,6 +22,9 @@ export async function saveSettingsAction(
const maxBytesRaw = String(formData.get("maxTranscriptPageBytes") ?? "").trim();
const transcribeBin = String(formData.get("transcribeBin") ?? "").trim();
const transcribeModel = String(formData.get("transcribeModel") ?? "").trim();
+ const cookiesFromBrowser = String(
+ formData.get("cookiesFromBrowser") ?? "",
+ ).trim();
const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? "");
const transcribeArgs = transcribeArgsRaw
.split("\n")
@@ -60,6 +63,7 @@ export async function saveSettingsAction(
transcribeBin,
transcribeModel,
transcribeArgs,
+ cookiesFromBrowser,
};
await writeSettings(next);
revalidatePath("/settings");
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -87,6 +87,12 @@ export function SettingsForm({ initial }: Props) {
</span>
</label>
</fieldset>
+ <Field
+ label="Cookies from browser (retry only)"
+ name="cookiesFromBrowser"
+ defaultValue={initial.cookiesFromBrowser}
+ hint="Browser spec passed to yt-dlp --cookies-from-browser only when the primary download attempt fails with an auth/age error and the channel doesn't have its own cookies set. e.g. firefox, chrome:Default. Leave blank to disable."
+ />
<div className="flex items-center gap-3">
<button
type="submit"
diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts
@@ -265,6 +265,93 @@ test("Mark untranscribable button is absent when transcript already exists", asy
).toHaveCount(0);
});
+test("text files render a preview, not the audio player", async ({ page }) => {
+ await resetData("one-youtube-channel-with-data");
+ const dir = resolvePath(
+ "test-transcripts/channels/test-youtube/data/20240101_test1234567",
+ );
+ // Stage misleadingly-named text files plus a real audio.mp3 in the same
+ // video dir. Old code matched name.startsWith("audio.") and would render
+ // <audio> for audio.vtt and audio.json.
+ await writeFile(`${dir}/audio.vtt`, "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nfake cue\n");
+ await writeFile(`${dir}/audio.json`, '{"hello":"world"}\n');
+ await writeFile(`${dir}/audio.mp3`, "binary audio bytes");
+
+ await page.goto("/channels/test-youtube/videos/20240101_test1234567");
+
+ const fileList = page.getByLabel("files for 20240101_test1234567");
+
+ // audio.mp3 — real audio, MediaPlayer renders.
+ await expect(fileList.getByLabel("media player for audio.mp3")).toBeVisible();
+ await expect(fileList.getByLabel("preview audio.mp3")).toHaveCount(0);
+
+ // audio.vtt — text file, preview details renders, no audio player.
+ await expect(fileList.getByLabel("media player for audio.vtt")).toHaveCount(0);
+ const vttPreview = fileList.getByLabel("preview audio.vtt");
+ await expect(vttPreview).toBeVisible();
+ // Content is not fetched until the user expands.
+ await expect(fileList.getByLabel("preview of audio.vtt")).toHaveCount(0);
+ await vttPreview.getByText("Preview contents").click();
+ await expect(fileList.getByLabel("preview of audio.vtt")).toContainText(
+ "fake cue",
+ );
+
+ // audio.json — text file, same shape.
+ await expect(fileList.getByLabel("media player for audio.json")).toHaveCount(0);
+ const jsonPreview = fileList.getByLabel("preview audio.json");
+ await expect(jsonPreview).toBeVisible();
+ await jsonPreview.getByText("Preview contents").click();
+ await expect(fileList.getByLabel("preview of audio.json")).toContainText(
+ '"hello":"world"',
+ );
+});
+
+test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => {
+ await resetData("one-youtube-channel-with-data");
+ const dir = resolvePath(
+ "test-transcripts/channels/test-youtube/data/20240101_test1234567",
+ );
+ await writeFile(`${dir}/audio.vtt`, "WEBVTT\n");
+ await writeFile(`${dir}/audio.json`, "{}\n");
+ await writeFile(`${dir}/audio.mp3`, "fake audio");
+
+ await page.goto("/channels/test-youtube/videos/20240101_test1234567");
+
+ // Download stage summary counts only the real audio file.
+ await expect(page.getByLabel("Redownload audio stage summary")).toContainText(
+ "1 audio file on disk.",
+ );
+ // Transcode summary too.
+ await expect(page.getByLabel("Transcode stage summary")).toContainText(
+ "Convert 1 audio file to another format.",
+ );
+ // Per-file transcode row exists for audio.mp3 only — not for the text files.
+ await expect(
+ page.getByRole("heading", { name: "Transcode audio.mp3" }),
+ ).toBeVisible();
+ await expect(
+ page.getByRole("heading", { name: "Transcode audio.vtt" }),
+ ).toHaveCount(0);
+ await expect(
+ page.getByRole("heading", { name: "Transcode audio.json" }),
+ ).toHaveCount(0);
+});
+
+test("existing transcript text files are inspectable from the file list", async ({
+ page,
+}) => {
+ await resetData("one-youtube-channel-with-data");
+ await page.goto("/channels/test-youtube/videos/20240101_test1234567");
+
+ const fileList = page.getByLabel("files for 20240101_test1234567");
+ const vttPreview = fileList.getByLabel("preview transcript.en.vtt");
+ await expect(vttPreview).toBeVisible();
+ await vttPreview.getByText("Preview contents").click();
+ await expect(fileList.getByLabel("preview of transcript.en.vtt")).toContainText(
+ "synthetic test transcript",
+ );
+});
+
test("Redownload writes audio in the chosen format", async ({ page }) => {
await resetData("one-youtube-channel-with-data");
// Stage a video dir whose name matches the URL id so fake-ytdlp lands its