Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 813939fc4fce3ee4d69645ced7bdd6980224de5e
parent 7ce693ef86310088415fa61fa19288316c81fae6
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 15 May 2026 23:09:52 -0400

rework dl to be managed

Diffstat:
Acommon/lib/downloadOutcome-server.ts | 48++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/downloadOutcome.ts | 45+++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/settings.ts | 12++++++++++++
Acommon/ytdlp/downloadOneManaged.ts | 449+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 249+++++++++++++++++++++++++++++++++++++++----------------------------------------
Aeditor/app/channels/[slug]/videos/[id]/components/TextFilePreview.tsx | 82+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 107+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----------
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 5+++++
Meditor/app/settings/actions.ts | 4++++
Meditor/app/settings/components/SettingsForm.tsx | 6++++++
Meditor/e2e/video-page.spec.ts | 87+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
11 files changed, 956 insertions(+), 138 deletions(-)

diff --git a/common/lib/downloadOutcome-server.ts b/common/lib/downloadOutcome-server.ts @@ -0,0 +1,48 @@ +import path from "node:path"; +import { readFile, rename, writeFile } from "node:fs/promises"; +import { + DOWNLOAD_OUTCOME_FILENAME, + DOWNLOAD_OUTCOME_STATUS_VALUES, + type DownloadOutcomeRecord, + type DownloadOutcomeStatus, +} from "./downloadOutcome"; + +export function downloadOutcomePath(videoDir: string): string { + return path.join(videoDir, DOWNLOAD_OUTCOME_FILENAME); +} + +export async function loadDownloadOutcome( + videoDir: string, +): Promise<DownloadOutcomeRecord | null> { + try { + const raw = await readFile(downloadOutcomePath(videoDir), "utf8"); + const parsed = JSON.parse(raw) as Partial<DownloadOutcomeRecord>; + if ( + typeof parsed?.status === "string" && + (DOWNLOAD_OUTCOME_STATUS_VALUES as string[]).includes(parsed.status) && + typeof parsed.videoId === "string" && + typeof parsed.startedAt === "string" && + typeof parsed.finishedAt === "string" && + Array.isArray(parsed.attempts) + ) { + // Cast: status is already narrowed to a member of the enum tuple. + return parsed as DownloadOutcomeRecord; + } + return null; + } catch { + return null; + } +} + +export async function writeDownloadOutcome( + videoDir: string, + record: DownloadOutcomeRecord, +): Promise<void> { + const file = downloadOutcomePath(videoDir); + const tmp = `${file}.tmp-${process.pid}`; + await writeFile(tmp, JSON.stringify(record, null, 2) + "\n"); + await rename(tmp, file); +} + +// Narrow re-export so callers don't have to import from both modules. +export type { DownloadOutcomeStatus }; diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts @@ -0,0 +1,45 @@ +// Client-safe types and constants for the per-video download outcome sidecar. +// Mirrors availability.ts: server-only I/O lives in downloadOutcome-server.ts. + +import type { Availability } from "./availability"; +import type { ChannelHandling } from "./channelConfig"; + +export type DownloadOutcomeStatus = + | "ok" + | "ok-with-cookies" + | "ok-auto-transcribed" + | "failed"; + +export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus> = [ + "ok", + "ok-with-cookies", + "ok-auto-transcribed", + "failed", +]; + +export type DownloadAttemptKind = + | "primary" + | "auth-retry" + | "no-subs-fallback"; + +export type DownloadAttempt = { + n: 1 | 2 | 3; + kind: DownloadAttemptKind; + handling: ChannelHandling; + usedCookies: boolean; + ytdlpExitCode: number | null; + availabilityClass?: Availability; + error?: string; +}; + +export type DownloadOutcomeRecord = { + videoId: string; + webpageUrl?: string; + status: DownloadOutcomeStatus; + startedAt: string; + finishedAt: string; + attempts: DownloadAttempt[]; + fellBackToTranscribe?: boolean; +}; + +export const DOWNLOAD_OUTCOME_FILENAME = "download-outcome.json"; diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -10,6 +10,10 @@ export type SiteSettings = { transcribeBin: string; transcribeArgs: string[]; transcribeModel: string; + // Browser spec (e.g. "firefox", "chrome:Default") passed to + // `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the + // managed per-video download flow. Empty string = disabled. + cookiesFromBrowser: string; }; export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; @@ -47,6 +51,7 @@ function defaults(): SiteSettings { transcribeBin: paths.whisperBin, transcribeArgs: [...DEFAULT_TRANSCRIBE_ARGS], transcribeModel: paths.whisperModel, + cookiesFromBrowser: "", }; } @@ -69,6 +74,9 @@ export function getSettings(): SiteSettings { if (typeof merged.transcribeModel !== "string") { merged.transcribeModel = defaults().transcribeModel; } + if (typeof merged.cookiesFromBrowser !== "string") { + merged.cookiesFromBrowser = ""; + } return merged; } @@ -115,6 +123,10 @@ export async function writeSettings(next: SiteSettings): Promise<void> { ...next, maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes), transcribeArgs: next.transcribeArgs.map((a) => String(a)), + cookiesFromBrowser: + typeof next.cookiesFromBrowser === "string" + ? next.cookiesFromBrowser.trim() + : "", }; const tmp = `${file}.tmp-${process.pid}`; await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n"); diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -0,0 +1,449 @@ +import path from "node:path"; +import { appendFile, mkdir, readdir, readFile } from "node:fs/promises"; +import { execa } from "execa"; +import { + parseUnavailableFromStderr, + type Availability, +} from "../lib/availability"; +import { + type AudioFormat, + type ChannelConfig, +} from "../lib/channelConfig"; +import { + type DownloadAttempt, + type DownloadOutcomeRecord, + type DownloadOutcomeStatus, +} from "../lib/downloadOutcome"; +import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; +import type { Paths } from "../lib/paths"; +import { transcribeOneVideo } from "../controller/transcribeOne"; +import { extractVideoId, isRumbleUrl } from "./runYtdlp"; + +const STDERR_TAIL_BYTES = 64 * 1024; +const ARCHIVE_MARKER = "DLOM_ARCHIVE"; + +export type ManagedDownloadOpts = { + channelSlug: string; + channelConfig: ChannelConfig; + paths: Paths; + videoUrl: string; + onLog: (s: string) => void; + signal: AbortSignal; + // Resolved global setting; only used when the primary attempt fails with an + // auth/age error AND the channel doesn't already have its own cookies set. + globalCookiesFromBrowser?: string; + // When false, suppress appending to the channel archive file. Mirrors + // `ignoreArchive` from the playlist-level callers. + appendArchive?: boolean; +}; + +const OUTPUT_ARGS: string[] = [ + "-o", + "data/%(id)s/audio.%(ext)s", + "-o", + "subtitle:data/%(id)s/transcript", + "-o", + "infojson:data/%(id)s/metadata", + "--no-write-playlist-metafiles", +]; + +function youtubeHandlingArgs(config: ChannelConfig): string[] { + return [ + "--write-auto-subs", + "--write-subs", + "--sub-langs", + config.subLangs ?? "en.*,live_chat", + "--write-info-json", + "--skip-download", + "-t", + "sleep", + ]; +} + +function transcribeHandlingArgs(config: ChannelConfig): string[] { + const fmt: AudioFormat = config.audioFormat ?? "mp3"; + const args = [ + "--write-info-json", + "-f", + "bestaudio/worst", + "-x", + "--audio-format", + fmt, + ]; + if (config.keepSourceVideo) args.push("-k"); + return args; +} + +function channelConfigArgs( + config: ChannelConfig, + overrideCookies?: string, +): string[] { + const args: string[] = []; + const cookies = overrideCookies ?? config.cookiesFromBrowser; + if (cookies) args.push("--cookies-from-browser", cookies); + if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs); + return args; +} + +type AttemptOutcome = { + exitCode: number | null; + stderrTail: string; + archiveLine: string | null; +}; + +async function runOneYtdlp( + opts: ManagedDownloadOpts, + cwd: string, + args: string[], +): Promise<AttemptOutcome> { + opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`); + const child = execa(opts.paths.ytdlpBin, args, { + cwd, + cancelSignal: opts.signal, + all: false, + buffer: false, + reject: false, + }); + + let stderrTail = ""; + child.stderr?.on("data", (c: Buffer) => { + const chunk = c.toString("utf8"); + opts.onLog(chunk); + stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_BYTES); + }); + + let stdoutBuf = ""; + let archiveLine: string | null = null; + child.stdout?.on("data", (c: Buffer) => { + const chunk = c.toString("utf8"); + opts.onLog(chunk); + stdoutBuf += chunk; + // Pull whole lines out of the buffer; keep the trailing partial line. + let nl: number; + while ((nl = stdoutBuf.indexOf("\n")) !== -1) { + const line = stdoutBuf.slice(0, nl).trim(); + stdoutBuf = stdoutBuf.slice(nl + 1); + if (line.startsWith(`${ARCHIVE_MARKER} `)) { + archiveLine = line.slice(ARCHIVE_MARKER.length + 1).trim(); + } + } + }); + + const result = await child; + // Flush any final partial line. + const trailing = stdoutBuf.trim(); + if (trailing.startsWith(`${ARCHIVE_MARKER} `)) { + archiveLine = trailing.slice(ARCHIVE_MARKER.length + 1).trim(); + } + return { + exitCode: result.exitCode ?? null, + stderrTail, + archiveLine, + }; +} + +function attemptSucceeded(exitCode: number | null): boolean { + // yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop). + return exitCode === 0 || exitCode === 101; +} + +const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([ + "needs_auth", + "members_only", + "private", +]); + +async function resolveVideoIdFromUrl( + url: string, + channelDir: string, +): Promise<string | null> { + // Most platforms: the URL slug IS the id used in data/<id>/. Rumble is the + // exception — its archive id (and on-disk dir name) is yt-dlp's internal + // RumbleEmbed id, not the URL slug. Build a minimal lookup using + // metadata.info.json files we just wrote. + const slug = extractVideoId(url); + if (!slug) return null; + if (!isRumbleUrl(url)) return slug; + const dataDir = path.join(channelDir, "data"); + const dirs = await readdir(dataDir).catch(() => [] as string[]); + for (const dir of dirs) { + try { + const raw = await readFile( + path.join(dataDir, dir, "metadata.info.json"), + "utf8", + ); + const parsed = JSON.parse(raw) as { webpage_url?: unknown }; + if (typeof parsed.webpage_url === "string") { + const dirSlug = extractVideoId(parsed.webpage_url); + if (dirSlug === slug) return dir; + } + } catch { + continue; + } + } + return slug; +} + +async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + return entries.some( + (e) => + e === "transcript.json" || + /^transcript\.[^.]+\.(?:vtt|json|json3|srv1|srv2|srv3)$/.test(e), + ); +} + +async function metadataReportsNoCaptions( + videoDir: string, +): Promise<boolean> { + let raw: string; + try { + raw = await readFile( + path.join(videoDir, "metadata.info.json"), + "utf8", + ); + } catch { + return false; + } + let parsed: { + subtitles?: Record<string, unknown>; + automatic_captions?: Record<string, unknown>; + }; + try { + parsed = JSON.parse(raw); + } catch { + return false; + } + const subs = parsed.subtitles ?? {}; + const auto = parsed.automatic_captions ?? {}; + return ( + typeof subs === "object" && + typeof auto === "object" && + Object.keys(subs).length === 0 && + Object.keys(auto).length === 0 + ); +} + +function trimError(stderrTail: string): string | undefined { + const trimmed = stderrTail.trim().split("\n").slice(-3).join("\n"); + return trimmed || undefined; +} + +async function appendArchiveLine( + channelDir: string, + line: string, +): Promise<void> { + await mkdir(channelDir, { recursive: true }); + await appendFile(path.join(channelDir, "archive"), line + "\n"); +} + +export async function downloadOneManaged( + opts: ManagedDownloadOpts, +): Promise<DownloadOutcomeRecord> { + const startedAt = new Date().toISOString(); + const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug); + await mkdir(channelDir, { recursive: true }); + + const attempts: DownloadAttempt[] = []; + let status: DownloadOutcomeStatus = "failed"; + let fellBackToTranscribe = false; + let lastArchiveLine: string | null = null; + + // ---------- Attempt 1: primary ---------- + const primaryArgs = [ + "--ignore-config", + "--restrict-filenames", + ...OUTPUT_ARGS, + ...(opts.channelConfig.handling === "youtube" + ? youtubeHandlingArgs(opts.channelConfig) + : transcribeHandlingArgs(opts.channelConfig)), + "--print", + `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, + ...channelConfigArgs(opts.channelConfig), + "--", + opts.videoUrl, + ]; + const primaryRes = await runOneYtdlp(opts, channelDir, primaryArgs); + const primaryAvail = attemptSucceeded(primaryRes.exitCode) + ? undefined + : parseUnavailableFromStderr(primaryRes.stderrTail); + attempts.push({ + n: 1, + kind: "primary", + handling: opts.channelConfig.handling, + usedCookies: Boolean(opts.channelConfig.cookiesFromBrowser), + ytdlpExitCode: primaryRes.exitCode, + availabilityClass: primaryAvail, + error: attemptSucceeded(primaryRes.exitCode) + ? undefined + : trimError(primaryRes.stderrTail), + }); + if (primaryRes.archiveLine) lastArchiveLine = primaryRes.archiveLine; + + let lastSucceeded = attemptSucceeded(primaryRes.exitCode); + if (lastSucceeded) status = "ok"; + + // ---------- Attempt 2: auth retry ---------- + const shouldAuthRetry = + !lastSucceeded && + primaryAvail !== undefined && + AUTH_RETRY_CLASSES.has(primaryAvail) && + Boolean(opts.globalCookiesFromBrowser) && + !opts.channelConfig.cookiesFromBrowser; + + if (shouldAuthRetry && !opts.signal.aborted) { + opts.onLog( + `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${opts.globalCookiesFromBrowser}\n`, + ); + const retryArgs = [ + "--ignore-config", + "--restrict-filenames", + ...OUTPUT_ARGS, + ...(opts.channelConfig.handling === "youtube" + ? youtubeHandlingArgs(opts.channelConfig) + : transcribeHandlingArgs(opts.channelConfig)), + "--print", + `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, + ...channelConfigArgs(opts.channelConfig, opts.globalCookiesFromBrowser), + "--", + opts.videoUrl, + ]; + const retryRes = await runOneYtdlp(opts, channelDir, retryArgs); + const retryAvail = attemptSucceeded(retryRes.exitCode) + ? undefined + : parseUnavailableFromStderr(retryRes.stderrTail); + attempts.push({ + n: 2, + kind: "auth-retry", + handling: opts.channelConfig.handling, + usedCookies: true, + ytdlpExitCode: retryRes.exitCode, + availabilityClass: retryAvail, + error: attemptSucceeded(retryRes.exitCode) + ? undefined + : trimError(retryRes.stderrTail), + }); + if (retryRes.archiveLine) lastArchiveLine = retryRes.archiveLine; + if (attemptSucceeded(retryRes.exitCode)) { + lastSucceeded = true; + status = "ok-with-cookies"; + } + } + + // ---------- Attempt 3: no-subs fallback (youtube handling only) ---------- + const videoId = + (await resolveVideoIdFromUrl(opts.videoUrl, channelDir)) ?? "unknown"; + const videoDir = path.join(channelDir, "data", videoId); + + if ( + lastSucceeded && + opts.channelConfig.handling === "youtube" && + !opts.signal.aborted + ) { + const hasTranscript = await hasAnyTranscriptOnDisk(videoDir); + const noCaptions = await metadataReportsNoCaptions(videoDir); + if (!hasTranscript && noCaptions) { + opts.onLog( + `No subs available for ${videoId}; falling back to audio download + whisper.\n`, + ); + const audioFmt: AudioFormat = opts.channelConfig.audioFormat ?? "mp3"; + const fallbackConfig: ChannelConfig = { + ...opts.channelConfig, + handling: "transcribe", + }; + // Use cookies if the auth retry succeeded with them; otherwise channel-level. + const fallbackCookieOverride = + status === "ok-with-cookies" + ? opts.globalCookiesFromBrowser + : undefined; + const fallbackArgs = [ + "--ignore-config", + "--restrict-filenames", + ...OUTPUT_ARGS, + ...transcribeHandlingArgs(fallbackConfig), + "--print", + `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, + ...channelConfigArgs(fallbackConfig, fallbackCookieOverride), + "--", + opts.videoUrl, + ]; + const fallbackRes = await runOneYtdlp(opts, channelDir, fallbackArgs); + const fallbackAvail = attemptSucceeded(fallbackRes.exitCode) + ? undefined + : parseUnavailableFromStderr(fallbackRes.stderrTail); + attempts.push({ + n: 3, + kind: "no-subs-fallback", + handling: "transcribe", + usedCookies: Boolean(fallbackCookieOverride), + ytdlpExitCode: fallbackRes.exitCode, + availabilityClass: fallbackAvail, + error: attemptSucceeded(fallbackRes.exitCode) + ? undefined + : trimError(fallbackRes.stderrTail), + }); + if (fallbackRes.archiveLine) lastArchiveLine = fallbackRes.archiveLine; + + if (attemptSucceeded(fallbackRes.exitCode)) { + // Inline whisper: matches whisperVideoAction's shape. + try { + await transcribeOneVideo({ + paths: opts.paths, + videoDir, + videoId, + audioFilename: `audio.${audioFmt}`, + onLog: opts.onLog, + signal: opts.signal, + }); + fellBackToTranscribe = true; + status = "ok-auto-transcribed"; + } catch (err) { + opts.onLog( + `Whisper failed after no-subs fallback: ${(err as Error).message}\n`, + ); + status = "failed"; + lastSucceeded = false; + } + } else { + status = "failed"; + lastSucceeded = false; + } + } + } + + // ---------- Archive append ---------- + if (lastSucceeded && opts.appendArchive !== false && lastArchiveLine) { + try { + await appendArchiveLine(channelDir, lastArchiveLine); + } catch (err) { + opts.onLog( + `Archive append failed (${lastArchiveLine}): ${(err as Error).message}\n`, + ); + } + } + + // ---------- Sidecar ---------- + const finishedAt = new Date().toISOString(); + const record: DownloadOutcomeRecord = { + videoId, + webpageUrl: opts.videoUrl, + status, + startedAt, + finishedAt, + attempts, + ...(fellBackToTranscribe ? { fellBackToTranscribe: true } : {}), + }; + // Only write the sidecar if we know which dir to put it in. If the very first + // attempt failed before metadata could be written, the data/<id> dir may not + // exist yet; in that case `mkdir -p` it so the sidecar lands somewhere. + try { + await mkdir(videoDir, { recursive: true }); + await writeDownloadOutcome(videoDir, record); + } catch (err) { + opts.onLog( + `Failed to write download-outcome.json: ${(err as Error).message}\n`, + ); + } + + return record; +} diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -8,6 +8,7 @@ import { writeFile, } from "node:fs/promises"; import { execa } from "execa"; +import pLimit from "p-limit"; import { readArchive } from "../lib/archive"; import { parseChannelConfig, @@ -15,9 +16,11 @@ import { type ChannelConfig, type ChannelHandling, } from "../lib/channelConfig"; +import { getSettings } from "../lib/settings"; import type { Paths } from "../lib/paths"; import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability"; import { resolveShardItems } from "../controller/shard"; +import { downloadOneManaged } from "./downloadOneManaged"; export type YtdlpMode = | "store-playlist" @@ -177,11 +180,23 @@ async function storePlaylist(opts: RunYtdlpOpts): Promise<void> { } async function downloadFromPlaylist(opts: RunYtdlpOpts): Promise<void> { + await downloadPlaylistManaged(opts, { prefilter: "archive" }); +} + +async function downloadMissing(opts: RunYtdlpOpts): Promise<void> { + await downloadPlaylistManaged(opts, { prefilter: "destination-exists" }); +} + +type PrefilterMode = "archive" | "destination-exists"; + +async function downloadPlaylistManaged( + opts: RunYtdlpOpts, + { prefilter }: { prefilter: PrefilterMode }, +): Promise<void> { const root = channelRoot(opts); const dataDir = path.join(root, "data"); const playlistPath = path.join(root, "playlist"); const archivePath = path.join(root, "archive"); - const toFetchPath = path.join(root, "playlist.tofetch"); let playlistText: string; try { @@ -196,152 +211,136 @@ async function downloadFromPlaylist(opts: RunYtdlpOpts): Promise<void> { .map((s) => s.trim()) .filter(Boolean); - // App-side prefilter: skip URLs whose ID is already in the archive so - // yt-dlp doesn't re-fetch metadata for entries we know are done. The - // --break-on-existing flag stays as defense in depth. - const archive = await readArchive(archivePath); const rumbleIndex = urls.some(isRumbleUrl) ? await buildRumbleSlugIndex(dataDir) : null; + const tofetch: string[] = []; - let skipped = 0; - for (const url of urls) { - const archiveId = lookupArchiveId(url, rumbleIndex); - if (archiveId && archive.ids.has(archiveId)) { - skipped++; - continue; + if (prefilter === "archive") { + // Skip URLs whose ID is already recorded in the channel's archive file. + const archive = await readArchive(archivePath); + let skipped = 0; + if (opts.ignoreArchive) { + tofetch.push(...urls); + } else { + for (const url of urls) { + const archiveId = lookupArchiveId(url, rumbleIndex); + if (archiveId && archive.ids.has(archiveId)) { + skipped++; + continue; + } + tofetch.push(url); + } } - tofetch.push(url); + opts.onLog( + `Prefilter: ${tofetch.length} new, ${skipped} already archived (of ${urls.length} total).\n`, + ); + } else { + // Skip URLs whose destination file (transcript.en.vtt for youtube + // handling, audio.* or transcript.json for transcribe handling) is + // already on disk. Catches videos missing from a stale archive file. + let alreadyComplete = 0; + let unidentifiable = 0; + for (const url of urls) { + const slug = extractVideoId(url); + if (!slug) { + unidentifiable++; + tofetch.push(url); + continue; + } + const dirId = + rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; + if ( + dirId && + (await destinationExists(dataDir, dirId, opts.channelConfig.handling)) + ) { + alreadyComplete++; + continue; + } + tofetch.push(url); + } + opts.onLog( + `Prefilter: ${tofetch.length} missing destination files, ${alreadyComplete} already complete (of ${urls.length} total)${ + unidentifiable ? `, ${unidentifiable} could not be identified` : "" + }.\n`, + ); } - opts.onLog( - `Prefilter: ${tofetch.length} new, ${skipped} already archived (of ${urls.length} total).\n`, - ); - if (tofetch.length === 0) { - await rm(toFetchPath, { force: true }); + // Sharding is only meaningful for the missing-files mode (matches prior + // behavior where `download-missing` accepted shardTotal / shardIndex). + const items = + prefilter === "destination-exists" + ? ( + await resolveShardItems({ + paths: opts.paths, + slug: opts.channelSlug, + op: "download-missing", + fullItems: tofetch, + totalShards: opts.shardTotal, + shardIndex: opts.shardIndex, + onLog: (m) => opts.onLog(`${m}\n`), + }) + ).items + : tofetch; + + if (items.length === 0) { opts.onLog("Nothing to fetch.\n"); await touchLastFullDownload(opts); return; } - await writeFile(toFetchPath, tofetch.join("\n") + "\n"); - - const args: string[] = [ - "--ignore-config", - "--restrict-filenames", - ...OUTPUT_ARGS, - ...handlingArgs(opts.channelConfig), - "--force-write-archive", - "--download-archive", - "archive", - ...(opts.abortOnError === false ? [] : ["--abort-on-error"]), - "-a", - "playlist.tofetch", - ...configArgs(opts.channelConfig), - ]; - await runChildAndStream(opts, root, args); - - await rm(toFetchPath, { force: true }); - await touchLastFullDownload(opts); - await safeBackfillAvailability(opts); -} - -async function downloadMissing(opts: RunYtdlpOpts): Promise<void> { - const root = channelRoot(opts); - const dataDir = path.join(root, "data"); - const playlistPath = path.join(root, "playlist"); - const toFetchPath = path.join(root, "playlist.tofetch"); - - let playlistText: string; - try { - playlistText = await readFile(playlistPath, "utf8"); - } catch { - throw new Error( - `No saved playlist at ${playlistPath}. Run "Store playlist" first.`, + const globalCookies = getSettings().cookiesFromBrowser; + if (opts.ignoreArchive) { + opts.onLog( + "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n", ); } - const urls = playlistText - .split("\n") - .map((s) => s.trim()) - .filter(Boolean); - // Prefilter by destination-file existence rather than the archive: skip - // URLs whose expected output (transcript.en.vtt for youtube handling, - // audio.* or transcript.json for transcribe handling) is already on disk. - // Catches videos missing from a stale or absent archive file, and avoids - // re-downloading audio for videos whose transcript has already been - // produced (audio may have been cleaned up post-transcription). - const rumbleIndex = urls.some(isRumbleUrl) - ? await buildRumbleSlugIndex(dataDir) - : null; - const tofetch: string[] = []; - let alreadyComplete = 0; - let unidentifiable = 0; - for (const url of urls) { - const slug = extractVideoId(url); - if (!slug) { - unidentifiable++; - tofetch.push(url); - continue; - } - const dirId = - rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; - if ( - dirId && - (await destinationExists(dataDir, dirId, opts.channelConfig.handling)) - ) { - alreadyComplete++; - continue; - } - tofetch.push(url); - } - opts.onLog( - `Prefilter: ${tofetch.length} missing destination files, ${alreadyComplete} already complete (of ${urls.length} total)${ - unidentifiable ? `, ${unidentifiable} could not be identified` : "" - }.\n`, + // Serialize within a run: keeps logs readable and avoids hammering the + // source with parallel requests (which is what often triggers needs_auth + // in the first place). The per-video startup cost is the price of being + // able to retry independently. + const limit = pLimit(1); + let failedCount = 0; + const abortOnError = opts.abortOnError !== false; + let firstFailure: Error | null = null; + + await Promise.all( + items.map((url) => + limit(async () => { + if (opts.signal.aborted) return; + if (firstFailure && abortOnError) return; + const outcome = await downloadOneManaged({ + channelSlug: opts.channelSlug, + channelConfig: opts.channelConfig, + paths: opts.paths, + videoUrl: url, + onLog: opts.onLog, + signal: opts.signal, + globalCookiesFromBrowser: globalCookies || undefined, + appendArchive: !opts.ignoreArchive, + }); + if (outcome.status === "failed") { + failedCount++; + if (abortOnError && !firstFailure) { + firstFailure = new Error( + `Managed download failed for ${url} (status=failed)`, + ); + } + } + }), + ), ); - const { items: shardedTofetch } = await resolveShardItems({ - paths: opts.paths, - slug: opts.channelSlug, - op: "download-missing", - fullItems: tofetch, - totalShards: opts.shardTotal, - shardIndex: opts.shardIndex, - onLog: (m) => opts.onLog(`${m}\n`), - }); - - if (shardedTofetch.length === 0) { - await rm(toFetchPath, { force: true }); - opts.onLog("Nothing to fetch.\n"); - await touchLastFullDownload(opts); - return; + if (firstFailure && abortOnError && !opts.signal.aborted) { + throw firstFailure; } - - await writeFile(toFetchPath, shardedTofetch.join("\n") + "\n"); - - if (opts.ignoreArchive) { + if (failedCount > 0) { opts.onLog( - "Ignoring archive: --download-archive omitted, videos will be fetched even if listed in the archive file.\n", + `Managed download complete with ${failedCount} failed video(s) (abort-on-error=${abortOnError}).\n`, ); } - const archiveArgs = opts.ignoreArchive - ? [] - : ["--force-write-archive", "--download-archive", "archive"]; - const args: string[] = [ - "--ignore-config", - "--restrict-filenames", - ...OUTPUT_ARGS, - ...handlingArgs(opts.channelConfig), - ...archiveArgs, - ...(opts.abortOnError === false ? [] : ["--abort-on-error"]), - "-a", - "playlist.tofetch", - ...configArgs(opts.channelConfig), - ]; - await runChildAndStream(opts, root, args); - await rm(toFetchPath, { force: true }); await touchLastFullDownload(opts); await safeBackfillAvailability(opts); } diff --git a/editor/app/channels/[slug]/videos/[id]/components/TextFilePreview.tsx b/editor/app/channels/[slug]/videos/[id]/components/TextFilePreview.tsx @@ -0,0 +1,82 @@ +"use client"; + +import { useEffect, useRef, useState } from "react"; + +type Phase = + | { kind: "idle" } + | { kind: "loading" } + | { kind: "ready"; text: string } + | { kind: "error"; message: string }; + +type Props = { + src: string; + filename: string; +}; + +export function TextFilePreview({ src, filename }: Props) { + const [phase, setPhase] = useState<Phase>({ kind: "idle" }); + const abortRef = useRef<AbortController | null>(null); + + useEffect( + () => () => { + abortRef.current?.abort(); + }, + [], + ); + + function handleToggle(e: React.SyntheticEvent<HTMLDetailsElement>) { + if (!e.currentTarget.open || phase.kind !== "idle") return; + const controller = new AbortController(); + abortRef.current = controller; + setPhase({ kind: "loading" }); + fetch(src, { signal: controller.signal }) + .then(async (res) => { + if (!res.ok) throw new Error(`HTTP ${res.status}`); + return res.text(); + }) + .then((text) => { + if (controller.signal.aborted) return; + setPhase({ kind: "ready", text }); + }) + .catch((err: unknown) => { + if (controller.signal.aborted) return; + const message = err instanceof Error ? err.message : String(err); + setPhase({ kind: "error", message }); + }); + } + + return ( + <details + onToggle={handleToggle} + className="text-sm" + aria-label={`preview ${filename}`} + > + <summary className="cursor-pointer text-zinc-600 dark:text-zinc-400 hover:text-zinc-900 dark:hover:text-zinc-100"> + Preview contents + </summary> + <div className="mt-2"> + {phase.kind === "loading" && ( + <p role="status" className="text-xs text-zinc-500"> + Loading… + </p> + )} + {phase.kind === "error" && ( + <p + role="alert" + className="text-xs text-red-600 dark:text-red-400" + > + Failed to load: {phase.message} + </p> + )} + {phase.kind === "ready" && ( + <pre + aria-label={`preview of ${filename}`} + className="text-xs font-mono bg-zinc-100 dark:bg-zinc-900 border border-zinc-200 dark:border-zinc-800 rounded p-3 h-96 overflow-auto whitespace-pre-wrap" + > + {phase.text} + </pre> + )} + </div> + </details> + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -7,10 +7,12 @@ import { type AudioFormat, type ChannelHandling, } from "yt-dlp-transcript-common/lib/channelConfig"; +import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome"; import { QueueControl } from "../../../../../components/QueueControl"; import { cancelJobAction } from "../../../../../jobs/actions"; import { PipelineStageCard } from "../../../components/PipelineStageCard"; import { MediaPlayer } from "./MediaPlayer"; +import { TextFilePreview } from "./TextFilePreview"; import { deleteVideoDirAction, deleteVideoFileAction, @@ -36,16 +38,18 @@ type Props = { audioFormat: AudioFormat; defaultQueueKey: string; existingQueues: string[]; + downloadOutcome: DownloadOutcomeRecord | null; }; -function isAudioFile(name: string): boolean { - return name.startsWith("audio."); -} - -function audioExt(name: string): string { - const dot = name.lastIndexOf("."); - return dot >= 0 ? name.slice(dot + 1) : ""; -} +const AUDIO_EXTS = new Set([ + "mp3", + "m4a", + "aac", + "ogg", + "opus", + "wav", + "flac", +]); const VIDEO_EXTS = new Set([ "mp4", @@ -57,14 +61,27 @@ const VIDEO_EXTS = new Set([ "avi", ]); +const TEXT_EXTS = new Set(["vtt", "json", "txt", "srt", "log"]); + function fileExt(name: string): string { const dot = name.lastIndexOf("."); return dot >= 0 ? name.slice(dot + 1).toLowerCase() : ""; } -function mediaKind(name: string): "audio" | "video" | null { - if (isAudioFile(name)) return "audio"; - if (VIDEO_EXTS.has(fileExt(name))) return "video"; +function isAudioFile(name: string): boolean { + return AUDIO_EXTS.has(fileExt(name)); +} + +function audioExt(name: string): string { + const dot = name.lastIndexOf("."); + return dot >= 0 ? name.slice(dot + 1) : ""; +} + +function mediaKind(name: string): "audio" | "video" | "text" | null { + const ext = fileExt(name); + if (AUDIO_EXTS.has(ext)) return "audio"; + if (VIDEO_EXTS.has(ext)) return "video"; + if (TEXT_EXTS.has(ext)) return "text"; return null; } @@ -94,6 +111,7 @@ export function VideoPanel({ audioFormat, defaultQueueKey, existingQueues, + downloadOutcome, }: Props) { const audioFiles = files.filter((f) => isAudioFile(f.name)); const hasTranscriptJson = files.some((f) => f.name === "transcript.json"); @@ -118,6 +136,7 @@ export function VideoPanel({ return ( <div className="flex flex-col gap-6"> + {downloadOutcome && <DownloadOutcomeBadge outcome={downloadOutcome} />} <PipelineStageCard id="download" title={noAudio ? "Download audio" : "Redownload audio"} @@ -492,13 +511,18 @@ function FilesList({ {formatSize(f.size)} · {new Date(f.mtime).toLocaleString()} </span> </div> - {kind && ( + {kind === "audio" || kind === "video" ? ( <MediaPlayer src={mediaUrl(slug, videoId, f.name)} kind={kind} filename={f.name} /> - )} + ) : kind === "text" ? ( + <TextFilePreview + src={mediaUrl(slug, videoId, f.name)} + filename={f.name} + /> + ) : null} <DeleteFileButton slug={slug} videoId={videoId} filename={f.name} /> </li> ); @@ -640,6 +664,63 @@ function DeleteVideoDirSection({ ); } +function DownloadOutcomeBadge({ + outcome, +}: { + outcome: DownloadOutcomeRecord; +}) { + if (outcome.status === "ok") return null; + const tone = + outcome.status === "failed" + ? "border-red-300 bg-red-50 text-red-900 dark:border-red-900 dark:bg-red-950 dark:text-red-200" + : "border-amber-300 bg-amber-50 text-amber-900 dark:border-amber-900 dark:bg-amber-950 dark:text-amber-200"; + const label = ((): string => { + switch (outcome.status) { + case "ok-with-cookies": + return "Retried with cookies"; + case "ok-auto-transcribed": + return "Auto-transcribed (no subs available)"; + case "failed": { + const lastAttempt = outcome.attempts[outcome.attempts.length - 1]; + const cls = lastAttempt?.availabilityClass; + return cls ? `Download failed: ${cls}` : "Download failed"; + } + default: + return outcome.status; + } + })(); + const lastErr = + outcome.status === "failed" + ? outcome.attempts[outcome.attempts.length - 1]?.error + : undefined; + return ( + <div + role="status" + aria-label="download outcome" + className={`rounded border px-3 py-2 text-sm ${tone}`} + > + <div className="flex flex-wrap items-baseline justify-between gap-2"> + <span className="font-medium">{label}</span> + <span className="text-xs opacity-70"> + {outcome.attempts.length} attempt + {outcome.attempts.length === 1 ? "" : "s"} ·{" "} + {new Date(outcome.finishedAt).toLocaleString()} + </span> + </div> + {lastErr && ( + <details className="mt-1"> + <summary className="cursor-pointer text-xs opacity-80"> + error tail + </summary> + <pre className="mt-1 whitespace-pre-wrap text-xs font-mono opacity-90"> + {lastErr} + </pre> + </details> + )} + </div> + ); +} + function Heading({ title, desc }: { title: string; desc: string }) { return ( <div> diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -5,6 +5,7 @@ import path from "node:path"; import { readdir, readFile, stat } from "node:fs/promises"; import type { Dirent } from "node:fs"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { loadDownloadOutcome } from "yt-dlp-transcript-common/lib/downloadOutcome-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { detectPlatform, @@ -91,6 +92,9 @@ export default async function VideoDetailPage({ if (!config) notFound(); const dirData = await loadVideoDir(slug, id); const meta = await loadMeta(slug, id); + const downloadOutcome = await loadDownloadOutcome( + path.join(getPaths().channelsDir, slug, "data", id), + ); const existingQueues = getRegistry().activeQueueNames(); const defaultQueueKey = platformQueueKey( @@ -146,6 +150,7 @@ export default async function VideoDetailPage({ audioFormat={config.audioFormat ?? "mp3"} defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} + downloadOutcome={downloadOutcome} /> </div> ); diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -22,6 +22,9 @@ export async function saveSettingsAction( const maxBytesRaw = String(formData.get("maxTranscriptPageBytes") ?? "").trim(); const transcribeBin = String(formData.get("transcribeBin") ?? "").trim(); const transcribeModel = String(formData.get("transcribeModel") ?? "").trim(); + const cookiesFromBrowser = String( + formData.get("cookiesFromBrowser") ?? "", + ).trim(); const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? ""); const transcribeArgs = transcribeArgsRaw .split("\n") @@ -60,6 +63,7 @@ export async function saveSettingsAction( transcribeBin, transcribeModel, transcribeArgs, + cookiesFromBrowser, }; await writeSettings(next); revalidatePath("/settings"); diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -87,6 +87,12 @@ export function SettingsForm({ initial }: Props) { </span> </label> </fieldset> + <Field + label="Cookies from browser (retry only)" + name="cookiesFromBrowser" + defaultValue={initial.cookiesFromBrowser} + hint="Browser spec passed to yt-dlp --cookies-from-browser only when the primary download attempt fails with an auth/age error and the channel doesn't have its own cookies set. e.g. firefox, chrome:Default. Leave blank to disable." + /> <div className="flex items-center gap-3"> <button type="submit" diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts @@ -265,6 +265,93 @@ test("Mark untranscribable button is absent when transcript already exists", asy ).toHaveCount(0); }); +test("text files render a preview, not the audio player", async ({ page }) => { + await resetData("one-youtube-channel-with-data"); + const dir = resolvePath( + "test-transcripts/channels/test-youtube/data/20240101_test1234567", + ); + // Stage misleadingly-named text files plus a real audio.mp3 in the same + // video dir. Old code matched name.startsWith("audio.") and would render + // <audio> for audio.vtt and audio.json. + await writeFile(`${dir}/audio.vtt`, "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nfake cue\n"); + await writeFile(`${dir}/audio.json`, '{"hello":"world"}\n'); + await writeFile(`${dir}/audio.mp3`, "binary audio bytes"); + + await page.goto("/channels/test-youtube/videos/20240101_test1234567"); + + const fileList = page.getByLabel("files for 20240101_test1234567"); + + // audio.mp3 — real audio, MediaPlayer renders. + await expect(fileList.getByLabel("media player for audio.mp3")).toBeVisible(); + await expect(fileList.getByLabel("preview audio.mp3")).toHaveCount(0); + + // audio.vtt — text file, preview details renders, no audio player. + await expect(fileList.getByLabel("media player for audio.vtt")).toHaveCount(0); + const vttPreview = fileList.getByLabel("preview audio.vtt"); + await expect(vttPreview).toBeVisible(); + // Content is not fetched until the user expands. + await expect(fileList.getByLabel("preview of audio.vtt")).toHaveCount(0); + await vttPreview.getByText("Preview contents").click(); + await expect(fileList.getByLabel("preview of audio.vtt")).toContainText( + "fake cue", + ); + + // audio.json — text file, same shape. + await expect(fileList.getByLabel("media player for audio.json")).toHaveCount(0); + const jsonPreview = fileList.getByLabel("preview audio.json"); + await expect(jsonPreview).toBeVisible(); + await jsonPreview.getByText("Preview contents").click(); + await expect(fileList.getByLabel("preview of audio.json")).toContainText( + '"hello":"world"', + ); +}); + +test("audioFiles count excludes audio.vtt and audio.json", async ({ page }) => { + await resetData("one-youtube-channel-with-data"); + const dir = resolvePath( + "test-transcripts/channels/test-youtube/data/20240101_test1234567", + ); + await writeFile(`${dir}/audio.vtt`, "WEBVTT\n"); + await writeFile(`${dir}/audio.json`, "{}\n"); + await writeFile(`${dir}/audio.mp3`, "fake audio"); + + await page.goto("/channels/test-youtube/videos/20240101_test1234567"); + + // Download stage summary counts only the real audio file. + await expect(page.getByLabel("Redownload audio stage summary")).toContainText( + "1 audio file on disk.", + ); + // Transcode summary too. + await expect(page.getByLabel("Transcode stage summary")).toContainText( + "Convert 1 audio file to another format.", + ); + // Per-file transcode row exists for audio.mp3 only — not for the text files. + await expect( + page.getByRole("heading", { name: "Transcode audio.mp3" }), + ).toBeVisible(); + await expect( + page.getByRole("heading", { name: "Transcode audio.vtt" }), + ).toHaveCount(0); + await expect( + page.getByRole("heading", { name: "Transcode audio.json" }), + ).toHaveCount(0); +}); + +test("existing transcript text files are inspectable from the file list", async ({ + page, +}) => { + await resetData("one-youtube-channel-with-data"); + await page.goto("/channels/test-youtube/videos/20240101_test1234567"); + + const fileList = page.getByLabel("files for 20240101_test1234567"); + const vttPreview = fileList.getByLabel("preview transcript.en.vtt"); + await expect(vttPreview).toBeVisible(); + await vttPreview.getByText("Preview contents").click(); + await expect(fileList.getByLabel("preview of transcript.en.vtt")).toContainText( + "synthetic test transcript", + ); +}); + test("Redownload writes audio in the chosen format", async ({ page }) => { await resetData("one-youtube-channel-with-data"); // Stage a video dir whose name matches the URL id so fake-ytdlp lands its