Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 931b48980c8611e874bb1ba04e9bea7eabc9ef8b
parent 20b533f2a9f19b3bd4d3ad498a1d2a5149188449
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 28 May 2026 13:55:15 -0400

app-handled video directories

Diffstat:
Acommon/bin/reconcile-video-dirs.ts | 81+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/channelSnapshot.ts | 51++++++++++++++++++++++++++++++++-------------------
Acommon/controller/reconcileVideoDirs.ts | 175+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/undownloadedVideos.ts | 18++++--------------
Mcommon/ytdlp/downloadOneManaged.ts | 119+++++++++++++++++++++++++++++++++++++++----------------------------------------
Mcommon/ytdlp/runYtdlp.ts | 466++++++++++++++++++++++++++++++++++++++++++++++---------------------------------
Meditor/CHANGELOG.md | 2++
Meditor/e2e/availability-backfill.spec.ts | 13++++++-------
Meditor/e2e/pipeline.spec.ts | 32++++++++++++++++++++++----------
Aeditor/e2e/reconcile.spec.ts | 84+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
10 files changed, 736 insertions(+), 305 deletions(-)

diff --git a/common/bin/reconcile-video-dirs.ts b/common/bin/reconcile-video-dirs.ts @@ -0,0 +1,81 @@ +#!/usr/bin/env tsx +import path from "node:path"; +import { readdir, stat } from "node:fs/promises"; +import { getPaths } from "../lib/paths"; +import { + reconcileVideoDirs, + type ReconcileResult, +} from "../controller/reconcileVideoDirs"; +import { parseFlags } from "./_parseFlags"; + +// Migrate / repair video data dirs so they match the app's canonical id layout +// (`data/<canonicalId>/`). See common/controller/reconcileVideoDirs.ts. +// +// Usage: +// reconcile-video-dirs.ts [--channel <slug>] [--dry-run] [--verbose] +// +// --channel reconcile only that channel (default: every channel with a data/) +// --dry-run report what would change without touching the filesystem +// --verbose print each rename/merge/skip line + +const flags = parseFlags(process.argv.slice(2)); +const dryRun = flags["dry-run"] === "true"; +const verbose = flags.verbose === "true"; +const paths = getPaths(); + +async function channelHasData(slug: string): Promise<boolean> { + try { + return ( + await stat(path.join(paths.channelsDir, slug, "data")) + ).isDirectory(); + } catch { + return false; + } +} + +async function listChannels(): Promise<string[]> { + const entries = await readdir(paths.channelsDir, { + withFileTypes: true, + }).catch(() => []); + const dirs = entries.filter((e) => e.isDirectory()).map((e) => e.name); + const withData: string[] = []; + for (const slug of dirs) if (await channelHasData(slug)) withData.push(slug); + return withData.sort(); +} + +function summarize(slug: string, r: ReconcileResult): void { + console.log( + `${slug}: renamed ${r.renamed.length}, merged ${r.merged.length}, ` + + `conflicts ${r.conflicts.length}, skipped ${r.skipped.length}`, + ); + for (const c of r.conflicts) { + console.error(` CONFLICT ${c.from} -> ${c.to}: ${c.reason}`); + } +} + +async function main(): Promise<void> { + const channels = flags.channel ? [flags.channel] : await listChannels(); + if (channels.length === 0) { + console.error("No channels with a data/ directory found."); + process.exit(2); + } + if (dryRun) console.log("(dry-run: no files will be moved)\n"); + + let totalConflicts = 0; + for (const slug of channels) { + const result = await reconcileVideoDirs({ + channelDir: path.join(paths.channelsDir, slug), + dryRun, + onLog: verbose ? (s) => process.stdout.write(` ${s}`) : undefined, + }); + summarize(slug, result); + totalConflicts += result.conflicts.length; + } + + if (totalConflicts > 0) process.exit(1); +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -19,7 +19,8 @@ import { resolveEffectiveAvailability, } from "../lib/availability-server"; import type { Paths } from "../lib/paths"; -import { extractVideoId, isRumbleUrl } from "../ytdlp/runYtdlp"; +import { extractVideoId } from "../ytdlp/runYtdlp"; +import { reconcileVideoDirs } from "./reconcileVideoDirs"; import { loadFailedTranscriptions } from "./failedTranscriptions"; import { readChannelConfig } from "./channels"; @@ -141,6 +142,14 @@ export async function generateChannelSnapshot( const archivePath = path.join(channelDir, "archive"); const playlistPath = path.join(channelDir, "playlist"); + // Heal any video dir that drifted from the canonical id layout before we read + // data/* (best-effort; never fail snapshot generation on a reconcile error). + try { + await reconcileVideoDirs({ channelDir }); + } catch { + /* ignore — the migration CLI can repair stragglers */ + } + const [dirEntries, urls, archive, failedListed, config] = await Promise.all([ readdir(dataDir, { withFileTypes: true }).catch(() => [] as Dirent[]), readPlaylistUrls(playlistPath), @@ -154,7 +163,13 @@ export async function generateChannelSnapshot( const videoDirNames = dirEntries .filter((d) => d.isDirectory()) .map((d) => d.name); - const wantRumbleIndex = urls.some(isRumbleUrl); + // The archive stores yt-dlp's native extractor ids, which equal the dir name + // (canonical id) only on YouTube. When the archive has any non-youtube + // extractor, read each video's native id from metadata so missingFromArchive + // compares like-for-like. + const wantNativeIndex = [...archive.byExtractor.keys()].some( + (e) => !e.toLowerCase().startsWith("youtube"), + ); const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY); const perVideo = await Promise.all( @@ -162,16 +177,16 @@ export async function generateChannelSnapshot( limit(async () => { const dir = path.join(dataDir, id); const files = await readVideoFiles(dir, { checkUntranscribable: true }); - let webpageUrl: string | null = null; - if (wantRumbleIndex && files.hasMeta) { + let nativeId: string | null = null; + if (wantNativeIndex && files.hasMeta) { try { const raw = await readFile( path.join(dir, "metadata.info.json"), "utf8", ); const parsed = JSON.parse(raw); - if (typeof parsed?.webpage_url === "string") { - webpageUrl = parsed.webpage_url; + if (typeof parsed?.id === "string") { + nativeId = parsed.id; } } catch { // ignore @@ -182,7 +197,7 @@ export async function generateChannelSnapshot( return { id, files, - webpageUrl, + nativeId, availability, effectiveAvailability, }; @@ -191,13 +206,13 @@ export async function generateChannelSnapshot( ); const filesById = new Map<string, VideoFiles>(); - const rumbleIndex = new Map<string, string>(); + // Native (yt-dlp extractor) ids present on disk, for the archive comparison. + // Dir names are canonical ids and double as native ids on YouTube. + const nativeIdsOnDisk = new Set<string>(); for (const v of perVideo) { filesById.set(v.id, v.files); - if (v.webpageUrl) { - const slugId = extractVideoId(v.webpageUrl); - if (slugId) rumbleIndex.set(slugId, v.id); - } + nativeIdsOnDisk.add(v.id); + if (v.nativeId) nativeIdsOnDisk.add(v.nativeId); } // Computed up-front so the buckets below can suppress IDs that can't be @@ -274,7 +289,7 @@ export async function generateChannelSnapshot( const missingFromArchive: string[] = []; for (const id of archive.ids) { - if (!filesById.has(id)) missingFromArchive.push(id); + if (!nativeIdsOnDisk.has(id)) missingFromArchive.push(id); } const excludedFromDownload = emptyExcludedFromDownload(); @@ -289,12 +304,10 @@ export async function generateChannelSnapshot( const undownloadedIds: string[] = []; for (const url of urls) { - const slugId = extractVideoId(url); - if (!slugId) continue; - const dirId = - isRumbleUrl(url) && rumbleIndex.has(slugId) - ? (rumbleIndex.get(slugId) as string) - : slugId; + // Post-reconcile a video's dir is its canonical id, so the URL's canonical + // id is the dir name directly. + const dirId = extractVideoId(url); + if (!dirId) continue; const f = filesById.get(dirId); if (f && videoHasAnyArtifact(f)) continue; if (excludedById.has(dirId)) continue; diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts @@ -0,0 +1,175 @@ +import path from "node:path"; +import { readdir, readFile, rename, rmdir, stat } from "node:fs/promises"; +import { extractVideoId } from "../ytdlp/runYtdlp"; + +// The app keys every video by its canonical, URL-derived id +// (`extractVideoId(webpage_url)`) and expects its data under +// `data/<canonicalId>/`. Historically the download pipeline let yt-dlp pick the +// directory via `%(id)s` (the extractor id), which diverges from the canonical +// id on Twitch (`v<id>`), Rumble, and Odysee. That split a video's bytes +// (`audio.*`, `metadata.info.json`) from its app-written sidecars +// (`download-outcome.json`). This module renames/merges those stray dirs back +// to the canonical layout. It is the one-time migration for existing data and a +// defensive safety pass before the snapshot reads `data/*`. + +export type ReconcileResult = { + renamed: { from: string; to: string }[]; + merged: { from: string; to: string; movedFiles: string[] }[]; + skipped: { + dir: string; + reason: "no-metadata" | "no-webpage-url" | "no-canonical-id"; + }[]; + conflicts: { from: string; to: string; reason: string }[]; +}; + +export type ReconcileOpts = { + // Absolute path to the channel root (…/channels/<slug>). + channelDir: string; + dryRun?: boolean; + onLog?: (s: string) => void; +}; + +// Files the canonical dir's copy should win on a name collision: these are +// written by the app keyed on the canonical id and are the ones the UI links +// to. Everything else (audio.*, transcript.*, metadata.info.json, download.log) +// comes from yt-dlp and the source dir holds the real bytes, so the source +// wins. +const CANONICAL_WINS = new Set(["download-outcome.json", "availability.json"]); + +function sourceWins(filename: string): boolean { + return !CANONICAL_WINS.has(filename); +} + +async function readWebpageUrl(metaPath: string): Promise<string | null> { + let raw: string; + try { + raw = await readFile(metaPath, "utf8"); + } catch { + return null; + } + try { + const url = (JSON.parse(raw) as { webpage_url?: unknown }).webpage_url; + return typeof url === "string" && url ? url : null; + } catch { + return null; + } +} + +async function dirExists(p: string): Promise<boolean> { + try { + return (await stat(p)).isDirectory(); + } catch { + return false; + } +} + +// Merge every file from `srcDir` into `dstDir`, returning the moved filenames. +// Collisions never destroy data: the loser is preserved as +// `<name>.dup-<srcName>` in the destination. +async function mergeDir( + srcDir: string, + dstDir: string, + srcName: string, + dryRun: boolean, +): Promise<string[]> { + const moved: string[] = []; + const [srcEntries, dstEntries] = await Promise.all([ + readdir(srcDir).catch(() => [] as string[]), + readdir(dstDir).catch(() => [] as string[]), + ]); + const dstSet = new Set(dstEntries); + for (const f of srcEntries) { + const src = path.join(srcDir, f); + if (!dstSet.has(f)) { + if (!dryRun) await rename(src, path.join(dstDir, f)); + moved.push(f); + continue; + } + if (sourceWins(f)) { + // Keep the source bytes; stash the destination's copy aside. + if (!dryRun) { + await rename(path.join(dstDir, f), path.join(dstDir, `${f}.dup-${srcName}`)); + await rename(src, path.join(dstDir, f)); + } + moved.push(f); + } else { + // Keep the destination copy; stash the source's aside. + if (!dryRun) await rename(src, path.join(dstDir, `${f}.dup-${srcName}`)); + moved.push(`${f}.dup-${srcName}`); + } + } + return moved; +} + +export async function reconcileVideoDirs( + opts: ReconcileOpts, +): Promise<ReconcileResult> { + const { channelDir, dryRun = false } = opts; + const log = opts.onLog ?? (() => {}); + const dataDir = path.join(channelDir, "data"); + const result: ReconcileResult = { + renamed: [], + merged: [], + skipped: [], + conflicts: [], + }; + + const entries = await readdir(dataDir, { withFileTypes: true }).catch( + () => [], + ); + const dirNames = entries.filter((e) => e.isDirectory()).map((e) => e.name); + + for (const name of dirNames) { + try { + const srcDir = path.join(dataDir, name); + const webpageUrl = await readWebpageUrl( + path.join(srcDir, "metadata.info.json"), + ); + if (!webpageUrl) { + result.skipped.push({ dir: name, reason: "no-metadata" }); + continue; + } + const canonical = extractVideoId(webpageUrl); + if (!canonical) { + result.skipped.push({ dir: name, reason: "no-canonical-id" }); + continue; + } + if (canonical === name) continue; + + const dstDir = path.join(dataDir, canonical); + if (!(await dirExists(dstDir))) { + if (!dryRun) await rename(srcDir, dstDir); + result.renamed.push({ from: name, to: canonical }); + log(`renamed ${name} -> ${canonical}\n`); + continue; + } + + const movedFiles = await mergeDir(srcDir, dstDir, name, dryRun); + if (!dryRun) { + // Source should be empty now; remove it. If something raced in, leave + // it and flag a conflict rather than risk deleting data. + try { + await rmdir(srcDir); + } catch (err) { + result.conflicts.push({ + from: name, + to: canonical, + reason: `source dir not empty after merge: ${(err as Error).message}`, + }); + continue; + } + } + result.merged.push({ from: name, to: canonical, movedFiles }); + log(`merged ${name} -> ${canonical} (${movedFiles.length} files)\n`); + } catch (err) { + result.conflicts.push({ + from: name, + to: name, + reason: (err as Error).message, + }); + log(`conflict on ${name}: ${(err as Error).message}\n`); + } + } + + return result; +} diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts @@ -1,11 +1,7 @@ import path from "node:path"; import { readFile } from "node:fs/promises"; import type { Paths } from "../lib/paths"; -import { - buildRumbleSlugIndex, - extractVideoId, - isRumbleUrl, -} from "../ytdlp/runYtdlp"; +import { extractVideoId } from "../ytdlp/runYtdlp"; async function readPlaylistUrls(playlistPath: string): Promise<string[]> { let raw: string; @@ -38,16 +34,10 @@ export async function findVideoSourceUrl( } const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); if (urls.length === 0) return null; - const dataDir = path.join(channelRoot, "data"); - const rumbleIndex = urls.some(isRumbleUrl) - ? await buildRumbleSlugIndex(dataDir) - : null; + // Post-reconcile a video's dir name is its canonical id, so match the + // requested videoId against each URL's canonical id directly. for (const url of urls) { - const slugId = extractVideoId(url); - if (!slugId) continue; - const dirId = - rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slugId) ?? slugId : slugId; - if (dirId === videoId) return url; + if (extractVideoId(url) === videoId) return url; } return null; } diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -1,5 +1,6 @@ import path from "node:path"; import { appendFile, mkdir, readdir, readFile } from "node:fs/promises"; +import { createWriteStream, type WriteStream } from "node:fs"; import { execa } from "execa"; import { parseUnavailableFromStderr, @@ -18,7 +19,7 @@ import { import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import type { Paths } from "../lib/paths"; import { transcribeOneVideo } from "../controller/transcribeOne"; -import { extractVideoId, isRumbleUrl } from "./runYtdlp"; +import { extractVideoId, outputArgsForUrl } from "./runYtdlp"; import { runAudioCheckedYtdlp } from "./audioCheckedDownload"; const STDERR_TAIL_BYTES = 64 * 1024; @@ -44,16 +45,6 @@ export type ManagedDownloadOpts = { inlineTranscribeOnFallback?: boolean; }; -const OUTPUT_ARGS: string[] = [ - "-o", - "data/%(id)s/audio.%(ext)s", - "-o", - "subtitle:data/%(id)s/transcript", - "-o", - "infojson:data/%(id)s/metadata", - "--no-write-playlist-metafiles", -]; - function youtubeHandlingArgs(config: ChannelConfig): string[] { return [ "--write-auto-subs", @@ -171,37 +162,6 @@ const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([ "private", ]); -async function resolveVideoIdFromUrl( - url: string, - channelDir: string, -): Promise<string | null> { - // Most platforms: the URL slug IS the id used in data/<id>/. Rumble is the - // exception — its archive id (and on-disk dir name) is yt-dlp's internal - // RumbleEmbed id, not the URL slug. Build a minimal lookup using - // metadata.info.json files we just wrote. - const slug = extractVideoId(url); - if (!slug) return null; - if (!isRumbleUrl(url)) return slug; - const dataDir = path.join(channelDir, "data"); - const dirs = await readdir(dataDir).catch(() => [] as string[]); - for (const dir of dirs) { - try { - const raw = await readFile( - path.join(dataDir, dir, "metadata.info.json"), - "utf8", - ); - const parsed = JSON.parse(raw) as { webpage_url?: unknown }; - if (typeof parsed.webpage_url === "string") { - const dirSlug = extractVideoId(parsed.webpage_url); - if (dirSlug === slug) return dir; - } - } catch { - continue; - } - } - return slug; -} - async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> { const entries = await readdir(videoDir).catch(() => [] as string[]); return entries.some((e) => { @@ -268,6 +228,49 @@ export async function downloadOneManaged( const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug); await mkdir(channelDir, { recursive: true }); + // We pin yt-dlp's output to data/<canonicalId>/ (see outputArgsForUrl), so we + // know the dir before yt-dlp runs. Tee every log line into a per-video + // download.log next to the sidecar (overwritten per managed download, like + // download-outcome.json). Falls back to no-log when the URL has no canonical + // id (outputArgsForUrl then uses %(id)s and the reconcile pass repairs it). + const canonicalId = extractVideoId(opts.videoUrl); + let logStream: WriteStream | null = null; + if (canonicalId) { + const dir = path.join(channelDir, "data", canonicalId); + try { + await mkdir(dir, { recursive: true }); + const stream = createWriteStream(path.join(dir, "download.log")); + logStream = stream; + const base = opts.onLog; + opts = { + ...opts, + onLog: (s) => { + try { + stream.write(s); + } catch { + /* logging is best-effort */ + } + base(s); + }, + }; + } catch { + /* per-video log is best-effort; fall back to opts.onLog */ + } + } + + try { + return await runManagedDownload(opts, channelDir, startedAt, canonicalId); + } finally { + logStream?.end(); + } +} + +async function runManagedDownload( + opts: ManagedDownloadOpts, + channelDir: string, + startedAt: string, + canonicalId: string | null, +): Promise<DownloadOutcomeRecord> { const attempts: DownloadAttempt[] = []; let status: DownloadOutcomeStatus = "failed"; let fellBackToTranscribe = false; @@ -290,7 +293,7 @@ export async function downloadOneManaged( const primaryArgs = [ "--ignore-config", "--restrict-filenames", - ...OUTPUT_ARGS, + ...outputArgsForUrl(opts.videoUrl), ...transcribeHandlingArgsForAudioCheck(opts.channelConfig), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, @@ -298,15 +301,12 @@ export async function downloadOneManaged( "--", opts.videoUrl, ]; - // Hint the orchestrator at which data/<id>/ subdir this launch will - // write to, so its .part discovery and final-file resolution don't - // latch onto stale .parts from prior interrupted attempts. For Rumble - // the on-disk dir is yt-dlp's internal id, only knowable after - // metadata.info.json appears — pass null and let the orchestrator - // fall back to the new-subdir heuristic. - const expectedVideoIdHint = isRumbleUrl(opts.videoUrl) - ? null - : extractVideoId(opts.videoUrl); + // Hint the orchestrator at which data/<id>/ subdir this launch will write + // to, so its .part discovery and final-file resolution don't latch onto + // stale .parts from prior interrupted attempts. outputArgsForUrl pins the + // output to data/<canonicalId>/ for every platform, so the canonical id is + // the dir. + const expectedVideoIdHint = canonicalId; const audioOutcome = await runAudioCheckedYtdlp({ paths: opts.paths, channelDir, @@ -336,7 +336,7 @@ export async function downloadOneManaged( const primaryArgs = [ "--ignore-config", "--restrict-filenames", - ...OUTPUT_ARGS, + ...outputArgsForUrl(opts.videoUrl), ...(opts.channelConfig.handling === "youtube" ? youtubeHandlingArgs(opts.channelConfig) : transcribeHandlingArgs(opts.channelConfig)), @@ -388,7 +388,7 @@ export async function downloadOneManaged( const retryArgs = [ "--ignore-config", "--restrict-filenames", - ...OUTPUT_ARGS, + ...outputArgsForUrl(opts.videoUrl), ...(opts.channelConfig.handling === "youtube" ? youtubeHandlingArgs(opts.channelConfig) : transcribeHandlingArgs(opts.channelConfig)), @@ -421,12 +421,9 @@ export async function downloadOneManaged( } // ---------- Attempt 3: no-subs fallback (youtube handling only) ---------- - const videoDir = audioCheckVideoDir - ?? path.join( - channelDir, - "data", - (await resolveVideoIdFromUrl(opts.videoUrl, channelDir)) ?? "unknown", - ); + const videoDir = + audioCheckVideoDir ?? + path.join(channelDir, "data", canonicalId ?? "unknown"); const videoId = path.basename(videoDir); if ( @@ -457,7 +454,7 @@ export async function downloadOneManaged( const fallbackArgs = [ "--ignore-config", "--restrict-filenames", - ...OUTPUT_ARGS, + ...outputArgsForUrl(opts.videoUrl), ...transcribeHandlingArgs(fallbackConfig), "--print", `after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`, diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -1,12 +1,5 @@ import path from "node:path"; -import { - mkdir, - readdir, - readFile, - rename, - rm, - writeFile, -} from "node:fs/promises"; +import { mkdir, readdir, readFile, rename, writeFile } from "node:fs/promises"; import { execa } from "execa"; import pLimit from "p-limit"; import { readArchive } from "../lib/archive"; @@ -17,6 +10,7 @@ import { type ChannelHandling, } from "../lib/channelConfig"; import { getSettings } from "../lib/settings"; +import { detectPlatform } from "../lib/platform"; import type { Paths } from "../lib/paths"; import { EXCLUDED_FROM_DOWNLOAD, @@ -144,33 +138,6 @@ function configArgs(config: ChannelConfig): string[] { return args; } -function handlingArgs(config: ChannelConfig): string[] { - if (config.handling === "youtube") { - return [ - "--write-auto-subs", - "--write-subs", - "--sub-langs", - config.subLangs ?? "en.*,live_chat", - "--write-info-json", - "--skip-download", - "-t", - "sleep", - ]; - } - // transcribe: download audio only or otherwise the smallest combined format; whisper runs as a follow-up job (Phase 7). - const fmt = config.audioFormat ?? "mp3"; - const args = [ - "--write-info-json", - "-f", - "bestaudio/worst", - "-x", - "--audio-format", - fmt, - ]; - if (config.keepSourceVideo) args.push("-k"); - return args; -} - const OUTPUT_ARGS: string[] = [ "-o", "data/%(id)s/audio.%(ext)s", @@ -183,20 +150,47 @@ const OUTPUT_ARGS: string[] = [ "--no-write-playlist-metafiles", ]; -async function storePlaylist(opts: RunYtdlpOpts): Promise<void> { - const root = channelRoot(opts); - await mkdir(root, { recursive: true }); - const playlistPath = path.join(root, "playlist"); - const tmpPath = `${playlistPath}.tmp-${process.pid}`; +// yt-dlp's `%(id)s` is the *extractor* id, which only matches our canonical, +// URL-derived id on YouTube. On Twitch (`v<id>` prefix), Rumble, and Odysee it +// diverges, splitting a video's bytes from the app's `data/<canonicalId>/` +// dir. For single-video spawns we know the URL up front, so we pin the output +// path literally to the canonical id and stop depending on the extractor id. +// Falls back to %(id)s for URLs whose canonical id is missing or not a safe +// directory name (the post-download reconcile pass cleans those up). +export function outputArgsForUrl(url: string): string[] { + const id = extractVideoId(url); + if (id && /^[\w.-]+$/.test(id) && id !== "." && id !== "..") { + return [ + "-o", + `data/${id}/audio.%(ext)s`, + "-o", + `subtitle:data/${id}/transcript`, + "-o", + `infojson:data/${id}/metadata`, + "--no-write-playlist-metafiles", + ]; + } + return OUTPUT_ARGS; +} +// Enumerate a channel's video URLs via `--flat-playlist --print url` (metadata +// only, no downloads). Pass `range` to fetch a single newest-first page via +// `-I start:end`; sync uses this to walk the channel incrementally. +async function enumeratePlaylistUrls( + opts: RunYtdlpOpts, + root: string, + range?: { start: number; end: number }, +): Promise<string[]> { const args: string[] = [ "--flat-playlist", "--skip-download", "--print", "url", - ...configArgs(opts.channelConfig), - opts.channelConfig.url!, ]; + if (range) { + args.push("--lazy-playlist", "-I", `${range.start}:${range.end}`); + } + args.push(...configArgs(opts.channelConfig), opts.channelConfig.url!); opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`); const child = execa(opts.paths.ytdlpBin, args, { @@ -210,19 +204,32 @@ async function storePlaylist(opts: RunYtdlpOpts): Promise<void> { const result = await child; // yt-dlp exit code convention: 101 = "break-on-existing" / "max-downloads" - // (clean stop, not an error). Treat it the same as 0. - if (result.exitCode !== 0 && result.exitCode !== 101) { + // (clean stop, not an error). Treat it the same as 0. A non-zero/101 exit + // while the run was cancelled is the cancel itself, not a failure. + if ( + result.exitCode !== 0 && + result.exitCode !== 101 && + !opts.signal.aborted + ) { throw new Error(`yt-dlp exited with code ${result.exitCode}`); } if (result.exitCode === 101) { opts.onLog(`yt-dlp stopped on existing entry (exit 101).\n`); } - const stdout = String(result.stdout ?? ""); - const urls = stdout + return String(result.stdout ?? "") .split("\n") .map((s) => s.trim()) .filter(Boolean); +} + +async function storePlaylist(opts: RunYtdlpOpts): Promise<void> { + const root = channelRoot(opts); + await mkdir(root, { recursive: true }); + const playlistPath = path.join(root, "playlist"); + const tmpPath = `${playlistPath}.tmp-${process.pid}`; + + const urls = await enumeratePlaylistUrls(opts, root); await writeFile(tmpPath, urls.join("\n") + (urls.length ? "\n" : "")); await rename(tmpPath, playlistPath); opts.onLog(`Wrote ${urls.length} URLs to ${playlistPath}\n`); @@ -272,22 +279,18 @@ async function downloadPlaylistManaged( .map((s) => s.trim()) .filter(Boolean); - const rumbleIndex = urls.some(isRumbleUrl) - ? await buildRumbleSlugIndex(dataDir) - : null; - + // Post-reconcile, a video's on-disk dir is always its canonical id + // (extractVideoId of the URL), so no slug→dir indirection is needed here. + // The one exception is the *archive* file, which keeps yt-dlp's native + // extractor ids — archiveIdForUrl bridges that. if (idAllowList) { const allow = new Set(idAllowList); const before = urls.length; const matched: string[] = []; const matchedIds = new Set<string>(); for (const url of urls) { - const slugId = extractVideoId(url); - if (!slugId) continue; - const dirId = - rumbleIndex && isRumbleUrl(url) - ? (rumbleIndex.get(slugId) ?? slugId) - : slugId; + const dirId = extractVideoId(url); + if (!dirId) continue; if (allow.has(dirId)) { matched.push(url); matchedIds.add(dirId); @@ -317,7 +320,7 @@ async function downloadPlaylistManaged( tofetch.push(...urls); } else { for (const url of urls) { - const archiveId = lookupArchiveId(url, rumbleIndex); + const archiveId = await archiveIdForUrl(url, dataDir); if (archiveId && archive.ids.has(archiveId)) { skipped++; continue; @@ -335,18 +338,13 @@ async function downloadPlaylistManaged( let alreadyComplete = 0; let unidentifiable = 0; for (const url of urls) { - const slug = extractVideoId(url); - if (!slug) { + const dirId = extractVideoId(url); + if (!dirId) { unidentifiable++; tofetch.push(url); continue; } - const dirId = - rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; - if ( - dirId && - (await destinationExists(dataDir, dirId, effectiveHandling)) - ) { + if (await destinationExists(dataDir, dirId, effectiveHandling)) { alreadyComplete++; continue; } @@ -366,12 +364,7 @@ async function downloadPlaylistManaged( const excludedCounts = { members_only: 0, deleted: 0, private: 0 }; const filteredTofetch: string[] = []; for (const url of tofetch) { - const slugId = extractVideoId(url); - const dirId = slugId - ? rumbleIndex && isRumbleUrl(url) - ? (rumbleIndex.get(slugId) ?? slugId) - : slugId - : null; + const dirId = extractVideoId(url); if (dirId) { const cls = await resolveEffectiveAvailability( path.join(dataDir, dirId), @@ -421,22 +414,52 @@ async function downloadPlaylistManaged( return; } + if (opts.ignoreArchive) { + opts.onLog( + "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n", + ); + } + + const { okCount, failedCount, processedCount, firstFailure } = + await runManagedDownloads(opts, items, effectiveChannelConfig); + if (firstFailure && !opts.signal.aborted) { + throw firstFailure; + } + opts.onLog( + `Managed download complete: ${okCount} succeeded, ${failedCount} failed (${processedCount}/${items.length} processed).\n`, + ); + + await touchLastFullDownload(opts); + await safeBackfillAvailability(opts); +} + +type ManagedRunResult = { + okCount: number; + failedCount: number; + processedCount: number; + // Non-null only when abortOnError is on and a non-per-video failure occurred. + firstFailure: Error | null; +}; + +// Serialize a list of video URLs through downloadOneManaged. Shared by the +// playlist-prefilter modes and `sync`. Serialized (pLimit(1)) to keep logs +// readable and avoid hammering the source with parallel requests (which is +// what often triggers needs_auth in the first place); the per-video startup +// cost is the price of being able to retry each independently. Honors the +// hard `signal`, the soft `drainSignal`, the per-channel sleep, and the +// per-operation tracker. +async function runManagedDownloads( + opts: RunYtdlpOpts, + urls: ReadonlyArray<string>, + effectiveChannelConfig: ChannelConfig, +): Promise<ManagedRunResult> { const settings = getSettings(); const globalCookies = settings.cookiesFromBrowser; const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback; const sleepSeconds = opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds; - if (opts.ignoreArchive) { - opts.onLog( - "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n", - ); - } - // Serialize within a run: keeps logs readable and avoids hammering the - // source with parallel requests (which is what often triggers needs_auth - // in the first place). The per-video startup cost is the price of being - // able to retry independently. const limit = pLimit(1); let failedCount = 0; const abortOnError = opts.abortOnError !== false; @@ -444,7 +467,7 @@ async function downloadPlaylistManaged( let processedCount = 0; await Promise.all( - items.map((url) => + urls.map((url) => limit(async () => { if (opts.signal.aborted) return; // Drain (soft-cancel): finish the in-flight download, start no more. @@ -474,8 +497,7 @@ async function downloadPlaylistManaged( if (outcome.status === "failed") { failedCount++; if (abortOnError && !firstFailure) { - const lastAttempt = - outcome.attempts[outcome.attempts.length - 1]; + const lastAttempt = outcome.attempts[outcome.attempts.length - 1]; const failureClass = classifyDownloadFailure( lastAttempt?.error ?? "", lastAttempt?.availabilityClass, @@ -491,7 +513,7 @@ async function downloadPlaylistManaged( } } processedCount++; - const isLast = processedCount >= items.length; + const isLast = processedCount >= urls.length; const willAbortLoop = firstFailure !== null && abortOnError; if ( sleepSeconds > 0 && @@ -506,16 +528,12 @@ async function downloadPlaylistManaged( ), ); - if (firstFailure && abortOnError && !opts.signal.aborted) { - throw firstFailure; - } - const okCount = processedCount - failedCount; - opts.onLog( - `Managed download complete: ${okCount} succeeded, ${failedCount} failed (${processedCount}/${items.length} processed).\n`, - ); - - await touchLastFullDownload(opts); - await safeBackfillAvailability(opts); + return { + okCount: processedCount - failedCount, + failedCount, + processedCount, + firstFailure, + }; } // yt-dlp's --sub-langs supports a comma-separated list with shell-glob style @@ -560,11 +578,37 @@ function trackOnDisk(entries: string[], track: string): boolean { return entries.some((e) => re.test(e)); } +// Fetch only the missing subtitle tracks for one already-downloaded video. +// Uses the literal canonical output path (outputArgsForUrl) so the .vtt files +// land in the same data/<canonicalId>/ dir as the rest of the video. +async function downloadSubsForUrl( + opts: RunYtdlpOpts, + root: string, + url: string, + subLangs: string, +): Promise<void> { + const args: string[] = [ + "--ignore-config", + "--restrict-filenames", + ...outputArgsForUrl(url), + "--write-auto-subs", + "--write-subs", + "--sub-langs", + subLangs, + "--skip-download", + "--no-write-info-json", + "--no-overwrites", + ...configArgs(opts.channelConfig), + "--", + url, + ]; + await runChildAndStream(opts, root, args); +} + async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { const root = channelRoot(opts); const dataDir = path.join(root, "data"); const playlistPath = path.join(root, "playlist"); - const toFetchPath = path.join(root, "playlist.tofetch-subs"); let playlistText: string; try { @@ -582,9 +626,6 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { const subLangs = opts.channelConfig.subLangs ?? "en.*,live_chat"; const matchesSubLang = compileSubLangMatcher(subLangs); - const rumbleIndex = urls.some(isRumbleUrl) - ? await buildRumbleSlugIndex(dataDir) - : null; const tofetch: string[] = []; let upToDate = 0; let unidentifiable = 0; @@ -592,17 +633,11 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { let noExpectedTracks = 0; for (const url of urls) { - const slug = extractVideoId(url); - if (!slug) { - unidentifiable++; - continue; - } - const dirId = - rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug; + // Post-reconcile a video lives in data/<canonicalId>/, so the URL's + // canonical id is the dir name directly — no slug→dir indirection. + const dirId = extractVideoId(url); if (!dirId) { - // No corresponding data dir yet — skip; download-from-playlist / - // download-missing should pull the video and its metadata first. - noMetadata++; + unidentifiable++; continue; } const videoDir = path.join(dataDir, dirId); @@ -613,6 +648,8 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { "utf8", ); } catch { + // No corresponding data dir yet — skip; download-from-playlist / + // download-missing should pull the video and its metadata first. noMetadata++; continue; } @@ -656,32 +693,48 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { ); if (tofetch.length === 0) { - await rm(toFetchPath, { force: true }); opts.onLog("Nothing to fetch.\n"); return; } - await writeFile(toFetchPath, tofetch.join("\n") + "\n"); - - const args: string[] = [ - "--ignore-config", - "--restrict-filenames", - ...OUTPUT_ARGS, - "--write-auto-subs", - "--write-subs", - "--sub-langs", - subLangs, - "--skip-download", - "--no-write-info-json", - "--no-overwrites", - ...(opts.abortOnError === false ? [] : ["--abort-on-error"]), - "-a", - "playlist.tofetch-subs", - ...configArgs(opts.channelConfig), - ]; - await runChildAndStream(opts, root, args); + // One yt-dlp spawn per video (serialized), mirroring the download paths. + const abortOnError = opts.abortOnError !== false; + const limit = pLimit(1); + let processed = 0; + let failed = 0; + let firstFailure: Error | null = null; + await Promise.all( + tofetch.map((url) => + limit(async () => { + if (opts.signal.aborted || opts.drainSignal?.aborted) return; + if (firstFailure) return; + const task = opts.tracker?.start({ + id: extractVideoId(url) ?? url, + label: extractVideoId(url) ?? url, + kind: "download", + }); + try { + await downloadSubsForUrl( + task ? { ...opts, onLog: task.onLog } : opts, + root, + url, + subLangs, + ); + } catch (err) { + failed++; + if (abortOnError && !firstFailure) firstFailure = err as Error; + } finally { + task?.end(); + } + processed++; + }), + ), + ); - await rm(toFetchPath, { force: true }); + if (firstFailure && !opts.signal.aborted) throw firstFailure; + opts.onLog( + `Sub download complete: ${processed - failed} succeeded, ${failed} failed (of ${tofetch.length}).\n`, + ); } async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { @@ -695,7 +748,7 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> { const args: string[] = [ "--ignore-config", "--restrict-filenames", - ...OUTPUT_ARGS, + ...outputArgsForUrl(opts.singleVideoUrl), "--write-info-json", "-f", "bestaudio/worst", @@ -737,25 +790,78 @@ export async function destinationExists( return false; } +// How many newest-first playlist entries to enumerate per sync page. Each page +// is one cheap flat-playlist call; new videos on it are fetched, and we stop at +// the first page that contains an already-archived entry. +const SYNC_PAGE_SIZE = 50; + async function sync(opts: RunYtdlpOpts): Promise<void> { const root = channelRoot(opts); + const dataDir = path.join(root, "data"); + const archivePath = path.join(root, "archive"); + // Create the channel root (cwd for enumeration) but NOT data/ — the per-video + // downloads create their own canonical dirs, so a sync that's cancelled + // before any download leaves no data/ behind. await mkdir(root, { recursive: true }); - const args: string[] = [ - "--ignore-config", - "--restrict-filenames", - ...OUTPUT_ARGS, - ...handlingArgs(opts.channelConfig), - "--force-write-archive", - "--download-archive", - "archive", - "--break-on-existing", - "--lazy-playlist", - ...configArgs(opts.channelConfig), - "--", - opts.channelConfig.url!, - ]; - await runChildAndStream(opts, root, args); + // Walk the channel newest-first a page at a time. On each page, download the + // entries not yet in the archive, then stop once we reach a page that + // contains an already-archived entry (we've caught up to a prior sync) or a + // short final page. Each video is fetched via downloadOneManaged — same + // auth-retry / audio-check / no-subs-fallback handling as every other + // download path — and lands directly in its canonical data/<id>/ dir. Deeper + // mid-channel gaps remain the job of "download missing", exactly as before. + let archive = await readArchive(archivePath); + let totalNew = 0; + let totalFailed = 0; + let firstFailure: Error | null = null; + + for ( + let page = 0; + !opts.signal.aborted && !opts.drainSignal?.aborted; + page++ + ) { + const start = page * SYNC_PAGE_SIZE + 1; + const end = start + SYNC_PAGE_SIZE - 1; + const pageUrls = await enumeratePlaylistUrls(opts, root, { start, end }); + if (pageUrls.length === 0) break; + + const newUrls: string[] = []; + let archivedHits = 0; + for (const url of pageUrls) { + const archiveId = await archiveIdForUrl(url, dataDir); + if (archiveId && archive.ids.has(archiveId)) { + archivedHits++; + continue; + } + newUrls.push(url); + } + opts.onLog( + `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`, + ); + + if (newUrls.length > 0) { + const res = await runManagedDownloads(opts, newUrls, opts.channelConfig); + totalNew += res.okCount; + totalFailed += res.failedCount; + if (res.firstFailure) { + firstFailure = res.firstFailure; + break; + } + // Successful downloads appended to the archive; refresh so the next + // page's diff sees them. + archive = await readArchive(archivePath); + } + + if (archivedHits > 0) break; // reached previously-synced content + if (pageUrls.length < SYNC_PAGE_SIZE) break; // last page + } + + if (firstFailure && !opts.signal.aborted) throw firstFailure; + + opts.onLog( + `Sync complete: ${totalNew} new downloaded, ${totalFailed} failed.\n`, + ); await touchLastSync(opts); await safeBackfillAvailability(opts); } @@ -834,58 +940,30 @@ async function updateConfigField( await rename(tmp, configPath); } -export function isRumbleUrl(url: string): boolean { +// The channel's archive file stores yt-dlp's native extractor ids (e.g. +// `twitch:vod v123`, `rumble <internal>`), which diverge from our canonical id +// on every platform except YouTube. A downloaded video records its native id as +// `id` inside metadata.info.json, which (post-reconcile) lives in the canonical +// dir. Resolve it from there; for not-yet-downloaded videos (no dir on disk) or +// YouTube (native == canonical) fall back to the canonical id. +export async function archiveIdForUrl( + url: string, + dataDir: string, +): Promise<string | null> { + const canonical = extractVideoId(url); + if (!canonical) return null; + if (detectPlatform(url) === "youtube") return canonical; try { - return new URL(url).hostname.toLowerCase().endsWith("rumble.com"); + const raw = await readFile( + path.join(dataDir, canonical, "metadata.info.json"), + "utf8", + ); + const id = (JSON.parse(raw) as { id?: unknown }).id; + if (typeof id === "string" && id) return id; } catch { - return false; - } -} - -// Rumble's URL slug (the part between "rumble.com/" and the first "-") is a -// different identifier from yt-dlp's RumbleEmbed extractor `id`, which is what -// ends up in the archive file and as the on-disk data dir name. To prefilter -// Rumble playlist URLs against the archive / data dir, we walk every existing -// data/{id}/metadata.info.json once, read its `webpage_url`, and build a -// slug → id map. -export async function buildRumbleSlugIndex( - dataDir: string, -): Promise<Map<string, string>> { - const index = new Map<string, string>(); - const dirs = await readdir(dataDir).catch(() => [] as string[]); - for (const id of dirs) { - let raw: string; - try { - raw = await readFile( - path.join(dataDir, id, "metadata.info.json"), - "utf8", - ); - } catch { - continue; - } - let url: unknown; - try { - url = JSON.parse(raw)?.webpage_url; - } catch { - continue; - } - if (typeof url !== "string") continue; - const slug = extractVideoId(url); - if (slug) index.set(slug, id); - } - return index; -} - -function lookupArchiveId( - url: string, - rumbleIndex: Map<string, string> | null, -): string | null { - const slug = extractVideoId(url); - if (!slug) return null; - if (rumbleIndex && isRumbleUrl(url)) { - return rumbleIndex.get(slug) ?? null; + /* not downloaded yet — fall through to canonical */ } - return slug; + return canonical; } export function extractVideoId(url: string): string | null { diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] - **Twitch.tv support.** Twitch is now a first-class platform: Twitch channel/VOD URLs are auto-detected, "Twitch" is selectable in the channel form's platform dropdown and the charts platform filter, videos play via an in-browser Twitch embed, and downloads are queued on a `platform:twitch` queue like the other platforms. +- **Downloads always land in the canonical `data/<id>/` dir.** Previously yt-dlp chose the directory from its own extractor id, which diverges from the app's URL-derived id on Twitch (`v<id>` prefix), Rumble, and Odysee — so a video's audio/metadata and its `download-outcome.json` could end up split across two sibling dirs and the editor couldn't resolve the video. The app now pins yt-dlp's output path per video, and a reconcile pass (run automatically when a channel's report regenerates) merges any pre-existing split dirs back together. A one-time migration command, `reconcile-video-dirs`, repairs all existing channels (`--dry-run` to preview). Sync and "download missing subtitles" now fetch one video per yt-dlp run (sync walks the channel a page at a time, downloading the diff against the archive and stopping once it reaches already-synced videos). +- **Per-video download logs.** Each managed download now writes a `download.log` alongside the video's data, capturing that run's full yt-dlp output for after-the-fact debugging. - **Smarter queues for unrecognized sources.** When a channel's platform can't be detected, its jobs are now queued per-domain (e.g. `platform:vimeo.com`, with subdomains stripped) instead of all sharing the single `platform:unknown` queue — so unrelated unknown sources no longer block one another. - **Charts authoring + stats dataset.** A new **Charts** tab lets you author the default chart dashboard that ships to the viewer — add/edit/remove charts and configure axes, metrics, series, filters, and search-derived series with a live preview; edits save automatically and bake into the next export build. Search-derived charts now use the full layered query builder (AND/OR/NOT, nesting, per-layer scope/regex), matching the search page. A new **Build stats dataset** action on the Build page extracts per-video stats (views, likes, comments, follower count, duration, categories, language, cue counts) into `export/public/stats/` for the charts to read; it's incremental (mtime short-circuit) and also runs automatically as part of the static export build. - **Per-operation progress bars on `/jobs/active`.** Each running download/transcription now shows its own live progress bar parsed from the tool's shell output — yt-dlp's download percent (and fragment count) and whisper's transcribed position against the audio length. Batch jobs (Transcribe missing, Download from playlist, Sync, …) list a bar per in-flight video underneath the batch's overall bar; standalone single-video jobs get one too. The screen now polls about once a second so the bars advance live. diff --git a/editor/e2e/availability-backfill.spec.ts b/editor/e2e/availability-backfill.spec.ts @@ -40,13 +40,12 @@ test("download backfills availability.json from metadata", async ({ page }) => { await seedVideo("seededPlain1", {}); await page.goto(`/channels/${CHANNEL}`); - // Sync triggers downloadOneAudio path? Actually "Sync" runs the sync mode - // (--lazy-playlist). The fake-ytdlp emits one fakeSync0001 video. After - // success, runYtdlp calls backfillAvailabilityFromMetadata, which sweeps - // ALL existing data dirs. + // Sync pages the flat-playlist (fake emits fake0000000{1..5}) and downloads + // each via the managed per-URL path. After it finishes, runYtdlp calls + // backfillAvailabilityFromMetadata, which sweeps ALL existing data dirs. await page.getByRole("button", { name: "Sync" }).click(); await expect(page.getByLabel("Sync output")).toContainText("backfill", { - timeout: 20_000, + timeout: 30_000, }); // Seeded videos should now have availability.json written from metadata. @@ -68,10 +67,10 @@ test("download backfills availability.json from metadata", async ({ page }) => { ); expect(plain.availability).toBe("public"); - // The newly-fetched fakeSync0001 also gets a sidecar from its metadata. + // A newly-fetched video also gets a sidecar from its metadata. expect( await pathExists( - `test-transcripts/channels/${CHANNEL}/data/fakeSync0001/availability.json`, + `test-transcripts/channels/${CHANNEL}/data/fake00000001/availability.json`, ), ).toBe(true); }); diff --git a/editor/e2e/pipeline.spec.ts b/editor/e2e/pipeline.spec.ts @@ -43,6 +43,12 @@ test("download from playlist fetches all 5 URLs and writes the archive", async ( ), ).toBe(true); } + // Each managed download writes a per-video log alongside the data. + expect( + await pathExists( + "test-transcripts/channels/test-pipeline/data/fake00000001/download.log", + ), + ).toBe(true); const config = await readJson<{ lastFullDownloadAt?: string }>( "test-transcripts/channels/test-pipeline/config.json", ); @@ -72,19 +78,25 @@ test("re-running download skips already-archived entries", async ({ page }) => { await expect(downloadLog).toContainText("Nothing to fetch"); }); -test("sync downloads one new entry and writes lastSyncedAt", async ({ page }) => { +test("sync enumerates the channel, downloads new entries, writes lastSyncedAt", async ({ + page, +}) => { + test.setTimeout(120_000); await resetData("test-pipeline"); await page.goto("/channels/test-pipeline"); await page.getByRole("button", { name: "Sync" }).click(); - await expect(page.getByLabel("Sync output")).toContainText( - "fakeSync0001 done", - { timeout: 30_000 }, - ); - expect( - await pathExists( - "test-transcripts/channels/test-pipeline/data/fakeSync0001/metadata.info.json", - ), - ).toBe(true); + // Sync now pages the flat-playlist (fake emits 5 entries) and downloads the + // archive diff per page via the managed per-URL path. + await expect(page.getByLabel("Sync output")).toContainText("Sync complete", { + timeout: 60_000, + }); + for (let i = 1; i <= 5; i++) { + expect( + await pathExists( + `test-transcripts/channels/test-pipeline/data/fake0000000${i}/metadata.info.json`, + ), + ).toBe(true); + } const config = await readJson<{ lastSyncedAt?: string }>( "test-transcripts/channels/test-pipeline/config.json", ); diff --git a/editor/e2e/reconcile.spec.ts b/editor/e2e/reconcile.spec.ts @@ -0,0 +1,84 @@ +import { mkdir, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath } from "./helpers"; + +// yt-dlp names Twitch VOD dirs `v<id>` (its extractor id) while the app keys +// videos by the canonical URL id `<id>`, so the audio/metadata land in +// data/v<id>/ and the app-written download-outcome.json lands in data/<id>/. +// generateChannelSnapshot (run on first channel-page load when no snapshot.json +// exists) now reconciles those dirs back to the canonical layout. + +const CHANNEL = "twitch-reconcile"; +const ROOT = `test-transcripts/channels/${CHANNEL}`; + +async function write(rel: string, contents: string): Promise<void> { + const full = resolvePath(`${ROOT}/${rel}`); + await mkdir(full.slice(0, full.lastIndexOf("/")), { recursive: true }); + await writeFile(full, contents); +} + +async function seedChannel(): Promise<void> { + await write( + "config.json", + JSON.stringify({ + handling: "transcribe", + name: "Twitch Reconcile", + url: "https://www.twitch.tv/reconcile/videos", + platform: "twitch", + }), + ); + + // Case 1 — merge: bytes in v123, sidecar already in canonical 123. + await write( + "data/v123/metadata.info.json", + JSON.stringify({ id: "v123", webpage_url: "https://www.twitch.tv/videos/123" }), + ); + await write("data/v123/audio.mp3", "fake audio 123"); + await write( + "data/123/download-outcome.json", + JSON.stringify({ videoId: "123", status: "ok" }), + ); + + // Case 2 — rename: bytes in v456, no canonical dir yet. + await write( + "data/v456/metadata.info.json", + JSON.stringify({ id: "v456", webpage_url: "https://www.twitch.tv/videos/456" }), + ); + await write("data/v456/audio.mp3", "fake audio 456"); + + // Case 3 — orphan with no metadata: can't determine canonical id, left alone. + await write( + "data/orphan999/download-outcome.json", + JSON.stringify({ videoId: "orphan999", status: "failed" }), + ); +} + +test("channel snapshot reconciles Twitch v<id> dirs into canonical <id> dirs", async ({ + page, +}) => { + await resetData(null); + await seedChannel(); + + // Loading the channel page generates the snapshot (no snapshot.json yet), + // which runs reconcileVideoDirs first. + await page.goto(`/channels/${CHANNEL}`); + await expect( + page.getByRole("heading", { name: "Twitch Reconcile" }), + ).toBeVisible({ timeout: 15_000 }); + + // Case 1: v123 merged into 123 — all three files now in 123, v123 gone. + expect(await pathExists(`${ROOT}/data/123/audio.mp3`)).toBe(true); + expect(await pathExists(`${ROOT}/data/123/metadata.info.json`)).toBe(true); + expect(await pathExists(`${ROOT}/data/123/download-outcome.json`)).toBe(true); + expect(await pathExists(`${ROOT}/data/v123`)).toBe(false); + + // Case 2: v456 renamed to 456. + expect(await pathExists(`${ROOT}/data/456/audio.mp3`)).toBe(true); + expect(await pathExists(`${ROOT}/data/456/metadata.info.json`)).toBe(true); + expect(await pathExists(`${ROOT}/data/v456`)).toBe(false); + + // Case 3: metadata-less orphan left untouched. + expect( + await pathExists(`${ROOT}/data/orphan999/download-outcome.json`), + ).toBe(true); +});