commit 931b48980c8611e874bb1ba04e9bea7eabc9ef8b
parent 20b533f2a9f19b3bd4d3ad498a1d2a5149188449
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 28 May 2026 13:55:15 -0400
app-handled video directories
Diffstat:
10 files changed, 736 insertions(+), 305 deletions(-)
diff --git a/common/bin/reconcile-video-dirs.ts b/common/bin/reconcile-video-dirs.ts
@@ -0,0 +1,81 @@
+#!/usr/bin/env tsx
+import path from "node:path";
+import { readdir, stat } from "node:fs/promises";
+import { getPaths } from "../lib/paths";
+import {
+ reconcileVideoDirs,
+ type ReconcileResult,
+} from "../controller/reconcileVideoDirs";
+import { parseFlags } from "./_parseFlags";
+
+// Migrate / repair video data dirs so they match the app's canonical id layout
+// (`data/<canonicalId>/`). See common/controller/reconcileVideoDirs.ts.
+//
+// Usage:
+// reconcile-video-dirs.ts [--channel <slug>] [--dry-run] [--verbose]
+//
+// --channel reconcile only that channel (default: every channel with a data/)
+// --dry-run report what would change without touching the filesystem
+// --verbose print each rename/merge/skip line
+
+const flags = parseFlags(process.argv.slice(2));
+const dryRun = flags["dry-run"] === "true";
+const verbose = flags.verbose === "true";
+const paths = getPaths();
+
+async function channelHasData(slug: string): Promise<boolean> {
+ try {
+ return (
+ await stat(path.join(paths.channelsDir, slug, "data"))
+ ).isDirectory();
+ } catch {
+ return false;
+ }
+}
+
+async function listChannels(): Promise<string[]> {
+ const entries = await readdir(paths.channelsDir, {
+ withFileTypes: true,
+ }).catch(() => []);
+ const dirs = entries.filter((e) => e.isDirectory()).map((e) => e.name);
+ const withData: string[] = [];
+ for (const slug of dirs) if (await channelHasData(slug)) withData.push(slug);
+ return withData.sort();
+}
+
+function summarize(slug: string, r: ReconcileResult): void {
+ console.log(
+ `${slug}: renamed ${r.renamed.length}, merged ${r.merged.length}, ` +
+ `conflicts ${r.conflicts.length}, skipped ${r.skipped.length}`,
+ );
+ for (const c of r.conflicts) {
+ console.error(` CONFLICT ${c.from} -> ${c.to}: ${c.reason}`);
+ }
+}
+
+async function main(): Promise<void> {
+ const channels = flags.channel ? [flags.channel] : await listChannels();
+ if (channels.length === 0) {
+ console.error("No channels with a data/ directory found.");
+ process.exit(2);
+ }
+ if (dryRun) console.log("(dry-run: no files will be moved)\n");
+
+ let totalConflicts = 0;
+ for (const slug of channels) {
+ const result = await reconcileVideoDirs({
+ channelDir: path.join(paths.channelsDir, slug),
+ dryRun,
+ onLog: verbose ? (s) => process.stdout.write(` ${s}`) : undefined,
+ });
+ summarize(slug, result);
+ totalConflicts += result.conflicts.length;
+ }
+
+ if (totalConflicts > 0) process.exit(1);
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -19,7 +19,8 @@ import {
resolveEffectiveAvailability,
} from "../lib/availability-server";
import type { Paths } from "../lib/paths";
-import { extractVideoId, isRumbleUrl } from "../ytdlp/runYtdlp";
+import { extractVideoId } from "../ytdlp/runYtdlp";
+import { reconcileVideoDirs } from "./reconcileVideoDirs";
import { loadFailedTranscriptions } from "./failedTranscriptions";
import { readChannelConfig } from "./channels";
@@ -141,6 +142,14 @@ export async function generateChannelSnapshot(
const archivePath = path.join(channelDir, "archive");
const playlistPath = path.join(channelDir, "playlist");
+ // Heal any video dir that drifted from the canonical id layout before we read
+ // data/* (best-effort; never fail snapshot generation on a reconcile error).
+ try {
+ await reconcileVideoDirs({ channelDir });
+ } catch {
+ /* ignore — the migration CLI can repair stragglers */
+ }
+
const [dirEntries, urls, archive, failedListed, config] = await Promise.all([
readdir(dataDir, { withFileTypes: true }).catch(() => [] as Dirent[]),
readPlaylistUrls(playlistPath),
@@ -154,7 +163,13 @@ export async function generateChannelSnapshot(
const videoDirNames = dirEntries
.filter((d) => d.isDirectory())
.map((d) => d.name);
- const wantRumbleIndex = urls.some(isRumbleUrl);
+ // The archive stores yt-dlp's native extractor ids, which equal the dir name
+ // (canonical id) only on YouTube. When the archive has any non-youtube
+ // extractor, read each video's native id from metadata so missingFromArchive
+ // compares like-for-like.
+ const wantNativeIndex = [...archive.byExtractor.keys()].some(
+ (e) => !e.toLowerCase().startsWith("youtube"),
+ );
const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY);
const perVideo = await Promise.all(
@@ -162,16 +177,16 @@ export async function generateChannelSnapshot(
limit(async () => {
const dir = path.join(dataDir, id);
const files = await readVideoFiles(dir, { checkUntranscribable: true });
- let webpageUrl: string | null = null;
- if (wantRumbleIndex && files.hasMeta) {
+ let nativeId: string | null = null;
+ if (wantNativeIndex && files.hasMeta) {
try {
const raw = await readFile(
path.join(dir, "metadata.info.json"),
"utf8",
);
const parsed = JSON.parse(raw);
- if (typeof parsed?.webpage_url === "string") {
- webpageUrl = parsed.webpage_url;
+ if (typeof parsed?.id === "string") {
+ nativeId = parsed.id;
}
} catch {
// ignore
@@ -182,7 +197,7 @@ export async function generateChannelSnapshot(
return {
id,
files,
- webpageUrl,
+ nativeId,
availability,
effectiveAvailability,
};
@@ -191,13 +206,13 @@ export async function generateChannelSnapshot(
);
const filesById = new Map<string, VideoFiles>();
- const rumbleIndex = new Map<string, string>();
+ // Native (yt-dlp extractor) ids present on disk, for the archive comparison.
+ // Dir names are canonical ids and double as native ids on YouTube.
+ const nativeIdsOnDisk = new Set<string>();
for (const v of perVideo) {
filesById.set(v.id, v.files);
- if (v.webpageUrl) {
- const slugId = extractVideoId(v.webpageUrl);
- if (slugId) rumbleIndex.set(slugId, v.id);
- }
+ nativeIdsOnDisk.add(v.id);
+ if (v.nativeId) nativeIdsOnDisk.add(v.nativeId);
}
// Computed up-front so the buckets below can suppress IDs that can't be
@@ -274,7 +289,7 @@ export async function generateChannelSnapshot(
const missingFromArchive: string[] = [];
for (const id of archive.ids) {
- if (!filesById.has(id)) missingFromArchive.push(id);
+ if (!nativeIdsOnDisk.has(id)) missingFromArchive.push(id);
}
const excludedFromDownload = emptyExcludedFromDownload();
@@ -289,12 +304,10 @@ export async function generateChannelSnapshot(
const undownloadedIds: string[] = [];
for (const url of urls) {
- const slugId = extractVideoId(url);
- if (!slugId) continue;
- const dirId =
- isRumbleUrl(url) && rumbleIndex.has(slugId)
- ? (rumbleIndex.get(slugId) as string)
- : slugId;
+ // Post-reconcile a video's dir is its canonical id, so the URL's canonical
+ // id is the dir name directly.
+ const dirId = extractVideoId(url);
+ if (!dirId) continue;
const f = filesById.get(dirId);
if (f && videoHasAnyArtifact(f)) continue;
if (excludedById.has(dirId)) continue;
diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts
@@ -0,0 +1,175 @@
+import path from "node:path";
+import { readdir, readFile, rename, rmdir, stat } from "node:fs/promises";
+import { extractVideoId } from "../ytdlp/runYtdlp";
+
+// The app keys every video by its canonical, URL-derived id
+// (`extractVideoId(webpage_url)`) and expects its data under
+// `data/<canonicalId>/`. Historically the download pipeline let yt-dlp pick the
+// directory via `%(id)s` (the extractor id), which diverges from the canonical
+// id on Twitch (`v<id>`), Rumble, and Odysee. That split a video's bytes
+// (`audio.*`, `metadata.info.json`) from its app-written sidecars
+// (`download-outcome.json`). This module renames/merges those stray dirs back
+// to the canonical layout. It is the one-time migration for existing data and a
+// defensive safety pass before the snapshot reads `data/*`.
+
+export type ReconcileResult = {
+ renamed: { from: string; to: string }[];
+ merged: { from: string; to: string; movedFiles: string[] }[];
+ skipped: {
+ dir: string;
+ reason: "no-metadata" | "no-webpage-url" | "no-canonical-id";
+ }[];
+ conflicts: { from: string; to: string; reason: string }[];
+};
+
+export type ReconcileOpts = {
+ // Absolute path to the channel root (…/channels/<slug>).
+ channelDir: string;
+ dryRun?: boolean;
+ onLog?: (s: string) => void;
+};
+
+// Files the canonical dir's copy should win on a name collision: these are
+// written by the app keyed on the canonical id and are the ones the UI links
+// to. Everything else (audio.*, transcript.*, metadata.info.json, download.log)
+// comes from yt-dlp and the source dir holds the real bytes, so the source
+// wins.
+const CANONICAL_WINS = new Set(["download-outcome.json", "availability.json"]);
+
+function sourceWins(filename: string): boolean {
+ return !CANONICAL_WINS.has(filename);
+}
+
+async function readWebpageUrl(metaPath: string): Promise<string | null> {
+ let raw: string;
+ try {
+ raw = await readFile(metaPath, "utf8");
+ } catch {
+ return null;
+ }
+ try {
+ const url = (JSON.parse(raw) as { webpage_url?: unknown }).webpage_url;
+ return typeof url === "string" && url ? url : null;
+ } catch {
+ return null;
+ }
+}
+
+async function dirExists(p: string): Promise<boolean> {
+ try {
+ return (await stat(p)).isDirectory();
+ } catch {
+ return false;
+ }
+}
+
+// Merge every file from `srcDir` into `dstDir`, returning the moved filenames.
+// Collisions never destroy data: the loser is preserved as
+// `<name>.dup-<srcName>` in the destination.
+async function mergeDir(
+ srcDir: string,
+ dstDir: string,
+ srcName: string,
+ dryRun: boolean,
+): Promise<string[]> {
+ const moved: string[] = [];
+ const [srcEntries, dstEntries] = await Promise.all([
+ readdir(srcDir).catch(() => [] as string[]),
+ readdir(dstDir).catch(() => [] as string[]),
+ ]);
+ const dstSet = new Set(dstEntries);
+ for (const f of srcEntries) {
+ const src = path.join(srcDir, f);
+ if (!dstSet.has(f)) {
+ if (!dryRun) await rename(src, path.join(dstDir, f));
+ moved.push(f);
+ continue;
+ }
+ if (sourceWins(f)) {
+ // Keep the source bytes; stash the destination's copy aside.
+ if (!dryRun) {
+ await rename(path.join(dstDir, f), path.join(dstDir, `${f}.dup-${srcName}`));
+ await rename(src, path.join(dstDir, f));
+ }
+ moved.push(f);
+ } else {
+ // Keep the destination copy; stash the source's aside.
+ if (!dryRun) await rename(src, path.join(dstDir, `${f}.dup-${srcName}`));
+ moved.push(`${f}.dup-${srcName}`);
+ }
+ }
+ return moved;
+}
+
+export async function reconcileVideoDirs(
+ opts: ReconcileOpts,
+): Promise<ReconcileResult> {
+ const { channelDir, dryRun = false } = opts;
+ const log = opts.onLog ?? (() => {});
+ const dataDir = path.join(channelDir, "data");
+ const result: ReconcileResult = {
+ renamed: [],
+ merged: [],
+ skipped: [],
+ conflicts: [],
+ };
+
+ const entries = await readdir(dataDir, { withFileTypes: true }).catch(
+ () => [],
+ );
+ const dirNames = entries.filter((e) => e.isDirectory()).map((e) => e.name);
+
+ for (const name of dirNames) {
+ try {
+ const srcDir = path.join(dataDir, name);
+ const webpageUrl = await readWebpageUrl(
+ path.join(srcDir, "metadata.info.json"),
+ );
+ if (!webpageUrl) {
+ result.skipped.push({ dir: name, reason: "no-metadata" });
+ continue;
+ }
+ const canonical = extractVideoId(webpageUrl);
+ if (!canonical) {
+ result.skipped.push({ dir: name, reason: "no-canonical-id" });
+ continue;
+ }
+ if (canonical === name) continue;
+
+ const dstDir = path.join(dataDir, canonical);
+ if (!(await dirExists(dstDir))) {
+ if (!dryRun) await rename(srcDir, dstDir);
+ result.renamed.push({ from: name, to: canonical });
+ log(`renamed ${name} -> ${canonical}\n`);
+ continue;
+ }
+
+ const movedFiles = await mergeDir(srcDir, dstDir, name, dryRun);
+ if (!dryRun) {
+ // Source should be empty now; remove it. If something raced in, leave
+ // it and flag a conflict rather than risk deleting data.
+ try {
+ await rmdir(srcDir);
+ } catch (err) {
+ result.conflicts.push({
+ from: name,
+ to: canonical,
+ reason: `source dir not empty after merge: ${(err as Error).message}`,
+ });
+ continue;
+ }
+ }
+ result.merged.push({ from: name, to: canonical, movedFiles });
+ log(`merged ${name} -> ${canonical} (${movedFiles.length} files)\n`);
+ } catch (err) {
+ result.conflicts.push({
+ from: name,
+ to: name,
+ reason: (err as Error).message,
+ });
+ log(`conflict on ${name}: ${(err as Error).message}\n`);
+ }
+ }
+
+ return result;
+}
diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts
@@ -1,11 +1,7 @@
import path from "node:path";
import { readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
-import {
- buildRumbleSlugIndex,
- extractVideoId,
- isRumbleUrl,
-} from "../ytdlp/runYtdlp";
+import { extractVideoId } from "../ytdlp/runYtdlp";
async function readPlaylistUrls(playlistPath: string): Promise<string[]> {
let raw: string;
@@ -38,16 +34,10 @@ export async function findVideoSourceUrl(
}
const urls = await readPlaylistUrls(path.join(channelRoot, "playlist"));
if (urls.length === 0) return null;
- const dataDir = path.join(channelRoot, "data");
- const rumbleIndex = urls.some(isRumbleUrl)
- ? await buildRumbleSlugIndex(dataDir)
- : null;
+ // Post-reconcile a video's dir name is its canonical id, so match the
+ // requested videoId against each URL's canonical id directly.
for (const url of urls) {
- const slugId = extractVideoId(url);
- if (!slugId) continue;
- const dirId =
- rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slugId) ?? slugId : slugId;
- if (dirId === videoId) return url;
+ if (extractVideoId(url) === videoId) return url;
}
return null;
}
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -1,5 +1,6 @@
import path from "node:path";
import { appendFile, mkdir, readdir, readFile } from "node:fs/promises";
+import { createWriteStream, type WriteStream } from "node:fs";
import { execa } from "execa";
import {
parseUnavailableFromStderr,
@@ -18,7 +19,7 @@ import {
import { writeDownloadOutcome } from "../lib/downloadOutcome-server";
import type { Paths } from "../lib/paths";
import { transcribeOneVideo } from "../controller/transcribeOne";
-import { extractVideoId, isRumbleUrl } from "./runYtdlp";
+import { extractVideoId, outputArgsForUrl } from "./runYtdlp";
import { runAudioCheckedYtdlp } from "./audioCheckedDownload";
const STDERR_TAIL_BYTES = 64 * 1024;
@@ -44,16 +45,6 @@ export type ManagedDownloadOpts = {
inlineTranscribeOnFallback?: boolean;
};
-const OUTPUT_ARGS: string[] = [
- "-o",
- "data/%(id)s/audio.%(ext)s",
- "-o",
- "subtitle:data/%(id)s/transcript",
- "-o",
- "infojson:data/%(id)s/metadata",
- "--no-write-playlist-metafiles",
-];
-
function youtubeHandlingArgs(config: ChannelConfig): string[] {
return [
"--write-auto-subs",
@@ -171,37 +162,6 @@ const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([
"private",
]);
-async function resolveVideoIdFromUrl(
- url: string,
- channelDir: string,
-): Promise<string | null> {
- // Most platforms: the URL slug IS the id used in data/<id>/. Rumble is the
- // exception — its archive id (and on-disk dir name) is yt-dlp's internal
- // RumbleEmbed id, not the URL slug. Build a minimal lookup using
- // metadata.info.json files we just wrote.
- const slug = extractVideoId(url);
- if (!slug) return null;
- if (!isRumbleUrl(url)) return slug;
- const dataDir = path.join(channelDir, "data");
- const dirs = await readdir(dataDir).catch(() => [] as string[]);
- for (const dir of dirs) {
- try {
- const raw = await readFile(
- path.join(dataDir, dir, "metadata.info.json"),
- "utf8",
- );
- const parsed = JSON.parse(raw) as { webpage_url?: unknown };
- if (typeof parsed.webpage_url === "string") {
- const dirSlug = extractVideoId(parsed.webpage_url);
- if (dirSlug === slug) return dir;
- }
- } catch {
- continue;
- }
- }
- return slug;
-}
-
async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> {
const entries = await readdir(videoDir).catch(() => [] as string[]);
return entries.some((e) => {
@@ -268,6 +228,49 @@ export async function downloadOneManaged(
const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug);
await mkdir(channelDir, { recursive: true });
+ // We pin yt-dlp's output to data/<canonicalId>/ (see outputArgsForUrl), so we
+ // know the dir before yt-dlp runs. Tee every log line into a per-video
+ // download.log next to the sidecar (overwritten per managed download, like
+ // download-outcome.json). Falls back to no-log when the URL has no canonical
+ // id (outputArgsForUrl then uses %(id)s and the reconcile pass repairs it).
+ const canonicalId = extractVideoId(opts.videoUrl);
+ let logStream: WriteStream | null = null;
+ if (canonicalId) {
+ const dir = path.join(channelDir, "data", canonicalId);
+ try {
+ await mkdir(dir, { recursive: true });
+ const stream = createWriteStream(path.join(dir, "download.log"));
+ logStream = stream;
+ const base = opts.onLog;
+ opts = {
+ ...opts,
+ onLog: (s) => {
+ try {
+ stream.write(s);
+ } catch {
+ /* logging is best-effort */
+ }
+ base(s);
+ },
+ };
+ } catch {
+ /* per-video log is best-effort; fall back to opts.onLog */
+ }
+ }
+
+ try {
+ return await runManagedDownload(opts, channelDir, startedAt, canonicalId);
+ } finally {
+ logStream?.end();
+ }
+}
+
+async function runManagedDownload(
+ opts: ManagedDownloadOpts,
+ channelDir: string,
+ startedAt: string,
+ canonicalId: string | null,
+): Promise<DownloadOutcomeRecord> {
const attempts: DownloadAttempt[] = [];
let status: DownloadOutcomeStatus = "failed";
let fellBackToTranscribe = false;
@@ -290,7 +293,7 @@ export async function downloadOneManaged(
const primaryArgs = [
"--ignore-config",
"--restrict-filenames",
- ...OUTPUT_ARGS,
+ ...outputArgsForUrl(opts.videoUrl),
...transcribeHandlingArgsForAudioCheck(opts.channelConfig),
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
@@ -298,15 +301,12 @@ export async function downloadOneManaged(
"--",
opts.videoUrl,
];
- // Hint the orchestrator at which data/<id>/ subdir this launch will
- // write to, so its .part discovery and final-file resolution don't
- // latch onto stale .parts from prior interrupted attempts. For Rumble
- // the on-disk dir is yt-dlp's internal id, only knowable after
- // metadata.info.json appears — pass null and let the orchestrator
- // fall back to the new-subdir heuristic.
- const expectedVideoIdHint = isRumbleUrl(opts.videoUrl)
- ? null
- : extractVideoId(opts.videoUrl);
+ // Hint the orchestrator at which data/<id>/ subdir this launch will write
+ // to, so its .part discovery and final-file resolution don't latch onto
+ // stale .parts from prior interrupted attempts. outputArgsForUrl pins the
+ // output to data/<canonicalId>/ for every platform, so the canonical id is
+ // the dir.
+ const expectedVideoIdHint = canonicalId;
const audioOutcome = await runAudioCheckedYtdlp({
paths: opts.paths,
channelDir,
@@ -336,7 +336,7 @@ export async function downloadOneManaged(
const primaryArgs = [
"--ignore-config",
"--restrict-filenames",
- ...OUTPUT_ARGS,
+ ...outputArgsForUrl(opts.videoUrl),
...(opts.channelConfig.handling === "youtube"
? youtubeHandlingArgs(opts.channelConfig)
: transcribeHandlingArgs(opts.channelConfig)),
@@ -388,7 +388,7 @@ export async function downloadOneManaged(
const retryArgs = [
"--ignore-config",
"--restrict-filenames",
- ...OUTPUT_ARGS,
+ ...outputArgsForUrl(opts.videoUrl),
...(opts.channelConfig.handling === "youtube"
? youtubeHandlingArgs(opts.channelConfig)
: transcribeHandlingArgs(opts.channelConfig)),
@@ -421,12 +421,9 @@ export async function downloadOneManaged(
}
// ---------- Attempt 3: no-subs fallback (youtube handling only) ----------
- const videoDir = audioCheckVideoDir
- ?? path.join(
- channelDir,
- "data",
- (await resolveVideoIdFromUrl(opts.videoUrl, channelDir)) ?? "unknown",
- );
+ const videoDir =
+ audioCheckVideoDir ??
+ path.join(channelDir, "data", canonicalId ?? "unknown");
const videoId = path.basename(videoDir);
if (
@@ -457,7 +454,7 @@ export async function downloadOneManaged(
const fallbackArgs = [
"--ignore-config",
"--restrict-filenames",
- ...OUTPUT_ARGS,
+ ...outputArgsForUrl(opts.videoUrl),
...transcribeHandlingArgs(fallbackConfig),
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -1,12 +1,5 @@
import path from "node:path";
-import {
- mkdir,
- readdir,
- readFile,
- rename,
- rm,
- writeFile,
-} from "node:fs/promises";
+import { mkdir, readdir, readFile, rename, writeFile } from "node:fs/promises";
import { execa } from "execa";
import pLimit from "p-limit";
import { readArchive } from "../lib/archive";
@@ -17,6 +10,7 @@ import {
type ChannelHandling,
} from "../lib/channelConfig";
import { getSettings } from "../lib/settings";
+import { detectPlatform } from "../lib/platform";
import type { Paths } from "../lib/paths";
import {
EXCLUDED_FROM_DOWNLOAD,
@@ -144,33 +138,6 @@ function configArgs(config: ChannelConfig): string[] {
return args;
}
-function handlingArgs(config: ChannelConfig): string[] {
- if (config.handling === "youtube") {
- return [
- "--write-auto-subs",
- "--write-subs",
- "--sub-langs",
- config.subLangs ?? "en.*,live_chat",
- "--write-info-json",
- "--skip-download",
- "-t",
- "sleep",
- ];
- }
- // transcribe: download audio only or otherwise the smallest combined format; whisper runs as a follow-up job (Phase 7).
- const fmt = config.audioFormat ?? "mp3";
- const args = [
- "--write-info-json",
- "-f",
- "bestaudio/worst",
- "-x",
- "--audio-format",
- fmt,
- ];
- if (config.keepSourceVideo) args.push("-k");
- return args;
-}
-
const OUTPUT_ARGS: string[] = [
"-o",
"data/%(id)s/audio.%(ext)s",
@@ -183,20 +150,47 @@ const OUTPUT_ARGS: string[] = [
"--no-write-playlist-metafiles",
];
-async function storePlaylist(opts: RunYtdlpOpts): Promise<void> {
- const root = channelRoot(opts);
- await mkdir(root, { recursive: true });
- const playlistPath = path.join(root, "playlist");
- const tmpPath = `${playlistPath}.tmp-${process.pid}`;
+// yt-dlp's `%(id)s` is the *extractor* id, which only matches our canonical,
+// URL-derived id on YouTube. On Twitch (`v<id>` prefix), Rumble, and Odysee it
+// diverges, splitting a video's bytes from the app's `data/<canonicalId>/`
+// dir. For single-video spawns we know the URL up front, so we pin the output
+// path literally to the canonical id and stop depending on the extractor id.
+// Falls back to %(id)s for URLs whose canonical id is missing or not a safe
+// directory name (the post-download reconcile pass cleans those up).
+export function outputArgsForUrl(url: string): string[] {
+ const id = extractVideoId(url);
+ if (id && /^[\w.-]+$/.test(id) && id !== "." && id !== "..") {
+ return [
+ "-o",
+ `data/${id}/audio.%(ext)s`,
+ "-o",
+ `subtitle:data/${id}/transcript`,
+ "-o",
+ `infojson:data/${id}/metadata`,
+ "--no-write-playlist-metafiles",
+ ];
+ }
+ return OUTPUT_ARGS;
+}
+// Enumerate a channel's video URLs via `--flat-playlist --print url` (metadata
+// only, no downloads). Pass `range` to fetch a single newest-first page via
+// `-I start:end`; sync uses this to walk the channel incrementally.
+async function enumeratePlaylistUrls(
+ opts: RunYtdlpOpts,
+ root: string,
+ range?: { start: number; end: number },
+): Promise<string[]> {
const args: string[] = [
"--flat-playlist",
"--skip-download",
"--print",
"url",
- ...configArgs(opts.channelConfig),
- opts.channelConfig.url!,
];
+ if (range) {
+ args.push("--lazy-playlist", "-I", `${range.start}:${range.end}`);
+ }
+ args.push(...configArgs(opts.channelConfig), opts.channelConfig.url!);
opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`);
const child = execa(opts.paths.ytdlpBin, args, {
@@ -210,19 +204,32 @@ async function storePlaylist(opts: RunYtdlpOpts): Promise<void> {
const result = await child;
// yt-dlp exit code convention: 101 = "break-on-existing" / "max-downloads"
- // (clean stop, not an error). Treat it the same as 0.
- if (result.exitCode !== 0 && result.exitCode !== 101) {
+ // (clean stop, not an error). Treat it the same as 0. A non-zero/101 exit
+ // while the run was cancelled is the cancel itself, not a failure.
+ if (
+ result.exitCode !== 0 &&
+ result.exitCode !== 101 &&
+ !opts.signal.aborted
+ ) {
throw new Error(`yt-dlp exited with code ${result.exitCode}`);
}
if (result.exitCode === 101) {
opts.onLog(`yt-dlp stopped on existing entry (exit 101).\n`);
}
- const stdout = String(result.stdout ?? "");
- const urls = stdout
+ return String(result.stdout ?? "")
.split("\n")
.map((s) => s.trim())
.filter(Boolean);
+}
+
+async function storePlaylist(opts: RunYtdlpOpts): Promise<void> {
+ const root = channelRoot(opts);
+ await mkdir(root, { recursive: true });
+ const playlistPath = path.join(root, "playlist");
+ const tmpPath = `${playlistPath}.tmp-${process.pid}`;
+
+ const urls = await enumeratePlaylistUrls(opts, root);
await writeFile(tmpPath, urls.join("\n") + (urls.length ? "\n" : ""));
await rename(tmpPath, playlistPath);
opts.onLog(`Wrote ${urls.length} URLs to ${playlistPath}\n`);
@@ -272,22 +279,18 @@ async function downloadPlaylistManaged(
.map((s) => s.trim())
.filter(Boolean);
- const rumbleIndex = urls.some(isRumbleUrl)
- ? await buildRumbleSlugIndex(dataDir)
- : null;
-
+ // Post-reconcile, a video's on-disk dir is always its canonical id
+ // (extractVideoId of the URL), so no slug→dir indirection is needed here.
+ // The one exception is the *archive* file, which keeps yt-dlp's native
+ // extractor ids — archiveIdForUrl bridges that.
if (idAllowList) {
const allow = new Set(idAllowList);
const before = urls.length;
const matched: string[] = [];
const matchedIds = new Set<string>();
for (const url of urls) {
- const slugId = extractVideoId(url);
- if (!slugId) continue;
- const dirId =
- rumbleIndex && isRumbleUrl(url)
- ? (rumbleIndex.get(slugId) ?? slugId)
- : slugId;
+ const dirId = extractVideoId(url);
+ if (!dirId) continue;
if (allow.has(dirId)) {
matched.push(url);
matchedIds.add(dirId);
@@ -317,7 +320,7 @@ async function downloadPlaylistManaged(
tofetch.push(...urls);
} else {
for (const url of urls) {
- const archiveId = lookupArchiveId(url, rumbleIndex);
+ const archiveId = await archiveIdForUrl(url, dataDir);
if (archiveId && archive.ids.has(archiveId)) {
skipped++;
continue;
@@ -335,18 +338,13 @@ async function downloadPlaylistManaged(
let alreadyComplete = 0;
let unidentifiable = 0;
for (const url of urls) {
- const slug = extractVideoId(url);
- if (!slug) {
+ const dirId = extractVideoId(url);
+ if (!dirId) {
unidentifiable++;
tofetch.push(url);
continue;
}
- const dirId =
- rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
- if (
- dirId &&
- (await destinationExists(dataDir, dirId, effectiveHandling))
- ) {
+ if (await destinationExists(dataDir, dirId, effectiveHandling)) {
alreadyComplete++;
continue;
}
@@ -366,12 +364,7 @@ async function downloadPlaylistManaged(
const excludedCounts = { members_only: 0, deleted: 0, private: 0 };
const filteredTofetch: string[] = [];
for (const url of tofetch) {
- const slugId = extractVideoId(url);
- const dirId = slugId
- ? rumbleIndex && isRumbleUrl(url)
- ? (rumbleIndex.get(slugId) ?? slugId)
- : slugId
- : null;
+ const dirId = extractVideoId(url);
if (dirId) {
const cls = await resolveEffectiveAvailability(
path.join(dataDir, dirId),
@@ -421,22 +414,52 @@ async function downloadPlaylistManaged(
return;
}
+ if (opts.ignoreArchive) {
+ opts.onLog(
+ "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n",
+ );
+ }
+
+ const { okCount, failedCount, processedCount, firstFailure } =
+ await runManagedDownloads(opts, items, effectiveChannelConfig);
+ if (firstFailure && !opts.signal.aborted) {
+ throw firstFailure;
+ }
+ opts.onLog(
+ `Managed download complete: ${okCount} succeeded, ${failedCount} failed (${processedCount}/${items.length} processed).\n`,
+ );
+
+ await touchLastFullDownload(opts);
+ await safeBackfillAvailability(opts);
+}
+
+type ManagedRunResult = {
+ okCount: number;
+ failedCount: number;
+ processedCount: number;
+ // Non-null only when abortOnError is on and a non-per-video failure occurred.
+ firstFailure: Error | null;
+};
+
+// Serialize a list of video URLs through downloadOneManaged. Shared by the
+// playlist-prefilter modes and `sync`. Serialized (pLimit(1)) to keep logs
+// readable and avoid hammering the source with parallel requests (which is
+// what often triggers needs_auth in the first place); the per-video startup
+// cost is the price of being able to retry each independently. Honors the
+// hard `signal`, the soft `drainSignal`, the per-channel sleep, and the
+// per-operation tracker.
+async function runManagedDownloads(
+ opts: RunYtdlpOpts,
+ urls: ReadonlyArray<string>,
+ effectiveChannelConfig: ChannelConfig,
+): Promise<ManagedRunResult> {
const settings = getSettings();
const globalCookies = settings.cookiesFromBrowser;
const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback;
const sleepSeconds =
opts.channelConfig.sleepBetweenDownloadsSeconds ??
settings.sleepBetweenDownloadsSeconds;
- if (opts.ignoreArchive) {
- opts.onLog(
- "Ignoring archive: archive entries will NOT be appended for successful downloads in this run.\n",
- );
- }
- // Serialize within a run: keeps logs readable and avoids hammering the
- // source with parallel requests (which is what often triggers needs_auth
- // in the first place). The per-video startup cost is the price of being
- // able to retry independently.
const limit = pLimit(1);
let failedCount = 0;
const abortOnError = opts.abortOnError !== false;
@@ -444,7 +467,7 @@ async function downloadPlaylistManaged(
let processedCount = 0;
await Promise.all(
- items.map((url) =>
+ urls.map((url) =>
limit(async () => {
if (opts.signal.aborted) return;
// Drain (soft-cancel): finish the in-flight download, start no more.
@@ -474,8 +497,7 @@ async function downloadPlaylistManaged(
if (outcome.status === "failed") {
failedCount++;
if (abortOnError && !firstFailure) {
- const lastAttempt =
- outcome.attempts[outcome.attempts.length - 1];
+ const lastAttempt = outcome.attempts[outcome.attempts.length - 1];
const failureClass = classifyDownloadFailure(
lastAttempt?.error ?? "",
lastAttempt?.availabilityClass,
@@ -491,7 +513,7 @@ async function downloadPlaylistManaged(
}
}
processedCount++;
- const isLast = processedCount >= items.length;
+ const isLast = processedCount >= urls.length;
const willAbortLoop = firstFailure !== null && abortOnError;
if (
sleepSeconds > 0 &&
@@ -506,16 +528,12 @@ async function downloadPlaylistManaged(
),
);
- if (firstFailure && abortOnError && !opts.signal.aborted) {
- throw firstFailure;
- }
- const okCount = processedCount - failedCount;
- opts.onLog(
- `Managed download complete: ${okCount} succeeded, ${failedCount} failed (${processedCount}/${items.length} processed).\n`,
- );
-
- await touchLastFullDownload(opts);
- await safeBackfillAvailability(opts);
+ return {
+ okCount: processedCount - failedCount,
+ failedCount,
+ processedCount,
+ firstFailure,
+ };
}
// yt-dlp's --sub-langs supports a comma-separated list with shell-glob style
@@ -560,11 +578,37 @@ function trackOnDisk(entries: string[], track: string): boolean {
return entries.some((e) => re.test(e));
}
+// Fetch only the missing subtitle tracks for one already-downloaded video.
+// Uses the literal canonical output path (outputArgsForUrl) so the .vtt files
+// land in the same data/<canonicalId>/ dir as the rest of the video.
+async function downloadSubsForUrl(
+ opts: RunYtdlpOpts,
+ root: string,
+ url: string,
+ subLangs: string,
+): Promise<void> {
+ const args: string[] = [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...outputArgsForUrl(url),
+ "--write-auto-subs",
+ "--write-subs",
+ "--sub-langs",
+ subLangs,
+ "--skip-download",
+ "--no-write-info-json",
+ "--no-overwrites",
+ ...configArgs(opts.channelConfig),
+ "--",
+ url,
+ ];
+ await runChildAndStream(opts, root, args);
+}
+
async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
const root = channelRoot(opts);
const dataDir = path.join(root, "data");
const playlistPath = path.join(root, "playlist");
- const toFetchPath = path.join(root, "playlist.tofetch-subs");
let playlistText: string;
try {
@@ -582,9 +626,6 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
const subLangs = opts.channelConfig.subLangs ?? "en.*,live_chat";
const matchesSubLang = compileSubLangMatcher(subLangs);
- const rumbleIndex = urls.some(isRumbleUrl)
- ? await buildRumbleSlugIndex(dataDir)
- : null;
const tofetch: string[] = [];
let upToDate = 0;
let unidentifiable = 0;
@@ -592,17 +633,11 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
let noExpectedTracks = 0;
for (const url of urls) {
- const slug = extractVideoId(url);
- if (!slug) {
- unidentifiable++;
- continue;
- }
- const dirId =
- rumbleIndex && isRumbleUrl(url) ? rumbleIndex.get(slug) : slug;
+ // Post-reconcile a video lives in data/<canonicalId>/, so the URL's
+ // canonical id is the dir name directly — no slug→dir indirection.
+ const dirId = extractVideoId(url);
if (!dirId) {
- // No corresponding data dir yet — skip; download-from-playlist /
- // download-missing should pull the video and its metadata first.
- noMetadata++;
+ unidentifiable++;
continue;
}
const videoDir = path.join(dataDir, dirId);
@@ -613,6 +648,8 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
"utf8",
);
} catch {
+ // No corresponding data dir yet — skip; download-from-playlist /
+ // download-missing should pull the video and its metadata first.
noMetadata++;
continue;
}
@@ -656,32 +693,48 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> {
);
if (tofetch.length === 0) {
- await rm(toFetchPath, { force: true });
opts.onLog("Nothing to fetch.\n");
return;
}
- await writeFile(toFetchPath, tofetch.join("\n") + "\n");
-
- const args: string[] = [
- "--ignore-config",
- "--restrict-filenames",
- ...OUTPUT_ARGS,
- "--write-auto-subs",
- "--write-subs",
- "--sub-langs",
- subLangs,
- "--skip-download",
- "--no-write-info-json",
- "--no-overwrites",
- ...(opts.abortOnError === false ? [] : ["--abort-on-error"]),
- "-a",
- "playlist.tofetch-subs",
- ...configArgs(opts.channelConfig),
- ];
- await runChildAndStream(opts, root, args);
+ // One yt-dlp spawn per video (serialized), mirroring the download paths.
+ const abortOnError = opts.abortOnError !== false;
+ const limit = pLimit(1);
+ let processed = 0;
+ let failed = 0;
+ let firstFailure: Error | null = null;
+ await Promise.all(
+ tofetch.map((url) =>
+ limit(async () => {
+ if (opts.signal.aborted || opts.drainSignal?.aborted) return;
+ if (firstFailure) return;
+ const task = opts.tracker?.start({
+ id: extractVideoId(url) ?? url,
+ label: extractVideoId(url) ?? url,
+ kind: "download",
+ });
+ try {
+ await downloadSubsForUrl(
+ task ? { ...opts, onLog: task.onLog } : opts,
+ root,
+ url,
+ subLangs,
+ );
+ } catch (err) {
+ failed++;
+ if (abortOnError && !firstFailure) firstFailure = err as Error;
+ } finally {
+ task?.end();
+ }
+ processed++;
+ }),
+ ),
+ );
- await rm(toFetchPath, { force: true });
+ if (firstFailure && !opts.signal.aborted) throw firstFailure;
+ opts.onLog(
+ `Sub download complete: ${processed - failed} succeeded, ${failed} failed (of ${tofetch.length}).\n`,
+ );
}
async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> {
@@ -695,7 +748,7 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> {
const args: string[] = [
"--ignore-config",
"--restrict-filenames",
- ...OUTPUT_ARGS,
+ ...outputArgsForUrl(opts.singleVideoUrl),
"--write-info-json",
"-f",
"bestaudio/worst",
@@ -737,25 +790,78 @@ export async function destinationExists(
return false;
}
+// How many newest-first playlist entries to enumerate per sync page. Each page
+// is one cheap flat-playlist call; new videos on it are fetched, and we stop at
+// the first page that contains an already-archived entry.
+const SYNC_PAGE_SIZE = 50;
+
async function sync(opts: RunYtdlpOpts): Promise<void> {
const root = channelRoot(opts);
+ const dataDir = path.join(root, "data");
+ const archivePath = path.join(root, "archive");
+ // Create the channel root (cwd for enumeration) but NOT data/ — the per-video
+ // downloads create their own canonical dirs, so a sync that's cancelled
+ // before any download leaves no data/ behind.
await mkdir(root, { recursive: true });
- const args: string[] = [
- "--ignore-config",
- "--restrict-filenames",
- ...OUTPUT_ARGS,
- ...handlingArgs(opts.channelConfig),
- "--force-write-archive",
- "--download-archive",
- "archive",
- "--break-on-existing",
- "--lazy-playlist",
- ...configArgs(opts.channelConfig),
- "--",
- opts.channelConfig.url!,
- ];
- await runChildAndStream(opts, root, args);
+ // Walk the channel newest-first a page at a time. On each page, download the
+ // entries not yet in the archive, then stop once we reach a page that
+ // contains an already-archived entry (we've caught up to a prior sync) or a
+ // short final page. Each video is fetched via downloadOneManaged — same
+ // auth-retry / audio-check / no-subs-fallback handling as every other
+ // download path — and lands directly in its canonical data/<id>/ dir. Deeper
+ // mid-channel gaps remain the job of "download missing", exactly as before.
+ let archive = await readArchive(archivePath);
+ let totalNew = 0;
+ let totalFailed = 0;
+ let firstFailure: Error | null = null;
+
+ for (
+ let page = 0;
+ !opts.signal.aborted && !opts.drainSignal?.aborted;
+ page++
+ ) {
+ const start = page * SYNC_PAGE_SIZE + 1;
+ const end = start + SYNC_PAGE_SIZE - 1;
+ const pageUrls = await enumeratePlaylistUrls(opts, root, { start, end });
+ if (pageUrls.length === 0) break;
+
+ const newUrls: string[] = [];
+ let archivedHits = 0;
+ for (const url of pageUrls) {
+ const archiveId = await archiveIdForUrl(url, dataDir);
+ if (archiveId && archive.ids.has(archiveId)) {
+ archivedHits++;
+ continue;
+ }
+ newUrls.push(url);
+ }
+ opts.onLog(
+ `Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`,
+ );
+
+ if (newUrls.length > 0) {
+ const res = await runManagedDownloads(opts, newUrls, opts.channelConfig);
+ totalNew += res.okCount;
+ totalFailed += res.failedCount;
+ if (res.firstFailure) {
+ firstFailure = res.firstFailure;
+ break;
+ }
+ // Successful downloads appended to the archive; refresh so the next
+ // page's diff sees them.
+ archive = await readArchive(archivePath);
+ }
+
+ if (archivedHits > 0) break; // reached previously-synced content
+ if (pageUrls.length < SYNC_PAGE_SIZE) break; // last page
+ }
+
+ if (firstFailure && !opts.signal.aborted) throw firstFailure;
+
+ opts.onLog(
+ `Sync complete: ${totalNew} new downloaded, ${totalFailed} failed.\n`,
+ );
await touchLastSync(opts);
await safeBackfillAvailability(opts);
}
@@ -834,58 +940,30 @@ async function updateConfigField(
await rename(tmp, configPath);
}
-export function isRumbleUrl(url: string): boolean {
+// The channel's archive file stores yt-dlp's native extractor ids (e.g.
+// `twitch:vod v123`, `rumble <internal>`), which diverge from our canonical id
+// on every platform except YouTube. A downloaded video records its native id as
+// `id` inside metadata.info.json, which (post-reconcile) lives in the canonical
+// dir. Resolve it from there; for not-yet-downloaded videos (no dir on disk) or
+// YouTube (native == canonical) fall back to the canonical id.
+export async function archiveIdForUrl(
+ url: string,
+ dataDir: string,
+): Promise<string | null> {
+ const canonical = extractVideoId(url);
+ if (!canonical) return null;
+ if (detectPlatform(url) === "youtube") return canonical;
try {
- return new URL(url).hostname.toLowerCase().endsWith("rumble.com");
+ const raw = await readFile(
+ path.join(dataDir, canonical, "metadata.info.json"),
+ "utf8",
+ );
+ const id = (JSON.parse(raw) as { id?: unknown }).id;
+ if (typeof id === "string" && id) return id;
} catch {
- return false;
- }
-}
-
-// Rumble's URL slug (the part between "rumble.com/" and the first "-") is a
-// different identifier from yt-dlp's RumbleEmbed extractor `id`, which is what
-// ends up in the archive file and as the on-disk data dir name. To prefilter
-// Rumble playlist URLs against the archive / data dir, we walk every existing
-// data/{id}/metadata.info.json once, read its `webpage_url`, and build a
-// slug → id map.
-export async function buildRumbleSlugIndex(
- dataDir: string,
-): Promise<Map<string, string>> {
- const index = new Map<string, string>();
- const dirs = await readdir(dataDir).catch(() => [] as string[]);
- for (const id of dirs) {
- let raw: string;
- try {
- raw = await readFile(
- path.join(dataDir, id, "metadata.info.json"),
- "utf8",
- );
- } catch {
- continue;
- }
- let url: unknown;
- try {
- url = JSON.parse(raw)?.webpage_url;
- } catch {
- continue;
- }
- if (typeof url !== "string") continue;
- const slug = extractVideoId(url);
- if (slug) index.set(slug, id);
- }
- return index;
-}
-
-function lookupArchiveId(
- url: string,
- rumbleIndex: Map<string, string> | null,
-): string | null {
- const slug = extractVideoId(url);
- if (!slug) return null;
- if (rumbleIndex && isRumbleUrl(url)) {
- return rumbleIndex.get(slug) ?? null;
+ /* not downloaded yet — fall through to canonical */
}
- return slug;
+ return canonical;
}
export function extractVideoId(url: string): string | null {
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -2,6 +2,8 @@
## [Unreleased]
- **Twitch.tv support.** Twitch is now a first-class platform: Twitch channel/VOD URLs are auto-detected, "Twitch" is selectable in the channel form's platform dropdown and the charts platform filter, videos play via an in-browser Twitch embed, and downloads are queued on a `platform:twitch` queue like the other platforms.
+- **Downloads always land in the canonical `data/<id>/` dir.** Previously yt-dlp chose the directory from its own extractor id, which diverges from the app's URL-derived id on Twitch (`v<id>` prefix), Rumble, and Odysee — so a video's audio/metadata and its `download-outcome.json` could end up split across two sibling dirs and the editor couldn't resolve the video. The app now pins yt-dlp's output path per video, and a reconcile pass (run automatically when a channel's report regenerates) merges any pre-existing split dirs back together. A one-time migration command, `reconcile-video-dirs`, repairs all existing channels (`--dry-run` to preview). Sync and "download missing subtitles" now fetch one video per yt-dlp run (sync walks the channel a page at a time, downloading the diff against the archive and stopping once it reaches already-synced videos).
+- **Per-video download logs.** Each managed download now writes a `download.log` alongside the video's data, capturing that run's full yt-dlp output for after-the-fact debugging.
- **Smarter queues for unrecognized sources.** When a channel's platform can't be detected, its jobs are now queued per-domain (e.g. `platform:vimeo.com`, with subdomains stripped) instead of all sharing the single `platform:unknown` queue — so unrelated unknown sources no longer block one another.
- **Charts authoring + stats dataset.** A new **Charts** tab lets you author the default chart dashboard that ships to the viewer — add/edit/remove charts and configure axes, metrics, series, filters, and search-derived series with a live preview; edits save automatically and bake into the next export build. Search-derived charts now use the full layered query builder (AND/OR/NOT, nesting, per-layer scope/regex), matching the search page. A new **Build stats dataset** action on the Build page extracts per-video stats (views, likes, comments, follower count, duration, categories, language, cue counts) into `export/public/stats/` for the charts to read; it's incremental (mtime short-circuit) and also runs automatically as part of the static export build.
- **Per-operation progress bars on `/jobs/active`.** Each running download/transcription now shows its own live progress bar parsed from the tool's shell output — yt-dlp's download percent (and fragment count) and whisper's transcribed position against the audio length. Batch jobs (Transcribe missing, Download from playlist, Sync, …) list a bar per in-flight video underneath the batch's overall bar; standalone single-video jobs get one too. The screen now polls about once a second so the bars advance live.
diff --git a/editor/e2e/availability-backfill.spec.ts b/editor/e2e/availability-backfill.spec.ts
@@ -40,13 +40,12 @@ test("download backfills availability.json from metadata", async ({ page }) => {
await seedVideo("seededPlain1", {});
await page.goto(`/channels/${CHANNEL}`);
- // Sync triggers downloadOneAudio path? Actually "Sync" runs the sync mode
- // (--lazy-playlist). The fake-ytdlp emits one fakeSync0001 video. After
- // success, runYtdlp calls backfillAvailabilityFromMetadata, which sweeps
- // ALL existing data dirs.
+ // Sync pages the flat-playlist (fake emits fake0000000{1..5}) and downloads
+ // each via the managed per-URL path. After it finishes, runYtdlp calls
+ // backfillAvailabilityFromMetadata, which sweeps ALL existing data dirs.
await page.getByRole("button", { name: "Sync" }).click();
await expect(page.getByLabel("Sync output")).toContainText("backfill", {
- timeout: 20_000,
+ timeout: 30_000,
});
// Seeded videos should now have availability.json written from metadata.
@@ -68,10 +67,10 @@ test("download backfills availability.json from metadata", async ({ page }) => {
);
expect(plain.availability).toBe("public");
- // The newly-fetched fakeSync0001 also gets a sidecar from its metadata.
+ // A newly-fetched video also gets a sidecar from its metadata.
expect(
await pathExists(
- `test-transcripts/channels/${CHANNEL}/data/fakeSync0001/availability.json`,
+ `test-transcripts/channels/${CHANNEL}/data/fake00000001/availability.json`,
),
).toBe(true);
});
diff --git a/editor/e2e/pipeline.spec.ts b/editor/e2e/pipeline.spec.ts
@@ -43,6 +43,12 @@ test("download from playlist fetches all 5 URLs and writes the archive", async (
),
).toBe(true);
}
+ // Each managed download writes a per-video log alongside the data.
+ expect(
+ await pathExists(
+ "test-transcripts/channels/test-pipeline/data/fake00000001/download.log",
+ ),
+ ).toBe(true);
const config = await readJson<{ lastFullDownloadAt?: string }>(
"test-transcripts/channels/test-pipeline/config.json",
);
@@ -72,19 +78,25 @@ test("re-running download skips already-archived entries", async ({ page }) => {
await expect(downloadLog).toContainText("Nothing to fetch");
});
-test("sync downloads one new entry and writes lastSyncedAt", async ({ page }) => {
+test("sync enumerates the channel, downloads new entries, writes lastSyncedAt", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
await resetData("test-pipeline");
await page.goto("/channels/test-pipeline");
await page.getByRole("button", { name: "Sync" }).click();
- await expect(page.getByLabel("Sync output")).toContainText(
- "fakeSync0001 done",
- { timeout: 30_000 },
- );
- expect(
- await pathExists(
- "test-transcripts/channels/test-pipeline/data/fakeSync0001/metadata.info.json",
- ),
- ).toBe(true);
+ // Sync now pages the flat-playlist (fake emits 5 entries) and downloads the
+ // archive diff per page via the managed per-URL path.
+ await expect(page.getByLabel("Sync output")).toContainText("Sync complete", {
+ timeout: 60_000,
+ });
+ for (let i = 1; i <= 5; i++) {
+ expect(
+ await pathExists(
+ `test-transcripts/channels/test-pipeline/data/fake0000000${i}/metadata.info.json`,
+ ),
+ ).toBe(true);
+ }
const config = await readJson<{ lastSyncedAt?: string }>(
"test-transcripts/channels/test-pipeline/config.json",
);
diff --git a/editor/e2e/reconcile.spec.ts b/editor/e2e/reconcile.spec.ts
@@ -0,0 +1,84 @@
+import { mkdir, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { pathExists, resetData, resolvePath } from "./helpers";
+
+// yt-dlp names Twitch VOD dirs `v<id>` (its extractor id) while the app keys
+// videos by the canonical URL id `<id>`, so the audio/metadata land in
+// data/v<id>/ and the app-written download-outcome.json lands in data/<id>/.
+// generateChannelSnapshot (run on first channel-page load when no snapshot.json
+// exists) now reconciles those dirs back to the canonical layout.
+
+const CHANNEL = "twitch-reconcile";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+
+async function write(rel: string, contents: string): Promise<void> {
+ const full = resolvePath(`${ROOT}/${rel}`);
+ await mkdir(full.slice(0, full.lastIndexOf("/")), { recursive: true });
+ await writeFile(full, contents);
+}
+
+async function seedChannel(): Promise<void> {
+ await write(
+ "config.json",
+ JSON.stringify({
+ handling: "transcribe",
+ name: "Twitch Reconcile",
+ url: "https://www.twitch.tv/reconcile/videos",
+ platform: "twitch",
+ }),
+ );
+
+ // Case 1 — merge: bytes in v123, sidecar already in canonical 123.
+ await write(
+ "data/v123/metadata.info.json",
+ JSON.stringify({ id: "v123", webpage_url: "https://www.twitch.tv/videos/123" }),
+ );
+ await write("data/v123/audio.mp3", "fake audio 123");
+ await write(
+ "data/123/download-outcome.json",
+ JSON.stringify({ videoId: "123", status: "ok" }),
+ );
+
+ // Case 2 — rename: bytes in v456, no canonical dir yet.
+ await write(
+ "data/v456/metadata.info.json",
+ JSON.stringify({ id: "v456", webpage_url: "https://www.twitch.tv/videos/456" }),
+ );
+ await write("data/v456/audio.mp3", "fake audio 456");
+
+ // Case 3 — orphan with no metadata: can't determine canonical id, left alone.
+ await write(
+ "data/orphan999/download-outcome.json",
+ JSON.stringify({ videoId: "orphan999", status: "failed" }),
+ );
+}
+
+test("channel snapshot reconciles Twitch v<id> dirs into canonical <id> dirs", async ({
+ page,
+}) => {
+ await resetData(null);
+ await seedChannel();
+
+ // Loading the channel page generates the snapshot (no snapshot.json yet),
+ // which runs reconcileVideoDirs first.
+ await page.goto(`/channels/${CHANNEL}`);
+ await expect(
+ page.getByRole("heading", { name: "Twitch Reconcile" }),
+ ).toBeVisible({ timeout: 15_000 });
+
+ // Case 1: v123 merged into 123 — all three files now in 123, v123 gone.
+ expect(await pathExists(`${ROOT}/data/123/audio.mp3`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/123/metadata.info.json`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/123/download-outcome.json`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/v123`)).toBe(false);
+
+ // Case 2: v456 renamed to 456.
+ expect(await pathExists(`${ROOT}/data/456/audio.mp3`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/456/metadata.info.json`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/v456`)).toBe(false);
+
+ // Case 3: metadata-less orphan left untouched.
+ expect(
+ await pathExists(`${ROOT}/data/orphan999/download-outcome.json`),
+ ).toBe(true);
+});