import path from "node:path"; import { readdir, readFile, rename, rmdir, stat } from "node:fs/promises"; import { extractVideoId } from "../ytdlp/runYtdlp"; import { CLIPS_DIR_NAME } from "../lib/clipWindow"; // The app keys every video by its canonical, URL-derived id // (`extractVideoId(webpage_url)`) and expects its data under // `data//`. Historically the download pipeline let yt-dlp pick the // directory via `%(id)s` (the extractor id), which diverges from the canonical // id on Twitch (`v`), Rumble, and Odysee. That split a video's bytes // (`audio.*`, `metadata.info.json`) from its app-written sidecars // (`download-outcome.json`). This module renames/merges those stray dirs back // to the canonical layout. It is the one-time migration for existing data and a // defensive safety pass before the snapshot reads `data/*`. export type ReconcileResult = { renamed: { from: string; to: string }[]; merged: { from: string; to: string; movedFiles: string[] }[]; skipped: { dir: string; reason: "no-metadata" | "no-webpage-url" | "no-canonical-id"; }[]; conflicts: { from: string; to: string; reason: string }[]; }; export type ReconcileOpts = { // Absolute path to the channel root (…/channels/). channelDir: string; dryRun?: boolean; onLog?: (s: string) => void; // Reconcile only the dirs this accepts (default: every dir). A caller that // knows which records it means to move (`archilyzer wayback refresh`, // controller/waybackRefresh.ts) leaves the rest to the snapshot's pass. only?: (dirName: string) => boolean; }; // Files the canonical dir's copy should win on a name collision: these are // written by the app keyed on the canonical id and are the ones the UI links // to. Everything else (audio.*, transcript.*, metadata.info.json, download.log) // comes from yt-dlp and the source dir holds the real bytes, so the source // wins. const CANONICAL_WINS = new Set(["download-outcome.json", "availability.json"]); function sourceWins(filename: string): boolean { return !CANONICAL_WINS.has(filename); } async function readWebpageUrl(metaPath: string): Promise { let raw: string; try { raw = await readFile(metaPath, "utf8"); } catch { return null; } try { const url = (JSON.parse(raw) as { webpage_url?: unknown }).webpage_url; return typeof url === "string" && url ? url : null; } catch { return null; } } async function dirExists(p: string): Promise { try { return (await stat(p)).isDirectory(); } catch { return false; } } // Merge every file from `srcDir` into `dstDir`, returning the moved filenames. // Collisions never destroy data: the loser is preserved as // `.dup-` in the destination. async function mergeDir( srcDir: string, dstDir: string, srcName: string, dryRun: boolean, ): Promise { const moved: string[] = []; const [srcEntries, dstEntries] = await Promise.all([ readdir(srcDir).catch(() => [] as string[]), readdir(dstDir).catch(() => [] as string[]), ]); const dstSet = new Set(dstEntries); for (const f of srcEntries) { const src = path.join(srcDir, f); // `clips/` IS A DIRECTORY, and the collision rule below is written for // files: stashing it as `clips.dup-` would hide every window it // holds from listClipWindows, which only reads `clips/`. Two clip stores // for one video are two halves of one cache, so they MERGE — and because a // window's name IS its span, a file present in both is the same bytes. if (f === CLIPS_DIR_NAME && dstSet.has(f)) { const inner = await mergeDir(src, path.join(dstDir, f), srcName, dryRun); moved.push(...inner.map((n) => path.join(f, n))); // rmdir, NOT rm: `src` is a DIRECTORY, and rm() without `recursive` // throws EISDIR on one. The catch swallowed it, the emptied source // `clips/` survived, and the rmdir(srcDir) below then failed ENOTEMPTY — // so every clips-on-both-sides reconcile reported a false "source dir not // empty" conflict and left the whole video dir unmerged. Still tolerant: // a stray file the recursion could not move is a reason to leave the dir // for the conflict path, not to delete it. if (!dryRun) await rmdir(src).catch(() => {}); continue; } if (!dstSet.has(f)) { if (!dryRun) await rename(src, path.join(dstDir, f)); moved.push(f); continue; } if (sourceWins(f)) { // Keep the source bytes; stash the destination's copy aside. if (!dryRun) { await rename(path.join(dstDir, f), path.join(dstDir, `${f}.dup-${srcName}`)); await rename(src, path.join(dstDir, f)); } moved.push(f); } else { // Keep the destination copy; stash the source's aside. if (!dryRun) await rename(src, path.join(dstDir, `${f}.dup-${srcName}`)); moved.push(`${f}.dup-${srcName}`); } } return moved; } export async function reconcileVideoDirs( opts: ReconcileOpts, ): Promise { const { channelDir, dryRun = false } = opts; const log = opts.onLog ?? (() => {}); const dataDir = path.join(channelDir, "data"); const result: ReconcileResult = { renamed: [], merged: [], skipped: [], conflicts: [], }; const entries = await readdir(dataDir, { withFileTypes: true }).catch( () => [], ); const dirNames = entries .filter((e) => e.isDirectory()) .map((e) => e.name) .filter((name) => !opts.only || opts.only(name)); for (const name of dirNames) { try { const srcDir = path.join(dataDir, name); const webpageUrl = await readWebpageUrl( path.join(srcDir, "metadata.info.json"), ); if (!webpageUrl) { result.skipped.push({ dir: name, reason: "no-metadata" }); continue; } const canonical = extractVideoId(webpageUrl); if (!canonical) { result.skipped.push({ dir: name, reason: "no-canonical-id" }); continue; } if (canonical === name) continue; const dstDir = path.join(dataDir, canonical); if (!(await dirExists(dstDir))) { if (!dryRun) await rename(srcDir, dstDir); result.renamed.push({ from: name, to: canonical }); log(`renamed ${name} -> ${canonical}\n`); continue; } const movedFiles = await mergeDir(srcDir, dstDir, name, dryRun); if (!dryRun) { // Source should be empty now; remove it. If something raced in, leave // it and flag a conflict rather than risk deleting data. try { await rmdir(srcDir); } catch (err) { result.conflicts.push({ from: name, to: canonical, reason: `source dir not empty after merge: ${(err as Error).message}`, }); continue; } } result.merged.push({ from: name, to: canonical, movedFiles }); log(`merged ${name} -> ${canonical} (${movedFiles.length} files)\n`); } catch (err) { result.conflicts.push({ from: name, to: name, reason: (err as Error).message, }); log(`conflict on ${name}: ${(err as Error).message}\n`); } } return result; }