#!/usr/bin/env node // Point a song at its source recordings in the archive's saved-video store: // media/.mp4 and wav48/.wav, for every video the plans use. // // render-poly reads both by absolute source seconds: wav48/.wav whole into // memory (a missing one aborts the render) and media/.mp4 per clip (a // missing or short one silently drops or slides that voice's layer). So these // are FULL recordings, and they are fetched by the editor, not here: // // pnpm ops persist-videos --file <{"items": [{"slug","id"}, ...], "format": "video_720"}> // // one paced job on the channel's download queue, with its cookies and backoff. // This step only reads what that job saved, through each video's own // saved-video.json pointer -- the same way the editor finds it, so a store that // is moved to another drive later is followed, not copied. // // media/.mp4 a SYMLINK to the saved file (no second copy; named .mp4 // whatever the container, because render-poly asks for that // name and ffmpeg reads the content). An older real file // there is replaced. // wav48/.wav 48 kHz mono s16 PCM, written once from the saved file. // // node link-sources.mjs ... [--channel the-quartering] // // CHANNELS_DIR the corpus's channels/ (default: /transcripts/channels) // // Re-run as the persist job progresses: an id not saved yet is skipped and // counted, and one already linked is left alone. sources.json in SONG_DATA // records each one. import { existsSync, lstatSync, mkdirSync, readFileSync, readlinkSync, renameSync, statSync, symlinkSync, unlinkSync, writeFileSync } from "node:fs"; import { execFileSync } from "node:child_process"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const argv = process.argv.slice(2); const ci = argv.indexOf("--channel"); const CHANNEL = ci < 0 ? "the-quartering" : argv[ci + 1]; const plans = argv.filter((a, i) => a.endsWith(".json") && !(ci >= 0 && i === ci + 1)); const DIR = path.dirname(new URL(import.meta.url).pathname); const CHANNELS = path.resolve(process.env.CHANNELS_DIR ?? path.join(DIR, "..", "..", "transcripts", "channels")); if (!plans.length) { console.error("usage: link-sources.mjs ... [--channel ]"); process.exit(2); } if (!existsSync(path.join(CHANNELS, CHANNEL))) { console.error(`no channel ${CHANNEL} under ${CHANNELS} -- set CHANNELS_DIR to the corpus's channels/`); process.exit(2); } const ids = [...new Set(plans.flatMap((p) => JSON.parse(readFileSync(p, "utf8")).voices.flatMap((v) => v.plan.map((n) => n.video)).filter(Boolean)))].sort(); const MEDIA = path.join(SONG_DATA, "media"), WAV = path.join(SONG_DATA, "wav48"); mkdirSync(MEDIA, { recursive: true }); mkdirSync(WAV, { recursive: true }); const LEDGER = path.join(SONG_DATA, "sources.json"); const ledger = existsSync(LEDGER) ? JSON.parse(readFileSync(LEDGER, "utf8")) : {}; const savedFile = (id) => { try { const p = JSON.parse(readFileSync(path.join(CHANNELS, CHANNEL, "data", id, "saved-video.json"), "utf8")); const f = path.join(p.dir, p.file); return existsSync(f) ? f : null; } catch { return null; } }; const lexists = (p) => { try { lstatSync(p); return true; } catch { return false; } }; let linked = 0, wavs = 0, waiting = 0, already = 0; for (const id of ids) { const file = savedFile(id); if (!file) { waiting += 1; continue; } const media = path.join(MEDIA, `${id}.mp4`); const isLink = lexists(media) && lstatSync(media).isSymbolicLink(); let changed = false; if (!isLink || path.resolve(MEDIA, readlinkSync(media)) !== file) { if (lexists(media)) unlinkSync(media); symlinkSync(file, media); linked += 1; changed = true; } const wav = path.join(WAV, `${id}.wav`); if (!existsSync(wav)) { execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-max_error_rate", "1.0", "-i", file, "-vn", "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", `${wav}.part.wav`]); renameSync(`${wav}.part.wav`, wav); wavs += 1; changed = true; } if (!changed) { already += 1; continue; } const secs = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", wav]).toString().trim()); ledger[id] = { channel: CHANNEL, saved: file, bytes: statSync(file).size, seconds: +secs.toFixed(2), linkedAt: new Date().toISOString() }; writeFileSync(`${LEDGER}.tmp`, JSON.stringify(ledger, null, 1) + "\n"); renameSync(`${LEDGER}.tmp`, LEDGER); console.log(`${id}: ${(secs / 60).toFixed(1)} min`); } console.log(`${ids.length} sources: ${linked} linked, ${wavs} wav48 written, ${already} already done, ${waiting} not saved yet`);