#!/usr/bin/env node // Render a multi-voice plan: mix every voice into one master, then build the // picture as the melody full-frame with the other voices as picture-in-picture, // each cutting on its OWN note onsets. import { readFileSync, writeFileSync, mkdirSync, existsSync, rmSync } from "node:fs"; import { execFile, execFileSync } from "node:child_process"; import path from "node:path"; import { regionPitch, midiOf, yinFrame } from "./pitch.mjs"; import { clipWindow } from "./clipwindow.mjs"; // parakeet word timings, used to keep a neighbouring word out of a note const asrCache = new Map(); const wordsFor = (v) => { if (!asrCache.has(v)) { let w = []; try { w = JSON.parse(readFileSync(path.join(SONG_DATA, "asr", `${v}.json`), "utf8")).words ?? []; } catch {} w.sort((a, b) => a.start - b.start); asrCache.set(v, w); } return asrCache.get(v); }; import { SONG_DATA } from "./paths.mjs"; const DIR = path.resolve(path.dirname(new URL(import.meta.url).pathname)); const OUT = process.argv[2] ?? path.join(SONG_DATA, "poly.mp4"); const SR = 48000, PAD = 0.025, FADE_IN = 0.008, FADE_OUT = 0.030, TARGET_RMS = 0.12; const W = 1280, H = 720, FPS = 30, JOBS = 6; const LAYOUT = process.env.LAYOUT ?? "pip"; // "pip" | "grid" | "bg" // Over a background the picture MUST pulse: the layer is composited by keying // out black, so a legato voice that fills its slot leaves no black and the // background never shows through it. bg implies pulse. const PULSE = process.env.PULSE === "1" || (process.env.LAYOUT ?? "pip") === "bg"; const DEBUG = process.env.DEBUG === "1"; const STRETCH_MAX = 2.8; // A tune that ends by LOOPING back to its opening does not end -- it stops dead // on the loop point. TAIL_FADE fades the last stretch of AUDIO; VIDEO_TAIL_FADE // fades the PICTURE over the same stretch, and defaults to it, so the two move // together wherever a tail fade exists at all. Where none does the default is 0 // and nothing changes -- Mario RPG ends on the background's own wipe to black // and must not be faded a second time. const TAIL_FADE = Number(process.env.TAIL_FADE ?? 0); const VIDEO_TAIL_FADE = Number(process.env.VIDEO_TAIL_FADE ?? TAIL_FADE); // DISSOLVE, off by default. It is deliberately NOT driven from the audio // envelope: that is FADE_IN 0.008 / FADE_OUT 0.030, both sub-frame at 30fps, so // a dissolve matched to it would be invisible. This is its own duration. // // Fading EVERY clip was judged wrong to look at, and it is: between two notes // that follow each other the picture should CUT, the way the sound does. A // dissolve only earns its place where the voice then stops for a while, so a // clip fades out only when at least CLIP_FADE_GAP of rest follows it, and it // fades INTO that rest -- the note is fully lit for its whole sounded length and // dissolves away afterwards, rather than dimming while it is still sounding. // There is no fade IN, deliberately: a note should arrive on its attack. const CLIP_FADE = Number(process.env.CLIP_FADE ?? 0); const CLIP_FADE_GAP = Number(process.env.CLIP_FADE_GAP ?? 0.30); const COMPRESS_MIN = Number(process.env.COMPRESS_MIN ?? 0.85); const KEEP_COMPRESS_MIN = Number(process.env.KEEP_COMPRESS_MIN ?? 0.72); // how much of the note a resampled clip must still fill to be worth it const RESAMPLE_MIN_FILL = Number(process.env.RESAMPLE_MIN_FILL ?? 0.80); function readWav(file) { const b = readFileSync(file); let p = 12, fmt = null, data = null; while (p + 8 <= b.length) { const id = b.toString("latin1", p, p + 4); const sz = b.readUInt32LE(p + 4); const body = p + 8; if (id === "fmt ") fmt = { ch: b.readUInt16LE(body + 2), sr: b.readUInt32LE(body + 4) }; if (id === "data") { data = b.subarray(body, body + sz); break; } p = body + sz + (sz & 1); } const n = Math.floor(data.length / 2 / fmt.ch); const x = new Float32Array(n); for (let i = 0; i < n; i += 1) { let s = 0; for (let c = 0; c < fmt.ch; c += 1) s += data.readInt16LE((i * fmt.ch + c) * 2) / 32768; x[i] = s / fmt.ch; } return { x, sr: fmt.sr }; } // Plain speed change: playing a clip f times faster raises it by 12*log2(f) // semitones AND shortens it by 1/f. When a note needs BOTH -- a higher pitch and // a shorter clip -- one resample delivers them together with no phase vocoder in // the path, so none of rubberband's smearing. Cubic interpolation, because // linear audibly dulls the top end at these rates. function resample(x, f) { const n = Math.max(1, Math.floor(x.length / f)); const y = new Float32Array(n); const at = (i) => (i < 0 ? x[0] : i >= x.length ? x[x.length - 1] : x[i]); for (let i = 0; i < n; i += 1) { const t = i * f, k = Math.floor(t), d = t - k; const a = at(k - 1), b = at(k), c = at(k + 1), e = at(k + 2); // Catmull-Rom y[i] = b + 0.5 * d * (c - a + d * (2 * a - 5 * b + 4 * c - e + d * (3 * (b - c) + e - a))); } return y; } function writeWav(file, x, sr) { const n = x.length, b = Buffer.alloc(44 + n * 2); b.write("RIFF", 0, "latin1"); b.writeUInt32LE(36 + n * 2, 4); b.write("WAVE", 8, "latin1"); b.write("fmt ", 12, "latin1"); b.writeUInt32LE(16, 16); b.writeUInt16LE(1, 20); b.writeUInt16LE(1, 22); b.writeUInt32LE(sr, 24); b.writeUInt32LE(sr * 2, 28); b.writeUInt16LE(2, 32); b.writeUInt16LE(16, 34); b.write("data", 36, "latin1"); b.writeUInt32LE(n * 2, 40); for (let i = 0; i < n; i += 1) b.writeInt16LE(Math.round(Math.max(-1, Math.min(1, x[i])) * 32767), 44 + i * 2); writeFileSync(file, b); } const SKIP_AUDIO = process.env.SKIP_AUDIO === "1"; // PLAN= names the plan explicitly. Defaulting to poly-plan.json is how the // Metal Slug arrangement ended up under the Mortal Kombat background: the render // took whatever the LAST arrange happened to leave in the shared scratch file. // Say which tune you are rendering and the mix-up cannot happen. const PLAN_FILE = process.env.PLAN ?? "poly-plan.json"; const PLAN_RAW = readFileSync(path.join(SONG_DATA, PLAN_FILE), "utf8"); const { voices } = JSON.parse(PLAN_RAW); // Print it, so a wrong plan is visible in the log rather than only in the audio. console.log(`plan ${PLAN_FILE}: ${voices.map((v) => `${v.name} ${v.plan.length}`).join(", ")} notes, ` + `${Math.min(...voices.flatMap((v) => v.plan.map((p) => p.slotStart))).toFixed(2)}s -> ` + `${Math.max(...voices.flatMap((v) => v.plan.map((p) => p.slotStart + p.slotDur))).toFixed(2)}s`); // Keep the plan NEXT TO the output. poly-plan.json is shared scratch, so the // next arrange overwrites it -- which meant a timestamped report about a // finished video ("static at 19s") could no longer be traced to a clip. writeFileSync(`${OUT.replace(/\.mp4$/, "")}.plan.json`, PLAN_RAW); // ...and a readable one. "Which clip is playing at 41s?" should not require // parsing JSON, and the answer needs to survive long after the render. { const rows = [["songTime", "voice", "clipId", "video", "srcStart", "srcEnd", "noteDur", "shift", "keepSide"]]; for (const v of voices) { for (const p of v.plan) { rows.push([p.slotStart.toFixed(3), v.name, `${p.video}@${(+p.srcStart).toFixed(2)}`, p.video, (+p.srcStart).toFixed(3), (+p.srcEnd).toFixed(3), (p.noteDur ?? 0).toFixed(3), (p.shift ?? 0).toFixed(2), String(p.keepSide ?? 0)]); } } rows.slice(1).sort((a, b) => Number(a[0]) - Number(b[0])); writeFileSync(`${OUT.replace(/\.mp4$/, "")}.clips.csv`, rows.map((r) => r.join(",")).join("\n")); } // Per-run scratch dir. This used to be a single shared "polytmp" that each run // wiped on startup, so two concurrent renders deleted each other's clips // mid-build (one died on a missing list.txt, another produced a file with no // moov atom). const tmp = path.join(SONG_DATA, `polytmp-${path.basename(OUT).replace(/\W+/g, "_")}-${process.pid}`); if (existsSync(tmp)) rmSync(tmp, { recursive: true, force: true }); mkdirSync(tmp, { recursive: true }); const srcCache = new Map(); const loadSrc = (vid) => { if (srcCache.has(vid)) return srcCache.get(vid); if (srcCache.size > 2) srcCache.clear(); const d = readWav(path.join(SONG_DATA, "wav48", `${vid}.wav`)); srcCache.set(vid, d); return d; }; // ---------- audio ------------------------------------------------------------ const endOf = (v) => v.plan.length ? Math.max(...v.plan.map((p) => p.slotStart + p.slotDur)) : 0; const TOTAL = Math.max(...voices.map(endOf)); // ---------- overlays --------------------------------------------------------- // An OVERLAY is a sound mixed OVER the arrangement rather than into it: it takes // no note slot and consumes no clip. The intro effect was the first of these, so // it is now one entry in the same list instead of a parallel code path -- a // second parallel path is how the clip-window logic drifted (see clipwindow.mjs). // // INTRO_SFX= one overlay placed to END on the first melody note // OVERLAYS= [{file, at, gain?}]; `at` is seconds, or "intro" const overlays = []; const SFX = process.env.INTRO_SFX; const SFX_VOICE = process.env.INTRO_SFX_VOICE ?? "bass"; const skipFirst = new Set(); if (SFX) { // the intro effect REPLACES the first note of its voice const v0 = voices.find((v) => v.name === SFX_VOICE); if (v0 && v0.plan.length) skipFirst.add(v0.plan[0]); overlays.push({ file: SFX, at: "intro", gain: 0.72 }); } if (process.env.OVERLAYS) { const list = JSON.parse(readFileSync(path.join(SONG_DATA, process.env.OVERLAYS), "utf8")); overlays.push(...(Array.isArray(list) ? list : list.overlays ?? [])); } // Load them BEFORE sizing the master buffer: an overlay near the end is longer // than the tail allowance, and would otherwise be silently truncated. const melody0 = (voices.find((v) => v.role === "full") ?? voices[0]).plan[0]?.slotStart ?? 0; for (const o of overlays) { try { o.buf = readWav(path.join(SONG_DATA, o.file)); } catch (e) { o.err = e.message; continue; } if (o.buf.sr !== SR) console.log(` WARNING overlay ${o.file} is ${o.buf.sr}Hz, not ${SR}Hz -- it will play at the wrong speed`); // "intro" ENDS on the first melody note, so it leads into the theme. o.at = o.at === "intro" ? Math.max(0, melody0 - o.buf.x.length / SR) : Number(o.at); } const overlayEnd = Math.max(0, ...overlays.filter((o) => o.buf).map((o) => o.at + o.buf.x.length / SR)); const master = new Float32Array(Math.ceil((Math.max(TOTAL, overlayEnd) + 1.0) * SR)); // TOTAL is the end of the last SLOT, and the buffer adds a second on top. A // voice whose final note is short leaves that slack as dead air -- one full // second of silent video at the end of the Yoshi render. END is recomputed from // the actual audio once the mix exists (see below). let END = TOTAL; for (const v of (SKIP_AUDIO ? [] : voices)) { let n = 0; for (const p of v.plan) { if (skipFirst.has(p)) continue; const { x } = loadSrc(p.video); // EXACTLY the window the review page showed -- same function, same words. // Two implementations had drifted: the page cleared neighbouring words and // the renderer did not, so notes sounded words nobody approved (one render // of Yoshi's Island ended on a bare "the"). let lo, hi; // A hand-set window is EXACT -- no padding. Padding it by 25ms each side // played 796 clips 14% longer than they were trimmed to, and dragged a word // back into 21 of them ("don't", "help", "Save") that the trim had removed. // The fades shape these edges; they do not need slack to work. if (p.noTrim) { lo = p.srcStart; hi = p.srcEnd; } else { const w = clipWindow(x, SR, { start: p.srcStart, end: p.srcEnd }, wordsFor(p.video)); lo = w.s; hi = w.e; } const a = Math.max(0, Math.round(lo * SR)); const b = Math.min(x.length, Math.round(hi * SR)); if (b - a < 0.05 * SR) continue; let raw = Float32Array.prototype.slice.call(x, a, b); // The melody fills its slot so the tune stays legato. Accompaniment voices // play their NATURAL length and stop, leaving a real gap -- audible and // visible, so the picture-in-picture can go black when that voice is not // sounding. Stretching every note to fill left nothing to blank (only 2.2% // of PiP frames were black). const fill = v.fill !== false; const maxLen = Math.round((fill ? p.slotDur * 0.97 : Math.min(p.slotDur * 0.97, p.noteDur * 1.15)) * SR); // A clip a little longer than its note can be TIME-COMPRESSED instead of // chopped, which keeps the whole um -- tail included -- and just plays it // slightly faster. Only mildly: past ~15% it starts to sound hurried, and // chopping the tail is the better trade. const ratio = maxLen / Math.max(1, raw.length); let stretch = fill ? Math.max(1, Math.min(STRETCH_MAX, ratio)) : 1; // A KEEP clip ("and um") is a compound sound -- chopping it removes the very // thing it was kept for -- so it may be squeezed harder before being cut. const floor = p.keepSide ? KEEP_COMPRESS_MIN : COMPRESS_MIN; if (ratio < 1 && ratio >= floor) stretch = ratio; const keep = Math.min(raw.length, maxLen); const head = regionPitch(raw, Math.round(0.01 * SR), Math.max(Math.round(0.02 * SR), keep - 1), SR, 55, 260); let shift = p.shift; if (head.f0 > 0) { const s2 = p.targetMidi - midiOf(head.f0); if (Math.abs(s2) <= 14) shift = s2; } const inF = path.join(tmp, `i.wav`), outF = path.join(tmp, `o.wav`); const shiftBy = (amt) => { writeWav(inF, raw, SR); const args = ["-3", "-p", amt.toFixed(3), "-F", "-c", "5"]; if (stretch > 1.02 || stretch < 0.98) args.push("-t", stretch.toFixed(3)); execFileSync("rubberband", [...args, inF, outF], { stdio: "ignore" }); return readWav(outF).x; }; let seg = raw; // Preferred path: if a straight speed change lands BOTH the pitch and a // length that fits the note, take it -- it is the only transformation here // that adds no artifacts at all. const speed = Math.pow(2, shift / 12); const resLen = raw.length / speed; const canResample = Math.abs(shift) > 0.15 && Math.abs(shift) <= 6 && resLen <= maxLen * 1.01 && resLen >= maxLen * RESAMPLE_MIN_FILL; if (canResample) { try { seg = resample(raw, speed); } catch { seg = raw; } } else if (Math.abs(shift) > 0.03 || stretch > 1.02 || stretch < 0.98) { try { seg = shiftBy(shift); } catch { seg = raw; } } try { const k2 = Math.min(seg.length, maxLen); const got = regionPitch(seg, Math.round(0.01 * SR), Math.max(Math.round(0.02 * SR), k2 - 1), SR, 55, 500); if (got.f0 > 0) { let r = p.targetMidi - midiOf(got.f0); while (r > 6) r -= 12; while (r < -6) r += 12; if (Math.abs(r) > 0.08 && Math.abs(r) <= 4) seg = shiftBy(shift + r); } } catch { /* keep first pass */ } if (seg.length > maxLen) { // Hand-set windows are trimmed to the MAX usable um, so when a note is // shorter than the clip there is slack at both ends. Taking it only off // the tail kept every edge artifact at the front; shave a little off the // front too and land on the steadier middle. Never shave the edge that // holds a deliberately-kept word. // Chop the TAIL, never the attack -- the onset is what makes it read as a // note. The one exception is a clip kept for a trailing word, where the // end is the point and the slack has to come off the front instead. const off = p.keepSide === 2 ? (seg.length - maxLen) : 0; seg = seg.slice(off, off + maxLen); } p.soundedDur = +(seg.length / SR).toFixed(4); // The picture may pulse shorter than the sound: a legato melody would never // blank otherwise, and a visualiser where one panel never moves reads as // idle video rather than as a voice. p.videoDur = +Math.min(p.soundedDur, Math.max(p.noteDur * 1.1, 0.10)).toFixed(4); let s = 0; for (let i = 0; i < seg.length; i += 1) s += seg[i] * seg[i]; const rms = Math.sqrt(s / Math.max(1, seg.length)); let g = rms > 1e-6 ? (TARGET_RMS * v.gain) / rms : 0; let peak = 0; for (let i = 0; i < seg.length; i += 1) peak = Math.max(peak, Math.abs(seg[i])); if (peak * g > 0.95) g = 0.95 / peak; const fi = Math.min(Math.round(FADE_IN * SR), seg.length >> 1); const fo = Math.min(Math.round(FADE_OUT * SR), seg.length >> 1); // A leading-word clip starts EARLY by the length of its word, so the um // itself lands on the beat and the word reads as a pickup into it. const lead = (p.keepSide === 1 && p.headOffset) ? p.headOffset : 0; const off = Math.round(Math.max(0, p.slotStart - lead) * SR); for (let i = 0; i < seg.length; i += 1) { let env = 1; if (i < fi) env = i / fi; if (i >= seg.length - fo) env = Math.min(env, (seg.length - i) / fo); const j = off + i; if (j < master.length) master[j] += seg[i] * g * env; } n += 1; } console.log(` voice ${v.name}: ${n}/${v.plan.length} notes mixed`); } // Overlays go in BEFORE the normalise-and-trim stage below, so the master is // levelled with them included -- adding them after would push the mix over full // scale and clip exactly on the loudest hooks. for (const o of (SKIP_AUDIO ? [] : overlays)) { if (!o.buf) { console.log(` overlay ${o.file} failed: ${o.err}`); continue; } const off = Math.round(o.at * SR); let pk = 0; for (let i = 0; i < o.buf.x.length; i += 1) pk = Math.max(pk, Math.abs(o.buf.x[i])); // Level each overlay to a PEAK target rather than an RMS one: these are shouts, // not sustained notes, and RMS-matching a short stab makes it far too loud. const want = o.gain ?? 0.72; const g = pk > 1e-4 ? want / pk : 1; let clipped = 0; for (let i = 0; i < o.buf.x.length; i += 1) { const j = off + i; if (j >= 0 && j < master.length) master[j] += o.buf.x[i] * g; else clipped += 1; } console.log(` overlay ${o.file} at ${o.at.toFixed(2)}s ${(o.buf.x.length / SR).toFixed(2)}s peak->${want}` + (o.label ? ` "${o.label}"` : "") + (clipped ? ` (${clipped} samples past the end)` : "")); } // Named for the OUTPUT, not shared. A single poly-song.wav is the same hazard as // a single poly-plan.json: two renders running at once had one overwrite the // other's mix between the write and the mux, so a finished video could carry the // wrong tune's audio. Per-output means the five tunes can render in parallel, // and SKIP_AUDIO=1 still finds the file because the name follows OUT. const songWav = path.join(SONG_DATA, `poly-song-${path.basename(OUT).replace(/\W+/g, "_")}.wav`); if (!SKIP_AUDIO) { let peak = 0; for (let i = 0; i < master.length; i += 1) peak = Math.max(peak, Math.abs(master[i])); { // trim to where the sound actually stops, plus a short release const floor = Math.max(1e-4, peak * 0.002); let last = 0; for (let i = master.length - 1; i >= 0; i -= 1) if (Math.abs(master[i]) > floor) { last = i; break; } if (last > 0) END = Math.min(TOTAL + 1.0, last / SR + 0.25); console.log(`audio ends at ${(last / SR).toFixed(2)}s; trimming output to ${END.toFixed(2)}s (slot end was ${TOTAL.toFixed(2)}s)`); } // Fading the last stretch turns the loop point into an ending. Cosine rather // than linear: a straight ramp is audible as a ramp. if (TAIL_FADE > 0) { const e = Math.min(master.length, Math.round(END * SR)); const n = Math.min(e, Math.round(TAIL_FADE * SR)); for (let i = 0; i < n; i += 1) master[e - n + i] *= 0.5 * (1 + Math.cos(Math.PI * (i / n))); for (let i = e; i < master.length; i += 1) master[i] = 0; console.log(` tail fade over the last ${TAIL_FADE}s, ending at ${END.toFixed(2)}s`); } const norm = peak > 0 ? 0.89 / peak : 1; for (let i = 0; i < master.length; i += 1) master[i] *= norm; writeWav(songWav, master, SR); console.log(`audio: ${(master.length / SR).toFixed(1)}s`); } else console.log("audio: reusing poly-song.wav"); // ---------- video ------------------------------------------------------------ const run = (args) => new Promise((res, rej) => execFile("ffmpeg", args, { maxBuffer: 1 << 26 }, (e) => (e ? rej(e) : res()))); // blankGaps: hold the clip for the note's sounding length, then pad to black // until this voice's next note. A PiP that keeps rolling reads as idle video; // blanking makes each voice visibly enter and leave. // Frame accounting, shared with the alpha mask below. Cuts are locked to the // beat by taking each clip's length from CUMULATIVE frame positions rather than // by rounding each duration on its own, so a mask built from the same numbers is // aligned by construction instead of by a second calculation that can drift. function voiceFrames(v) { const edges = v.plan.map((p) => Math.round(p.slotStart * FPS)); const endEdge = Math.round(END * FPS); return { // a voice may start after 0: the head is padded with black so overlays stay aligned lead: edges[0] ?? 0, frames: v.plan.map((_, i) => Math.max(1, (i + 1 < v.plan.length ? edges[i + 1] : endEdge) - edges[i])), }; } // Seconds this voice is SILENT after each slot -- 0 where the next note follows // straight on. One definition, used by the clip chain and by the alpha mask, so // the two cannot disagree about which notes end a phrase. function restsOf(v) { const { frames } = voiceFrames(v); return v.plan.map((p, i) => { const slotSec = frames[i] / FPS; const showFor = PULSE ? (p.videoDur ?? p.soundedDur) : (p.soundedDur ?? slotSec); const sounded = Math.max(1 / FPS, Math.min(showFor ?? slotSec, slotSec)); return Math.max(0, slotSec - sounded); }); } // How long this note's dissolve is: nothing unless a real rest follows it. const fadeAfter = (rest) => (CLIP_FADE > 0 && rest >= CLIP_FADE_GAP ? Math.min(CLIP_FADE, rest) : 0); async function buildVoiceVideo(v, w, h, outFile, blankGaps = false) { const dir = path.join(tmp, `v_${v.name}`); mkdirSync(dir, { recursive: true }); const { lead, frames } = voiceFrames(v); const jobs = v.plan.map((p, i) => ({ p, i, frames: frames[i], out: path.join(dir, `c${String(i).padStart(4, "0")}.mp4`), })); const queue = [...jobs]; let done = 0; const worker = async () => { while (queue.length) { const j = queue.shift(); const slotSec = j.frames / FPS; const showFor = PULSE ? (j.p.videoDur ?? j.p.soundedDur) : (j.p.soundedDur ?? slotSec); const sounded = Math.max(1 / FPS, Math.min(showFor ?? slotSec, slotSec)); const gap = Math.max(0, slotSec - sounded); // DEBUG=1 stamps each clip with its ID and the song time it lands at, so a // problem seen in a finished video can be named exactly ("the speaker at // 41s is ") instead of hunted for. const dbg = DEBUG ? `,drawtext=text='${ `${j.p.video}@${(+j.p.srcStart).toFixed(2)} t=${j.p.slotStart.toFixed(2)}s ${v.name}` .replace(/[\\:']/g, "") }':x=8:y=8:fontsize=${Math.max(11, Math.round(h / 26))}:fontcolor=white:box=1:boxcolor=black@0.65:boxborderw=4` : ""; // Dissolve into the rest that follows. Over a background the surround is // the GAME, so fading to black would show a black rectangle instead of // revealing it -- that layout uses the alpha mask below instead. Here the // surround is already black, so fading to black IS the dissolve. The clip // is held cf seconds longer and the black pad shortened to match, so the // picture dissolves during the silence rather than dimming under the note. const blank = blankGaps && gap > 1 / FPS; const cf = (blank && LAYOUT !== "bg") ? fadeAfter(gap) : 0; const vf = `scale=${w}:${h}:force_original_aspect_ratio=decrease,pad=${w}:${h}:(ow-iw)/2:(oh-ih)/2,setsar=1,fps=${FPS}` + dbg + (cf > 0 ? `,setpts=PTS-STARTPTS,fade=t=out:st=${sounded.toFixed(3)}:d=${cf.toFixed(3)}` : "") + (blank ? `,tpad=stop_mode=add:stop_duration=${(gap - cf).toFixed(3)}:color=black` : ""); const args = ["-nostdin", "-v", "error", "-y", "-ss", String(Math.max(0, j.p.srcStart - 0.02))]; if (blank) args.push("-t", (sounded + cf).toFixed(3)); args.push("-i", path.join(SONG_DATA, "media", `${j.p.video}.mp4`), "-frames:v", String(j.frames), "-an", "-vf", vf, "-c:v", "libx264", "-preset", "veryfast", "-crf", "21", "-pix_fmt", "yuv420p", "-video_track_timescale", "30000", j.out); try { await run(args); } catch { /* skip a bad clip rather than fail the render */ } done += 1; if (done % 50 === 0) console.log(` ${v.name} clips ${done}/${jobs.length}`); } }; await Promise.all(Array.from({ length: JOBS }, worker)); // A clip whose encode failed is skipped rather than failing the render, so the // SURVIVORS are what the layer actually contains -- and what an alpha mask has // to be built from, or every note after the gap would be masked by its // neighbour's window. const kept = jobs.filter((j) => existsSync(j.out)); const list = kept.map((j) => `file '${j.out}'`).join("\n"); const lp = path.join(dir, "list.txt"); writeFileSync(lp, `${list}\n`); const body = path.join(dir, "body.mp4"); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-f", "concat", "-safe", "0", "-i", lp, "-c", "copy", body], { maxBuffer: 1 << 26 }); if (lead > 0) { const black = path.join(dir, "black.mp4"); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-f", "lavfi", "-i", `color=c=black:s=${w}x${h}:r=${FPS}:d=${(lead / FPS).toFixed(3)}`, "-c:v", "libx264", "-preset", "veryfast", "-crf", "21", "-pix_fmt", "yuv420p", "-video_track_timescale", "30000", black], { maxBuffer: 1 << 26 }); const lp2 = path.join(dir, "list2.txt"); writeFileSync(lp2, `file '${black}'\nfile '${body}'\n`); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-f", "concat", "-safe", "0", "-i", lp2, "-c", "copy", outFile], { maxBuffer: 1 << 26 }); } else { execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-i", body, "-c", "copy", outFile], { maxBuffer: 1 << 26 }); } console.log(` ${v.name} video built (${jobs.length} clips)`); return { kept, lead }; } // The bg layout's per-note dissolve, as a parallel sequence of solid WHITE // blocks with exactly the frame counts the layer was built with. alphamerge // turns that luma into the layer's alpha, so the mask carries BOTH the on/off // window and its ramps -- and the layer can then be overlaid with no `enable` // gate at all. Alpha is the only way to dissolve here: the panel sits over the // game, so fading it to black would show a black rectangle rather than reveal // the background, and alpha cannot survive the libx264/yuv420p concat the // layer itself goes through. // // It also sidesteps the reason `enable` is chunked: ffmpeg cannot allocate a // 200-term expression, and a per-window ALPHA expression would hit that wall // harder. Solid colour encodes to almost nothing, and distinct (frames, on) // pairs repeat heavily across a tune, so each one is encoded once and the // concat list simply names it again. async function buildMaskVideo(v, kept, lead, w, h, outFile) { const dir = path.join(tmp, `m_${v.name}`); mkdirSync(dir, { recursive: true }); const rests = restsOf(v); const items = kept.map((j) => { const slotSec = j.frames / FPS; const dur = Math.max(1 / FPS, Math.min(j.p.videoDur ?? j.p.soundedDur ?? j.p.slotDur, j.p.slotDur, slotSec)); const onF = Math.max(1, Math.min(j.frames, Math.round(dur * FPS))); // Only a note followed by a real rest dissolves; everything else cuts, which // is one frame of ramp rather than a special case. const cf = Math.max(1 / FPS, fadeAfter(rests[j.i])); return { frames: j.frames, onF, cf, key: `${j.frames}_${onF}_${cf.toFixed(3)}` }; }); const uniq = [...new Map(items.map((it) => [it.key, it])).values()]; const files = new Map(); const queue = [...uniq]; const worker = async () => { while (queue.length) { const it = queue.shift(); const f = path.join(dir, `m_${it.key}.mp4`); const onSec = it.onF / FPS; // fade=t=out HOLDS black past its duration, so one white block covers the // whole slot: opaque for the note, then a ramp (or a single frame, which // is a cut) into the rest, then black until the next note. const vf = `fade=t=out:st=${onSec.toFixed(3)}:d=${it.cf.toFixed(3)}`; await run(["-nostdin", "-v", "error", "-y", "-f", "lavfi", "-i", `color=c=white:s=${w}x${h}:r=${FPS}:d=${((it.frames + 2) / FPS).toFixed(3)}`, "-frames:v", String(it.frames), "-vf", vf, "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-pix_fmt", "yuv420p", "-video_track_timescale", "30000", f]); files.set(it.key, f); } }; await Promise.all(Array.from({ length: JOBS }, worker)); const parts = []; if (lead > 0) { const black = path.join(dir, "lead.mp4"); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-f", "lavfi", "-i", `color=c=black:s=${w}x${h}:r=${FPS}:d=${((lead + 2) / FPS).toFixed(3)}`, "-frames:v", String(lead), "-c:v", "libx264", "-preset", "veryfast", "-crf", "18", "-pix_fmt", "yuv420p", "-video_track_timescale", "30000", black], { maxBuffer: 1 << 26 }); parts.push(black); } for (const it of items) parts.push(files.get(it.key)); const lp = path.join(dir, "list.txt"); writeFileSync(lp, `${parts.map((f) => `file '${f}'`).join("\n")}\n`); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-f", "concat", "-safe", "0", "-i", lp, "-c", "copy", outFile], { maxBuffer: 1 << 26 }); const dissolves = items.filter((it) => it.cf > 1.5 / FPS).length; console.log(` ${v.name} mask built (${items.length} blocks, ${uniq.length} distinct; ` + `${dissolves} dissolve over ${CLIP_FADE}s, ${items.length - dissolves} cut)`); } const full = voices.find((v) => v.role === "full"); const pips = voices.filter((v) => v.role !== "full" && v.plan.length); const PW = 384, PH = 216, M = 26; let inputs, fc; if (LAYOUT === "bg") { // Gameplay runs continuously behind. Each voice's layer is built with black in // its gaps, then black is keyed out, so a silent voice reveals the game rather // than cutting to a hard black rectangle. if (!process.env.BG_VIDEO) throw new Error("LAYOUT=bg needs BG_VIDEO"); // The melody panel is INSET rather than full-frame, so the game shows as a // border at all times instead of only in the gaps between notes. const SC = Number(process.env.BG_MAIN_SCALE ?? 0.70); const MW = Math.round((W * SC) / 2) * 2, MH = Math.round((H * SC) / 2) * 2; const MX = Math.round((W - MW) / 2); // BG_MAIN_TOP pins the melody panel's TOP instead of centring it vertically. // With a small BG_MAIN_SCALE that turns the three panels into a triangle -- // melody top-centre, the two accompaniment panels in the bottom corners -- // which on Metal Slug frames the character at the bottom centre instead of // covering the middle of the screen with one big panel. const MY = process.env.BG_MAIN_TOP !== undefined ? Math.round(Number(process.env.BG_MAIN_TOP)) : Math.round((H - MH) / 2); // No colour keying: a key would punch holes in dark parts of a clip. Instead // each layer is simply NOT DRAWN between its notes -- overlay is gated by an // enable expression built from the plan, so the background is untouched // wherever a voice is silent. // ffmpeg cannot allocate an enable expression with 200+ terms, so the windows // are split into small chunks and applied as a chain of overlays over a split // copy of the layer. Windows are disjoint, so the result is identical. const CHUNK = Number(process.env.ENABLE_CHUNK ?? 24); const windowChunks = (v) => { const terms = v.plan.map((p) => { const dur = Math.max(1 / FPS, Math.min(p.videoDur ?? p.soundedDur ?? p.slotDur, p.slotDur)); return `between(t,${p.slotStart.toFixed(3)},${(p.slotStart + dur).toFixed(3)})`; }); const out = []; for (let i = 0; i < terms.length; i += CHUNK) out.push(terms.slice(i, i + CHUNK).join("+")); return out; }; const baseFile = path.join(tmp, "base.mp4"); const built = [{ v: full, file: baseFile, w: MW, h: MH, x: String(MX), y: String(MY), ...(await buildVoiceVideo(full, MW, MH, baseFile, false)) }]; for (const [i, v] of pips.entries()) { const f = path.join(tmp, `pip_${v.name}.mp4`); built.push({ v, file: f, w: PW, h: PH, x: i === 0 ? `${M}` : `W-w-${M}`, y: `H-h-${M}`, ...(await buildVoiceVideo(v, PW, PH, f, false)) }); } // With a dissolve asked for, each layer gets a mask of the same length; the // masks are extra inputs after every layer, so the layer indexes are unchanged. if (CLIP_FADE > 0) { for (const L of built) { L.mask = path.join(tmp, `mask_${L.v.name}.mp4`); await buildMaskVideo(L.v, L.kept, L.lead, L.w, L.h, L.mask); } } inputs = ["-stream_loop", "-1", "-i", process.env.BG_VIDEO, ...built.flatMap((L) => ["-i", L.file]), ...built.filter((L) => L.mask).flatMap((L) => ["-i", L.mask])]; const layers = built.map((L, i) => ({ ...L, idx: i + 1, maskIdx: built.length + 1 + i })); // The background is DIMMED so the um panels read against it. That dim is why // the picture visibly darkens the moment the song starts: an intro segment // concatenated in front is not dimmed, so the join is a brightness step. // BG_DIM=0 removes it; matching the intro's own grade is the other fix. // NO DIM by default (2026-08-17). The background used to be darkened by -0.10 so the // panels read against it, and that is exactly what makes the picture CHANGE the moment // the song starts: an intro concatenated in front is not dimmed, so the join is a // brightness step. Every build then needed a matching dim on its bg-derived tails to // hide the same seam at the other end. // // The instruction is that the background must not change at all when the ums come in // -- no dim, no resize -- so the grade is now flat and the seam has nothing to show. // Set BG_DIM=-0.10 to get the old behaviour back. const BG_DIM = Number(process.env.BG_DIM ?? 0); const dim = BG_DIM ? `,eq=brightness=${BG_DIM}` : ""; const parts = [(process.env.BG_FIT === "pad" // a 4:3 capture must be padded, not cropped: cropping to 16:9 cuts the // top and bottom out of the gameplay ? `[0:v]scale=${W}:${H}:force_original_aspect_ratio=decrease,pad=${W}:${H}:(ow-iw)/2:(oh-ih)/2,setsar=1,fps=${FPS}${dim}[bg]` : `[0:v]scale=${W}:${H}:force_original_aspect_ratio=increase,crop=${W}:${H},setsar=1,fps=${FPS}${dim}[bg]`)]; let cur = "bg", stage = 0; for (const L of layers) { if (L.mask) { // The mask carries the window AND its ramps, so there is no enable gate -- // outside a note the alpha is simply 0 and the game shows through. // format=gray expands the mask's limited-range luma to full, so a white // block is alpha 255 and the panel is opaque where it sounds. parts.push(`[${L.idx}:v]format=yuva420p[L${L.idx}]`); parts.push(`[${L.maskIdx}:v]format=gray[M${L.idx}]`); parts.push(`[L${L.idx}][M${L.idx}]alphamerge[A${L.idx}]`); parts.push(`[${cur}][A${L.idx}]overlay=x=${L.x}:y=${L.y}:shortest=0:eof_action=pass[st${stage}]`); cur = `st${stage}`; stage += 1; continue; } const chunks = windowChunks(L.v); parts.push(`[${L.idx}:v]split=${chunks.length}${chunks.map((_, k) => `[L${L.idx}_${k}]`).join("")}`); chunks.forEach((expr, k) => { const outName = `st${stage}`; parts.push(`[${cur}][L${L.idx}_${k}]overlay=x=${L.x}:y=${L.y}:shortest=0:eof_action=pass:enable='${expr}'[${outName}]`); cur = outName; stage += 1; }); } parts.push(`[${cur}]null[v]`); fc = parts.join(";"); } else if (LAYOUT === "grid") { const CW = W / 2, CH = H / 2; const order = [full, ...pips].filter(Boolean); const files = []; for (const v of order) { const f = path.join(tmp, `cell_${v.name}.mp4`); await buildVoiceVideo(v, CW, CH, f, true); files.push(f); } inputs = ["-f", "lavfi", "-i", `color=c=black:s=${W}x${H}:r=${FPS}`, ...files.flatMap((f) => ["-i", f])]; const pos = order.length === 2 ? [[0, CH / 2], [CW, CH / 2]] : [[0, 0], [CW, 0], [0, CH], [CW, CH]]; const parts = []; let cur = "0:v"; files.forEach((_, i) => { const [x, y] = pos[i]; const last = i === files.length - 1; parts.push(`[${cur}][${i + 1}:v]overlay=x=${x}:y=${y}:shortest=${last ? 1 : 0}${last ? "[v]" : `[t${i}]`}`); cur = `t${i}`; }); fc = parts.join(";"); } else { const baseFile = path.join(tmp, "base.mp4"); await buildVoiceVideo(full, W, H, baseFile, PULSE); const pipFiles = []; for (const v of pips) { const f = path.join(tmp, `pip_${v.name}.mp4`); await buildVoiceVideo(v, PW, PH, f, true); pipFiles.push(f); } inputs = ["-i", baseFile, ...pipFiles.flatMap((f) => ["-i", f])]; const parts = []; let cur = "0:v"; pipFiles.forEach((_, i) => { const x = i === 0 ? `${M}` : `W-w-${M}`; parts.push(`[${cur}][${i + 1}:v]overlay=x=${x}:y=H-h-${M}:shortest=0${i === pipFiles.length - 1 ? "[v]" : `[t${i}]`}`); cur = `t${i}`; }); fc = parts.join(";"); } // Fade the PICTURE out over the same stretch the audio fades, ending at the same // END, in every layout. // // NOT the `fade` filter: it is a straight linear ramp and takes no curve (the // qsin/tri/etc curves belong to the AUDIO crossfade filters, `fade` has only // type/start_time/duration/color). Against the audio's raised cosine a linear // picture fade reads as a ramp, so the curve is generated instead -- a 64x64 // greyscale `geq` ramp of exactly 0.5*(1+cos) becomes the alpha of a black // plate laid over the composite. The ramp is written in LIMITED range (16..235) // because format=gray expands it back to a full 0..255 alpha, and it evaluates // to 0 before the fade starts, so the overlay is also gated by `enable` and // costs nothing for the rest of the tune. if (VIDEO_TAIL_FADE > 0) { const st = Math.max(0, END - VIDEO_TAIL_FADE), d = VIDEO_TAIL_FADE; const bIdx = inputs.filter((a) => a === "-i").length; inputs.push("-f", "lavfi", "-i", `color=c=black:s=${W}x${H}:r=${FPS}`, "-f", "lavfi", "-i", `color=c=gray:s=64x64:r=${FPS}`); fc = `${fc.replace(/\[v\]$/, "[vpre]")};` + `[${bIdx}:v]format=yuva420p[tfb];` + `[${bIdx + 1}:v]geq=lum='16+219*(0.5-0.5*cos(PI*clip((T-${st.toFixed(3)})/${d.toFixed(3)},0,1)))':cb=128:cr=128,` + `format=gray,scale=${W}:${H},setsar=1[tfr];` + `[tfb][tfr]alphamerge[tfp];` + `[vpre][tfp]overlay=0:0:shortest=0:eof_action=pass:enable='gte(t,${st.toFixed(3)})'[v]`; console.log(` video tail fade ${VIDEO_TAIL_FADE}s (raised cosine), ${st.toFixed(2)}s -> ${END.toFixed(2)}s`); } const nIn = inputs.filter((a) => a === "-i").length; // Stop at END, which is where the audio actually stops (plus a 0.25s release). // -shortest only bounds the output by the master BUFFER, which is allocated a // second past the last slot, so every build was carrying a tail of silent video // -- 2.3s of black and silence on the Metal Slug rebuild, after its fade had // already finished. Anything downstream that needs the true end still measures // the last audible sample rather than trusting a container duration. // // OUT_END overrides it, and OUT_END=0 turns it off. Some builds NEED background // past the last note: Mario RPG ends on the game's own wipe to black at 65.583s // while its last note is at 64.53s, so trimming to the audio cuts the ending off // -- measured as 7 trailing black frames where v4 had 32. const OUT_END = process.env.OUT_END !== undefined ? Number(process.env.OUT_END) : END; const endArgs = OUT_END > 0 ? ["-t", OUT_END.toFixed(3)] : []; execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", ...inputs, "-i", songWav, "-filter_complex", fc, "-map", "[v]", "-map", `${nIn}:a`, ...endArgs, "-c:v", "libx264", "-preset", "medium", "-crf", "22", "-pix_fmt", "yuv420p", "-c:a", "aac", "-b:a", "192k", "-ar", "48000", "-ac", "2", "-shortest", "-movflags", "+faststart", OUT], { maxBuffer: 1 << 26 }); const dur = execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", OUT]).toString().trim(); if (process.env.KEEP_TMP !== "1") rmSync(tmp, { recursive: true, force: true }); console.log(`${OUT} (${dur}s)`);