#!/usr/bin/env node // Make a rising line READ as rising, when pitch alone cannot carry it. // // The run voice plays one unique um per note, and a 1-2 semitone step is inaudible // across clips that each bring their own vowel, formants and loudness. Exaggerating // the intervals helps but runs out. So carry the direction with cues that do NOT // depend on the clip: // // AMP loudness rises with pitch // BRI brightness rises with pitch -- the audio is split into two fixed bands and // the MIX between them moves, because ffmpeg will not automate a filter's // frequency // PAN stereo position rises with pitch; completely independent of timbre, so it // survives the thing that defeats the pitch cue // // AUTOMATION IS A CONTROL SIGNAL, NOT AN EXPRESSION. // // The first version built a nested if() per note. That works for a 79-note test window // and dies on the real 516-note voice: ffmpeg cannot allocate an expression with // hundreds of terms -- the same wall render-poly documents for `enable`, which it // works around by chunking. A volume expression cannot be chunked, so instead each // cue is rendered as a mono WAV at audio rate and applied with `amultiply`. No // expression, no term limit, and the resolution is per-sample rather than per-note. // // Control wavs are 16-bit, so they cannot carry a value above 1.0. Every curve is // normalised by its own peak and the peak is put back as a static `volume` afterwards. // // node climb-fx.mjs // AMP=0.5 BRI=0.8 PAN=0.7 depth of each cue, 0 disables import { readFileSync, writeFileSync, mkdirSync, existsSync } from "node:fs"; import { execFileSync } from "node:child_process"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const [PLANF, VOICE, IN, OUT] = process.argv.slice(2); const AMP = Number(process.env.AMP ?? 0.5); const BRI = Number(process.env.BRI ?? 0.8); const PAN = Number(process.env.PAN ?? 0.7); const SPLIT_HZ = Number(process.env.SPLIT_HZ ?? 900); const SR = 48000; const RAMP = Number(process.env.RAMP_MS ?? 25) / 1000; // glide between note values const TMP = path.join(SONG_DATA, "climbctrl"); if (!existsSync(TMP)) mkdirSync(TMP, { recursive: true }); const plan = JSON.parse(readFileSync(PLANF, "utf8")); const v = plan.voices.find((x) => x.name === VOICE) ?? plan.voices[0]; const notes = [...v.plan].sort((a, b) => a.slotStart - b.slotStart); const mids = notes.map((n) => n.targetMidi); const lo = Math.min(...mids), hi = Math.max(...mids); const dur = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", IN]).toString().trim()); console.log(`${notes.length} notes, targetMidi ${lo}-${hi}, over ${dur.toFixed(2)}s`); // p(t) in [-1,+1]: where the note sounding at t sits in the line's range, held until // the next note and glided over RAMP so the cues do not click between notes. const n = Math.ceil(dur * SR) + SR; const p = new Float32Array(n); { let cursor = 0, prev = 0; for (const nt of notes) { const at = Math.round(nt.slotStart * SR); const val = hi > lo ? ((nt.targetMidi - lo) / (hi - lo)) * 2 - 1 : 0; const r = Math.min(Math.round(RAMP * SR), Math.max(1, at - cursor)); for (let i = cursor; i < Math.min(at, n); i += 1) p[i] = prev; for (let i = 0; i < r && at + i < n; i += 1) p[at + i] = prev + (val - prev) * (i / r); cursor = Math.min(at + r, n); prev = val; } for (let i = cursor; i < n; i += 1) p[i] = prev; } /** write a control curve as a mono wav, normalised, returning the factor taken out */ function ctrl(name, fn) { const x = new Float32Array(n); let peak = 0; for (let i = 0; i < n; i += 1) { x[i] = Math.max(0, fn(p[i])); if (x[i] > peak) peak = x[i]; } peak = peak || 1; const b = Buffer.alloc(44 + n * 2); b.write("RIFF", 0, "latin1"); b.writeUInt32LE(36 + n * 2, 4); b.write("WAVE", 8, "latin1"); b.write("fmt ", 12, "latin1"); b.writeUInt32LE(16, 16); b.writeUInt16LE(1, 20); b.writeUInt16LE(1, 22); b.writeUInt32LE(SR, 24); b.writeUInt32LE(SR * 2, 28); b.writeUInt16LE(2, 32); b.writeUInt16LE(16, 34); b.write("data", 36, "latin1"); b.writeUInt32LE(n * 2, 40); for (let i = 0; i < n; i += 1) b.writeInt16LE(Math.round((x[i] / peak) * 32767), 44 + i * 2); const f = path.join(TMP, `${name}.wav`); writeFileSync(f, b); return { f, peak }; } const cLo = ctrl("lo", (q) => (1 + AMP * q) * Math.max(0, 1 - BRI * q * 0.5)); const cHi = ctrl("hi", (q) => (1 + AMP * q) * Math.max(0, 1 + BRI * q)); const cL = ctrl("panL", (q) => Math.max(0, 1 - PAN * q * 0.5)); const cR = ctrl("panR", (q) => Math.max(0, 1 + PAN * q * 0.5)); console.log(`AMP ${AMP} BRI ${BRI} PAN ${PAN} split ${SPLIT_HZ}Hz ramp ${RAMP * 1000}ms`); // mono through the bands, then split to stereo for the pan; amultiply needs both of // its inputs in the same layout, which is why everything stays mono until the join const fc = [ `[0:a]aformat=sample_fmts=fltp:sample_rates=${SR}:channel_layouts=mono,asplit=2[a1][a2]`, `[a1]lowpass=f=${SPLIT_HZ}[loF]`, `[a2]highpass=f=${SPLIT_HZ}[hiF]`, `[1:a]aformat=sample_fmts=fltp:sample_rates=${SR}:channel_layouts=mono[cl]`, `[2:a]aformat=sample_fmts=fltp:sample_rates=${SR}:channel_layouts=mono[ch]`, `[3:a]aformat=sample_fmts=fltp:sample_rates=${SR}:channel_layouts=mono[cpl]`, `[4:a]aformat=sample_fmts=fltp:sample_rates=${SR}:channel_layouts=mono[cpr]`, `[loF][cl]amultiply,volume=${cLo.peak.toFixed(4)}[loG]`, `[hiF][ch]amultiply,volume=${cHi.peak.toFixed(4)}[hiG]`, `[loG][hiG]amix=inputs=2:duration=first:normalize=0,asplit=2[mL][mR]`, `[mL][cpl]amultiply,volume=${cL.peak.toFixed(4)}[pl]`, `[mR][cpr]amultiply,volume=${cR.peak.toFixed(4)}[pr]`, `[pl][pr]join=inputs=2:channel_layout=stereo,alimiter=limit=0.98:level=disabled[a]`, ].join(";"); execFileSync("ffmpeg", ["-nostdin", "-v", "warning", "-y", "-i", IN, "-i", cLo.f, "-i", cHi.f, "-i", cL.f, "-i", cR.f, "-filter_complex", fc, "-map", "0:v?", "-map", "[a]", "-t", dur.toFixed(3), "-c:v", "copy", "-c:a", "aac", "-b:a", "192k", OUT], { stdio: ["ignore", "inherit", "inherit"] }); console.log(`wrote ${OUT}`);