#!/usr/bin/env node // A voice made by CHOPPING ONE um into small pieces, for runs too fast to give a // clip each. // // The palette has a 0.14s minimum clip because a clip has to last long enough to // READ as an um. PkmRB-Battle1's track 1 runs at 78ms a note, so the arranger can // only ever return fragments and the track was left out of every build -- which is // why the melody's rests (20.3% of the song, six ~2.1s stretches where only the // quiet bass sounds) have nothing over them. // // The user's idea, and it dissolves the constraint: a fast run does not want 500 // separate ums, it wants ONE voice articulated rapidly. So take a single sustained // um and cut successive slices out of it. Timbre and speaker identity stay // continuous, and the rapid articulation reads as a stutter rather than as debris. // // Pitch comes from a plain speed change, the same "preferred path" render-poly // uses: playing a slice f times faster raises it 12*log2(f) semitones AND shortens // it by 1/f, so one resample delivers both at once with no phase vocoder anywhere // near it. For 78ms notes that matters more than usual - rubberband smears at // these lengths. // // DONOR= node chop-voice.mjs // // env: TEMPO_SCALE (default 1.15), GAIN (0.45), TOTAL (song seconds, pads the tail), // SHIFT (seconds to add to every note time), TRANSPOSE (semitones; default // auto, whole octaves only), MIN_DUR (0 - there is deliberately no floor here). import { readFileSync, writeFileSync } from "node:fs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const [notesFile, outFile] = process.argv.slice(2); const DONOR = process.env.DONOR ?? "0pHYgMHhi60@297.78"; const SCALE = Number(process.env.TEMPO_SCALE ?? 1.15); const GAIN = Number(process.env.GAIN ?? 0.45); const SHIFT_T = Number(process.env.SHIFT ?? 0); const TOTAL = Number(process.env.TOTAL ?? 0); const SR = 48000; // How far a slice may be resampled before an octave is folded instead, and how much // worse a downshift is than an upshift. See the note in the placement loop. // How far a slice may be resampled before an octave is folded INSTEAD -- and folding // is a last resort here, because this track's whole value is its contour. const FOLD_ABOVE = Number(process.env.FOLD_ABOVE ?? 6); const DOWN_W = Number(process.env.DOWN_W ?? 2.5); const UP_W = Number(process.env.UP_W ?? 1.0); // How much extra pitch cost is worth paying to get a DIFFERENT voice on the next // note. 0 disables rotation and always takes the nearest donor. const MIX_TOL = Number(process.env.MIX_TOL ?? 1.5); // A fast run articulates on its attacks. 4ms in was soft enough on a ~90ms note to // round every onset off, which is the other half of why the climb smeared; the // release can stay long because it is masked by the next note anyway. const ATT_MS = Number(process.env.ATT_MS ?? 1.2); const REL_MS = Number(process.env.REL_MS ?? 6); // How far to step through the donor between slices, as a fraction of the slice. // Below 1.0 the slices overlap and neighbouring notes share material. const ADVANCE = Number(process.env.ADVANCE ?? 1.0); function readWav(file) { const b = readFileSync(file); let p = 12, fmt = null, data = null; while (p + 8 <= b.length) { const id = b.toString("latin1", p, p + 4), sz = b.readUInt32LE(p + 4), body = p + 8; if (id === "fmt ") fmt = { ch: b.readUInt16LE(body + 2), sr: b.readUInt32LE(body + 4), bits: b.readUInt16LE(body + 14) }; if (id === "data") { data = b.subarray(body, body + sz); break; } p = body + sz + (sz & 1); } const by = fmt.bits / 8, n = Math.floor(data.length / by / fmt.ch), x = new Float32Array(n); for (let i = 0; i < n; i += 1) { let s = 0; for (let c = 0; c < fmt.ch; c += 1) s += data.readInt16LE((i * fmt.ch + c) * by) / 32768; x[i] = s / fmt.ch; } return { x, sr: fmt.sr }; } function writeWav(file, x, sr) { const n = x.length, b = Buffer.alloc(44 + n * 2); b.write("RIFF", 0, "latin1"); b.writeUInt32LE(36 + n * 2, 4); b.write("WAVE", 8, "latin1"); b.write("fmt ", 12, "latin1"); b.writeUInt32LE(16, 16); b.writeUInt16LE(1, 20); b.writeUInt16LE(1, 22); b.writeUInt32LE(sr, 24); b.writeUInt32LE(sr * 2, 28); b.writeUInt16LE(2, 32); b.writeUInt16LE(16, 34); b.write("data", 36, "latin1"); b.writeUInt32LE(n * 2, 40); for (let i = 0; i < n; i += 1) { let v = Math.max(-1, Math.min(1, x[i])); b.writeInt16LE(Math.round(v * 32767), 44 + i * 2); } writeFileSync(file, b); } // Catmull-Rom, as render-poly does: linear audibly dulls the top end at these rates. function resample(x, f) { const n = Math.max(1, Math.floor(x.length / f)); const y = new Float32Array(n); const at = (i) => (i < 0 ? x[0] : i >= x.length ? x[x.length - 1] : x[i]); for (let i = 0; i < n; i += 1) { const t = i * f, k = Math.floor(t), d = t - k; const a = at(k - 1), b = at(k), c = at(k + 1), e = at(k + 2); y[i] = b + 0.5 * d * (c - a + d * (2 * a - 5 * b + 4 * c - e + d * (3 * (b - c) + e - a))); } return y; } // One donor has to cover the whole run, and this run spans 22 semitones (A3-G5), // so a single um means a median 4.86 and a max 11.14 semitones of resampling -- far // enough that the top of the run chipmunks and the bottom growls. Several donors // across the pitch bands, with each note taking the NEAREST, keeps the shifts small // while still being a handful of chopped ums rather than a clip per note. const man = JSON.parse(readFileSync(path.join(SONG_DATA, "um-manifest.json"), "utf8")).items; const donors = DONOR.split(",").map((k) => k.trim()).filter(Boolean).map((k) => { const d = man.find((e) => e.k === k); if (!d) throw new Error(`donor ${k} not in the manifest`); const src = readWav(path.join(SONG_DATA, "wav48", `${d.v}.wav`)); const x = src.x.subarray(Math.round((d.clipStart + d.selA) * src.sr), Math.round((d.clipStart + d.selB) * src.sr)); return { k, x, sr: src.sr, midi: 69 + 12 * Math.log2(d.f0 / 440), f0: d.f0, tok: d.tok, cursor: 0, used: 0 }; }); for (const d of donors) { console.log(`donor ${d.k}: ${(d.x.length / d.sr).toFixed(3)}s sounded, f0 ${d.f0}Hz (MIDI ${d.midi.toFixed(2)}), "${d.tok}"`); } const donorMidi = donors.reduce((s, d) => s + d.midi, 0) / donors.length; const notes = JSON.parse(readFileSync(notesFile, "utf8")); // Octaves only, like the arranger's global transpose: it keeps the line's intervals // intact and only moves it into the register the donor can actually reach. let TR = process.env.TRANSPOSE !== undefined ? Number(process.env.TRANSPOSE) : null; if (TR === null) { const med = [...notes.map((n) => n.midi)].sort((x, y) => x - y)[Math.floor(notes.length / 2)]; TR = Math.round((donorMidi - med) / 12) * 12; } console.log(`transpose ${TR >= 0 ? "+" : ""}${TR} semitones (whole octaves)`); const endT = notes.reduce((m, n) => Math.max(m, n.t * SCALE + n.dur * SCALE), 0) + SHIFT_T; const outLen = Math.ceil(Math.max(endT, TOTAL) * SR) + SR; const out = new Float32Array(outLen); const ATT = Math.max(1, Math.round(ATT_MS / 1000 * SR)); const REL = Math.max(1, Math.round(REL_MS / 1000 * SR)); let placed = 0, clipped = 0, folded = 0, rr = 0; const shifts = [], intended = [], played = [], emitted = []; for (const n of notes) { const dur = n.dur * SCALE; const at = Math.round((n.t * SCALE + SHIFT_T) * SR); const target = n.midi + TR; // Pick the donor AND the octave together, preferring to FOLD rather than drag. // // Nearest-donor alone is not enough: whole-octave transposition put this run's // median target at MIDI 39 while the lowest donor sits at 40.33, so the entire // voice fell below every donor and 372 of 516 notes (72%) were dragged DOWN, // median -2.33 and worst -7.33. Reported by ear as "a lot of random notes pitch // waaay down for no reason", which is exactly what that is. // // Downshifts are penalised harder than upshifts (DOWN_W vs UP_W): slowing a slice // down thickens and growls it, while speeding it up only thins it, so the same // interval is uglier downwards. That asymmetry is the user's ask -- "harsh pitch // downs should be softened" -- rather than a tuning of mine. // // Folding an octave changes the line's contour at that note. For a MELODY that is // a real cost (see contourMax); for a granular texture under the tune it is // close to free, which is why the limit here is tighter than the melody's. // FOLDING IS A LAST RESORT, not the default. // // Folding an octave per note fixed the growl and destroyed the thing that made // this track worth adding: track 1 is a chromatic CLIMB, and folding 433 of its // 516 notes turned the climb into a sawtooth. Reported by ear as "it still // doesn't seem like a climb... just the previous version with some helium ums" — // the helium being the upward folds, the missing climb being the broken contour. // // So take the nearest donor at the written octave first, which plays the interval // the composer wrote. Only if that needs more than FOLD_ABOVE semitones is an // octave considered, and then the down-penalty applies. const cost = (s) => (s >= 0 ? s * UP_W : -s * DOWN_W); const opts = donors.map((c) => ({ c, s: target - c.midi })); let best = opts.reduce((b, o) => (cost(o.s) < cost(b.s) ? o : b), opts[0]); if (Math.abs(best.s) > FOLD_ABOVE) { for (const c of donors) { for (const k of [-1, 1]) { const s = target - c.midi + 12 * k; if (cost(s) < cost(best.s)) best = { c, s }; } } folded += 1; } // Rotate between donors that are near-equally good, instead of always taking the // single nearest. // // At this octave one donor is nearest for most of the run, so consecutive notes // were slices of the SAME sustained vowel taken a few tens of ms apart -- nearly // identical timbre, which smears a fast run into one continuous sound. Reported as // "the climb blends together a bit". Rotating costs a little pitch accuracy and // buys a different voice on adjacent notes, which is what makes a run articulate. let d = best.c, sh = best.s; if (MIX_TOL > 0) { const near = opts.filter((o) => cost(o.s) <= cost(best.s) + MIX_TOL); if (near.length > 1) { const pick = near[rr % near.length]; d = pick.c; sh = pick.s; rr += 1; } } d.used += 1; shifts.push(sh); // played pitch = donor + shift, which is the written target unless an octave was // folded in. Keeping both is how the contour check below can tell. intended.push(target); played.push(d.midi + sh); emitted.push({ t: +(n.t * SCALE + SHIFT_T).toFixed(4), dur: +dur.toFixed(4), donor: d.k, shift: +sh.toFixed(2) }); const f = Math.pow(2, sh / 12); const want = Math.max(2, Math.round(dur * SR * f)); // source samples needed if (d.cursor + want > d.x.length) d.cursor = 0; // walk the donor, wrap at the end let slice = d.x.subarray(d.cursor, d.cursor + want); if (slice.length < want) { d.cursor = 0; slice = d.x.subarray(0, want); } d.cursor += Math.max(1, Math.round(want * ADVANCE)); const seg = resample(slice, f); for (let i = 0; i < seg.length; i += 1) { const j = at + i; if (j < 0 || j >= outLen) { clipped += 1; continue; } let g = GAIN; if (i < ATT) g *= i / ATT; if (i >= seg.length - REL) g *= (seg.length - i) / REL; out[j] += seg[i] * g; } placed += 1; } // Only guard against summing overlaps into clipping; do not normalise, or the voice's // level stops being comparable between renders. let peak = 0; for (let i = 0; i < outLen; i += 1) peak = Math.max(peak, Math.abs(out[i])); if (peak > 0.98) { const k = 0.98 / peak; for (let i = 0; i < outLen; i += 1) out[i] *= k; } // Report shifts SIGNED and by direction. The absolute figure hid the actual fault: // |shift| p50 1.33 looked fine while 72% of the notes were being dragged DOWN, which // is what put the voice under the bass and made it inaudible as a line. const s = [...shifts].sort((x, y) => x - y); const q = (p) => s[Math.floor(p * (s.length - 1))]; const down = shifts.filter((x) => x < 0), up = shifts.filter((x) => x > 0); console.log(`placed ${placed} notes over ${(outLen / SR).toFixed(2)}s, peak ${peak.toFixed(3)}` + (clipped ? `, ${clipped} samples past the end` : "")); console.log(`shift: ${down.length} down / ${up.length} up / ${shifts.length - down.length - up.length} exact`); console.log(` worst down ${(s[0] ?? 0).toFixed(2)}, worst up ${(s[s.length - 1] ?? 0).toFixed(2)}, median ${q(0.5).toFixed(2)} semitones`); console.log(` beyond -2 semitones: ${shifts.filter((x) => x < -2).length} notes`); // Does it still CLIMB? A fold makes the played interval differ from the written one // by an octave, so a run of rising semitones becomes a sawtooth. Counting the breaks // is the only way to tell without listening, and the ear caught this before any // number did: |shift| looked healthy while 84% of the contour had been broken. let breaks = 0, rising = 0, risingKept = 0; for (let i = 1; i < intended.length; i += 1) { const want = intended[i] - intended[i - 1]; const got = played[i] - played[i - 1]; if (Math.abs(got - want) > 0.5) breaks += 1; if (want > 0 && want <= 2) { rising += 1; if (got > 0) risingKept += 1; } } console.log(`contour: ${folded} notes folded an octave, ${breaks}/${intended.length - 1} intervals broken ` + `(${(100 * breaks / (intended.length - 1)).toFixed(0)}%)`); console.log(` rising steps preserved: ${risingKept}/${rising} (${(100 * risingKept / (rising || 1)).toFixed(0)}%) ` + `— this is the "does it still climb" number`); console.log("notes per donor: " + donors.map((d) => `${d.k} ${d.used}`).join(", ")); writeWav(outFile, out, SR); console.log(`wrote ${outFile}`); // EMIT_NOTES records which donor each note actually came from, so a visual panel can // show the clip that is really sounding rather than a plausible-looking stand-in. if (process.env.EMIT_NOTES) { writeFileSync(process.env.EMIT_NOTES, JSON.stringify(emitted, null, 1)); console.log(`wrote ${process.env.EMIT_NOTES} (${emitted.length} notes with their donor)`); }