#!/usr/bin/env node // For every line the record says more than once, find the CLEAREST take of it. // // The catalogue holds 40 hooks but only 27 distinct lines: "Test your might." is // shouted six times, "MORDO GRANDE!" four, the fighter names three each. They are not // equally good -- some are further off-mic, some land under the music, some are clipped // by the separator. Using the best take everywhere costs nothing and is the difference // between a shout that reads and one that does not. // // "Most defined" is measured, not assumed, on three things a shout needs: // // presence the share of energy in 1-4 kHz. That band carries consonants and // articulation; a take that is all low-mid is a mumble however loud. // snr the loud frames against the quiet ones, in dB. A take separated cleanly // has a low floor; one dragged out from under the music does not. // attack mean positive frame-to-frame energy jump -- how sharply it starts. // // Each is turned into a z-score across the takes of that line, so the comparison is // within the group and no term's raw units dominate. // // PRESENCE CARRIES THE DECISION (weight 1.0); attack is half and SNR only a quarter. // The weights are not arbitrary: measured across the 13 repeated lines, "snr" as // computed here is really *how much quiet surrounds the shout*, which a tightly-cut // take has none of -- hook-17 "Finish him!" reads 13.5 dB against hook-06's 45.5 dB // purely because it is 1.20s to hook-06's 3.95s, and hook-17 is the take // mk-fatal-finish3.sh already uses for the finale. Weighting it equally would rank // takes by their padding. Presence alone picks the same winner in 11 of 13 groups, so // the choice is robust rather than tuned. // // node pick-take.mjs [--apply ] import { readFileSync, writeFileSync } from "node:fs"; import { execFileSync } from "node:child_process"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const [CAT, HOOKDIR, ...rest] = process.argv.slice(2); const applyIdx = rest.indexOf("--apply"); const [OVL, OUT] = applyIdx >= 0 ? rest.slice(applyIdx + 1) : []; const abs = (f) => (path.isAbsolute(f) ? f : path.join(SONG_DATA, f)); const SR = 48000, N = 1024; function fft(re, im) { const n = re.length; for (let i = 1, j = 0; i < n; i += 1) { let bit = n >> 1; for (; j & bit; bit >>= 1) j ^= bit; j ^= bit; if (i < j) { [re[i], re[j]] = [re[j], re[i]]; [im[i], im[j]] = [im[j], im[i]]; } } for (let len = 2; len <= n; len <<= 1) { const ang = -2 * Math.PI / len; for (let i = 0; i < n; i += len) for (let k = 0; k < len / 2; k += 1) { const wr = Math.cos(ang * k), wi = Math.sin(ang * k), ur = re[i + k], ui = im[i + k]; const vr = re[i + k + len / 2] * wr - im[i + k + len / 2] * wi, vi = re[i + k + len / 2] * wi + im[i + k + len / 2] * wr; re[i + k] = ur + vr; im[i + k] = ui + vi; re[i + k + len / 2] = ur - vr; im[i + k + len / 2] = ui - vi; } } } function score(file) { const b = execFileSync("ffmpeg", ["-v", "error", "-i", file, "-ac", "1", "-ar", String(SR), "-f", "s16le", "-"], { maxBuffer: 1 << 28 }); const a = new Int16Array(b.buffer, b.byteOffset, b.length / 2); const hop = Math.round(0.010 * SR); const frames = []; for (let i = 0; i + N <= a.length; i += hop) { const re = new Float64Array(N), im = new Float64Array(N); for (let k = 0; k < N; k += 1) re[k] = (a[i + k] / 32768) * (0.5 - 0.5 * Math.cos(2 * Math.PI * k / (N - 1))); fft(re, im); let band = 0, tot = 0; for (let k = 1; k < N / 2; k += 1) { const m = re[k] * re[k] + im[k] * im[k], f = k * SR / N; tot += m; if (f >= 1000 && f <= 4000) band += m; } frames.push({ tot, band }); } if (frames.length < 4) return null; const byTot = frames.map((f) => f.tot).sort((x, y) => y - x); const loudN = Math.max(1, Math.round(frames.length * 0.30)); const quietN = Math.max(1, Math.round(frames.length * 0.20)); const loud = byTot.slice(0, loudN).reduce((s, x) => s + x, 0) / loudN; const quiet = byTot.slice(-quietN).reduce((s, x) => s + x, 0) / quietN; // presence measured on the LOUD frames only -- the quiet ones are the tail and the // floor, and averaging them in rewards a take for being mostly silence. const thresh = byTot[loudN - 1]; let band = 0, tot = 0; for (const f of frames) if (f.tot >= thresh) { band += f.band; tot += f.tot; } let jump = 0, nj = 0; for (let i = 1; i < frames.length; i += 1) { const d = 10 * Math.log10((frames[i].tot + 1e-12) / (frames[i - 1].tot + 1e-12)); if (d > 0) { jump += d; nj += 1; } } return { presence: tot > 0 ? band / tot : 0, snr: 10 * Math.log10((loud + 1e-12) / (quiet + 1e-12)), attack: nj ? jump / nj : 0, dur: a.length / SR, }; } const cat = JSON.parse(readFileSync(abs(CAT), "utf8")); const hooks = Array.isArray(cat) ? cat : (cat.hooks || []); const groups = new Map(); for (const h of hooks) { const key = (h.text || "").trim().toLowerCase(); if (!groups.has(key)) groups.set(key, []); groups.get(key).push(h); } const z = (xs) => { const m = xs.reduce((s, x) => s + x, 0) / xs.length; const sd = Math.sqrt(xs.reduce((s, x) => s + (x - m) ** 2, 0) / xs.length) || 1; return xs.map((x) => (x - m) / sd); }; const winner = new Map(); // hook id -> the id whose audio it should use for (const [line, hs] of groups) { if (hs.length < 2) continue; const sc = hs.map((h) => ({ h, s: score(path.join(abs(HOOKDIR), h.file)) })).filter((r) => r.s); if (sc.length < 2) continue; const zp = z(sc.map((r) => r.s.presence)), zs = z(sc.map((r) => r.s.snr)), za = z(sc.map((r) => r.s.attack)); sc.forEach((r, i) => { r.total = zp[i] + 0.25 * zs[i] + 0.5 * za[i]; }); const best = sc.slice().sort((a, b) => b.total - a.total)[0]; console.log(`\n"${line}" ${sc.length} takes -> ${best.h.id}`); for (const r of sc.slice().sort((a, b) => b.total - a.total)) { console.log(` ${r.h.id.padEnd(8)} score ${r.total.toFixed(2).padStart(6)} ` + `presence ${(100 * r.s.presence).toFixed(1)}% snr ${r.s.snr.toFixed(1)}dB ` + `attack ${r.s.attack.toFixed(2)}dB ${r.s.dur.toFixed(2)}s` + (r.h.id === best.h.id ? " <== clearest" : "")); } for (const r of sc) if (r.h.id !== best.h.id) winner.set(r.h.id, best.h); } // Does the user's hunch hold -- are the later takes the clearer ones? const nums = [...winner.entries()].map(([from, to]) => [Number(from.split("-")[1]), Number(to.id.split("-")[1])]); const later = nums.filter(([f, t]) => t > f).length; console.log(`\n${winner.size} occurrences would change take; the winner is the LATER recording in ` + `${later} of ${nums.length} (${(100 * later / Math.max(1, nums.length)).toFixed(0)}%)`); if (OVL && OUT) { // The winning audio is COPIED UNDER EACH OCCURRENCE'S OWN NAME rather than // repointing the overlay at the winner's file. // // That is not tidiness, it is required. `mk-mix-hooks.mjs` keys HOOK_GAINS by the // hook id, and it derives that id from the filename -- so pointing hook-08, hook-19 // and hook-32 all at hook-19.wav collapses three placements onto one id and one // gain. Measured, that is exactly what happened: the fitter's spread went from 1.00x // to 1.54x on the short and to 1383x on the long, with one hook reading a ratio of // 0.002, because three occurrences were fighting over a single gain. // // Copying keeps every id distinct, so each occurrence keeps its own level, its own // window and its own placement -- and only the recording changes. const dir = process.env.TAKE_DIR ?? "mkvocals/hooks-best"; execFileSync("mkdir", ["-p", abs(dir)]); const ovl = JSON.parse(readFileSync(abs(OVL), "utf8")); let n = 0; for (const o of ovl) { const id = path.basename(o.file, ".wav"); const w = winner.get(id); const cur = hooks.find((h) => h.id === id); let srcName = `${id}.wav`; if (w && cur && w.dur > cur.dur + 0.25) { // A longer take would run past the window it was placed in -- hook-01 has to // finish before the game's gong at 18.086s, for one. console.log(` ${id}: keeping its own take (${w.id} is ${(w.dur - cur.dur).toFixed(2)}s longer)`); } else if (w) { srcName = w.file; o.label = `${o.label ?? id} [take ${w.id}]`; n += 1; } execFileSync("cp", [path.join(abs(HOOKDIR), srcName), path.join(abs(dir), `${id}.wav`)]); o.file = path.join(dir, `${id}.wav`); } writeFileSync(abs(OUT), JSON.stringify(ovl, null, 1)); console.log(`\n${n} of ${ovl.length} occurrences now use another take; all ${ovl.length} ` + `audio files written to ${dir}/ under their OWN ids -> ${OUT}`); }