#!/usr/bin/env node // Multi-voice arranger: one plan per voice, from DISJOINT sample pools. // // Two things make this sound like an arrangement rather than a pile of notes: // * each voice draws from its own set of SOURCE VIDEOS, so a part keeps one // consistent timbre (and one consistent face in the picture-in-picture); // * pools are disjoint, so no clip can sound in two voices at the same instant. // Sources are assigned by pitch: the lowest-voiced videos play the bass. import { readFileSync, writeFileSync, readdirSync } from "node:fs"; import path from "node:path"; import { HEAD_CODES, DELIBERATE_CODES } from "./reasons.mjs"; import { SONG_DATA } from "./paths.mjs"; const DIR = path.resolve(path.dirname(new URL(import.meta.url).pathname)); const SCALE = Number(process.env.TEMPO_SCALE ?? 1.15); const MAX_IOI = Number(process.env.MAX_IOI ?? 0.70); const CONTOUR_MAX = Number(process.env.CONTOUR_MAX ?? 14); const WINDOW = Number(process.env.WINDOW ?? 40); const IMPURE_PENALTY = Number(process.env.IMPURE_PENALTY ?? 25); const PROV_PENALTY = Number(process.env.PROV_PENALTY ?? 8); const DELIBERATE_BONUS = Number(process.env.DELIBERATE_BONUS ?? 10); // Chronological ordering never really read in the finished cut, so locality can // be relaxed to buy closer pitch matches (less processing of each um). const LOCAL = Number(process.env.LOCAL ?? 0.006); const WOBBLE_OK = Number(process.env.WOBBLE_OK ?? 120); // cents of drift tolerated free const WOBBLE_W = Number(process.env.WOBBLE_W ?? 0.03); // What a semitone of pitch-shift COSTS, against everything else in the score. // // 12 was never chosen against a measurement, and it makes a semitone worth // exactly two tenths of a second of missing clip. Measured on the Pokemon plan, // that balance produces a BIMODAL result: 71 melody notes land under 0.5 // semitones and then 47 pile up between 2 and 3, with almost nothing between 1 // and 2. A gap-then-cluster is the arranger giving up on pitch and buying // length instead, and 2-3 semitones on a voice clip is audible as a note that // has been dragged rather than sung. const SHIFT_W = Number(process.env.SHIFT_W ?? 12); const dates = JSON.parse(readFileSync(path.join(DIR, "dates.json"), "utf8")); const midiOf = (hz) => 69 + 12 * Math.log2(hz / 440); const med = (a) => { const s = [...a].sort((x, y) => x - y); return s[s.length >> 1]; }; // voices: file, label, gain, video role, share of the palette const VOICES = JSON.parse(process.env.VOICES ?? JSON.stringify([ { name: "melody", lead: "pkmn-lead.json", gain: 1.0, role: "full", unique: true, reuseGap: 20, contourMax: 14 }, { name: "bass", lead: "pkmn-bass.json", gain: 0.62, role: "pip1", unique: false, reuseGap: 12, contourMax: 5 }, { name: "arp", lead: "pkmn-arp.json", gain: 0.42, role: "pip2", unique: false, reuseGap: 12, contourMax: 9 }, ])); // ---- palette (same gates as arrange.mjs) ------------------------------------ let all = []; for (const f of readdirSync(path.join(SONG_DATA, "cand2")).filter((f) => f.endsWith(".json"))) { all.push(...JSON.parse(readFileSync(path.join(SONG_DATA, "cand2", f), "utf8")).candidates); } { const seen = new Set(); all = all.filter((c) => { const k = `${c.video}@${c.start}-${c.end}`; if (seen.has(k)) return false; seen.add(k); return true; }); } const ACCEPTED = new Set(JSON.parse(readFileSync(path.join(DIR, "accepted.json"), "utf8")).accepted); let CORE = {}; try { CORE = JSON.parse(readFileSync(path.join(DIR, "corepitch.json"), "utf8")); } catch {} const akey = (c) => `${c.video}@${(+c.start).toFixed(2)}`; const DROP = new Set(["JVD2nOEJ9t8_none"]); // The word a HUMAN says the clip contains, which beats what ASR guessed. // // The ASR token rode along unquestioned and is often wrong: the stray "the"s and // "errr"s that got into Yoshi's melody were mislabelled clips this gate could // not filter, because it only ever saw `c.token`. The triage tool records the // real word in words.json; ASR stays as the fallback for the ~99% of clips // nobody has labelled yet. let WORDS = {}; try { WORDS = JSON.parse(readFileSync(path.join(DIR, "words.json"), "utf8")); } catch {} const wordOf = (c) => WORDS[akey(c)]?.word ?? c.token ?? ""; // ---- how a clip STARTS ------------------------------------------------------ // attack.mjs separates a stopped onset ("d'ummm" -- an aperiodic burst sitting in // front of the voicing, which is the /d/) from a smooth one ("ummm"). It is the // difference between a run that articulates and one that smears into a single // sound, and until now nothing in the arranger read the file. // // Two settings, and the split between them is measured rather than chosen: // attackPenalty costs a clip that is not percussive at all. 2,118 of the 3,331 // clips free of the melody and bass qualify -- 4.1x a 516-note // voice -- so this can bite hard without starving anything. // plosiveBonus rewards a STRONG burst (>=2.0). Only 360 clips qualify, which // is 0.7x that voice, so it can never be a filter; it has to be // a preference that runs out gracefully. let ATTACK = {}; try { ATTACK = JSON.parse(readFileSync(path.join(DIR, "attack.json"), "utf8")); } catch {} const attackOf = (c) => ATTACK[akey(c)]; const isPercussive = (c) => { const a = attackOf(c); return !!a && ((a.burst ?? 1) >= 1.6 || (a.riseMs ?? 99) <= 4); }; const isPlosive = (c) => { const a = attackOf(c); return !!a && (a.burst ?? 1) >= 2.0; }; { // Say how much the labels actually change. Zero overrides with a non-empty // words.json means the keys do not line up and the gate is silently still // running on ASR -- which is the failure mode worth catching loudly, because // nothing else about the render would look wrong. // Count KEYS, not rows: `all` is deduped on video@start-end, so one clip key // can still appear several times and would inflate every figure here. const byKey = new Map(); for (const c of all) if (!byKey.has(akey(c))) byKey.set(akey(c), c); const labelled = [...byKey.entries()].filter(([k]) => WORDS[k]); const differ = labelled.filter( ([k, c]) => WORDS[k].word.toLowerCase().trim() !== String(c.token ?? "").toLowerCase().trim(), ); const inPalette = differ.filter(([k]) => ACCEPTED.has(k)); console.log( `words.json: ${Object.keys(WORDS).length} labels, ${labelled.length} matched to a clip, ` + `${differ.length} differ from ASR (${inPalette.length} of them in the palette)`, ); if (Object.keys(WORDS).length && !labelled.length) { console.error(" !! words.json has labels but NONE matched a clip key -- the override is unwired"); } } let IMPURE = new Set(); try { IMPURE = new Set(JSON.parse(readFileSync(path.join(DIR, "impure.json"), "utf8"))); } catch {} // Deliberately-dirty clips split by WHICH side the stray word sits on, taken // from the reason recorded during review. A leading word ("and um") belongs at // the start of a run of notes; a trailing word ("um and") belongs at the end. const HEAD_OFFSET = new Map(); // clip key -> seconds from clip start to the end of the leading word let IMPURE_HEAD = new Set(), IMPURE_BOTH = new Set(), DELIBERATE = new Set(); try { const R = JSON.parse(readFileSync(path.join(DIR, "reasons.json"), "utf8")); for (const [k, v] of Object.entries(R)) { if (v.verdict !== 2) continue; if (HEAD_CODES.has(v.code)) IMPURE_HEAD.add(k); if (DELIBERATE_CODES.has(v.code)) DELIBERATE.add(k); } } catch {} // keepside.json OVERRIDES the reason code. The review pass tagged every // deliberate keep as "leading" whether the word led or trailed, so the side is // only trustworthy once the dedicated keep pass has decided it. try { const KS = JSON.parse(readFileSync(path.join(DIR, "keepside.json"), "utf8")); for (const [k, side] of Object.entries(KS)) { DELIBERATE.add(k); IMPURE_HEAD.delete(k); IMPURE_BOTH.delete(k); if (side === 1) IMPURE_HEAD.add(k); else if (side === 3) IMPURE_BOTH.add(k); } } catch {} try { const HO = JSON.parse(readFileSync(path.join(DIR, "headoffset.json"), "utf8")); for (const [k, v] of Object.entries(HO)) HEAD_OFFSET.set(k, v); } catch {} // Clips auto-graded by isolation rather than by ear. They measure ~89% good against // a 75% baseline, which is worth having, but not worth preferring: carry a penalty // so a human-graded clip always wins the slot and these only fill what is left. let PROVISIONAL = new Set(); try { PROVISIONAL = new Set(JSON.parse(readFileSync(path.join(DIR, "provisional.json"), "utf8"))); } catch {} // Clips from a source video where some clip was judged "other speaker, not Jer". // This list EXISTED for a long time before anything read it, and every build in // between was free to pick from it: Yoshi's "another speaker" at 45s was // vfMRAdhSSrc@946.46, chosen after a single bad clip from the SAME video had // been rejected by hand. Rejecting one clip does not help -- the arranger simply // picks another from that video. // // These are INTERVIEWS, not the wrong person: they hold Jer AND a guest, so most // of their clips are good and banning the video throws them away. PITCH splits // the speakers -- the same signal the fitted order model already ranks highly // ("high f0 usually means a different speaker"). vfMRAdhSSrc has two clusters, // 78.9-92.8Hz (Jer) and 125-218Hz (the guest), and BOTH complained-about clips // sat in the high one. So inside a flagged video keep only clips at or below // SUSPECT_F0. SUSPECT=0 disables the gate. const SUSPECT_F0 = Number(process.env.SUSPECT_F0 ?? 115); let SUSPECT = new Set(); if (process.env.SUSPECT !== "0") { try { const S = JSON.parse(readFileSync(path.join(DIR, "suspect-sources.json"), "utf8")); SUSPECT = new Set((Array.isArray(S) ? S : S.sources ?? []).map((s) => (typeof s === "string" ? s : s.k))); } catch {} } // A ceiling on pitch for EVERY source, not just the flagged ones. // // SUSPECT_F0 only screens videos somebody has already flagged, so a high-pitched // clip from an unflagged source sails through: 201 of Yoshi's 723 picks measure // above 115Hz and ten of them above 150, against a corpus median of 103 and p95 // of 127. Those are the picks that get reported as "another speaker", and they // cannot be fixed with swap-clip -- the arranger chose them BECAUSE their pitch // matched, so no clip near the corpus median is within a swap's 3-semitone // window. The lever has to be here. Default 200 keeps the old behaviour. const F0_CEIL = Number(process.env.F0_CEIL ?? 200); let good = all.filter((c) => ACCEPTED.has(akey(c)) && dates[c.video] && !DROP.has(c.video) && !(SUSPECT.has(akey(c)) && (CORE[akey(c)] ? CORE[akey(c)].f0 : c.f0) > SUSPECT_F0) && c.dur >= 0.14 && c.f0 >= 70 && c.f0 <= F0_CEIL) .map((c) => { const k = akey(c), core = CORE[k]; return { ...c, midi: midiOf(core ? core.f0 : c.f0), dur: core ? core.dur : c.dur, date: dates[c.video], impure: IMPURE.has(k), head: IMPURE_HEAD.has(k), both: IMPURE_BOTH.has(k), want: DELIBERATE.has(k), prov: PROVISIONAL.has(k) }; }); if (SUSPECT.size) console.log(`suspect sources: ${SUSPECT.size} flagged; those above ${SUSPECT_F0}Hz dropped as the other speaker`); { const sliced = new Set(good.filter((c) => c.noTrim).map(akey)); good = good.filter((c) => c.noTrim || !sliced.has(akey(c))); } { const seen = new Set(); good = good.filter((c) => { const k = akey(c); if (seen.has(k)) return false; seen.add(k); return true; }); } // ---- build note lists ------------------------------------------------------- const TIME_LIMIT = Number(process.env.TIME_LIMIT ?? Infinity); const voiceNotes = VOICES.map((v) => { const lead = JSON.parse(readFileSync(path.join(SONG_DATA, v.lead), "utf8")) .filter((n) => n.t < TIME_LIMIT); const notes = []; for (let i = 0; i < lead.length; i += 1) { const n = lead[i], next = lead[i + 1]; const slotStart = n.t * SCALE; const rawIoi = (next ? next.t - n.t : n.dur) * SCALE; notes.push({ i, midi: n.midi, name: n.name, slotStart, slotDur: Math.max(0.09, Math.min(rawIoi, MAX_IOI)), noteDur: Math.min(n.dur * SCALE, MAX_IOI) }); } // A note next to a rest can tolerate a clip carrying a trace of a neighbouring // word, because a rest is there anyway. Which SIDE matters: "and um" wants a // phrase START (the stray word leads into the run) and "um and" wants a phrase // END. Mid-phrase notes can take neither. for (let i = 0; i < notes.length; i += 1) { const n = notes[i], nx = notes[i + 1], pv = notes[i - 1]; const gapAfter = nx ? nx.slotStart - (n.slotStart + n.noteDur) : Infinity; const gapBefore = pv ? n.slotStart - (pv.slotStart + pv.noteDur) : Infinity; n.phraseEnd = gapAfter > 0.12; n.phraseStart = gapBefore > 0.12; } return notes; }); // ---- assign SOURCES to voices by pitch ------------------------------------- // A bass part wants the deepest voices; give each part whole source videos so a // part keeps one timbre and one face. const bySource = new Map(); for (const c of good) { if (!bySource.has(c.video)) bySource.set(c.video, []); bySource.get(c.video).push(c); } const sources = [...bySource.entries()] .map(([v, cs]) => ({ v, n: cs.length, midi: med(cs.map((c) => c.midi)) })) .sort((a, b) => a.midi - b.midi); // lowest-voiced first const need = VOICES.map((v, i) => voiceNotes[i].length); const order = VOICES.map((v, i) => ({ i, name: v.name, need: need[i] })); // bass takes from the low end, arp from the high end, melody the middle const lowFirst = order.filter((o) => o.name === "bass"); const highFirst = order.filter((o) => o.name === "arp"); const mid = order.filter((o) => o.name !== "bass" && o.name !== "arp"); const pools = new Map(VOICES.map((v) => [v.name, []])); const takeFrom = (list, target, want) => { let got = 0; while (list.length && got < want) { const s = list.shift(); pools.get(target).push(...bySource.get(s.v)); got += s.n; } }; const claimed = new Set(); const claim = (arr, target, want) => { let got = 0; for (const s of arr) { if (claimed.has(s.v) || got >= want) continue; claimed.add(s.v); pools.get(target).push(...bySource.get(s.v)); got += s.n; } return got; }; // Accompaniment may reuse, so it needs a pool smaller than its note count. The // melody is the voice that must stay unique, so it is claimed FIRST and takes // the middle of the pitch range -- claiming it last left it 85 clips for 140 // notes and a p90 shift of 7.7 semitones. // Pitch-constrained voices claim FIRST: the bass can only use deep-voiced // sources, while the melody can use almost anything. A unique voice needs a pool // at least as large as its note count, so ask for 1.2x; a reusing voice needs // less. Claiming the melody first starved the bass (65 clips for 151 notes). const wantFor = (o) => { const v = VOICES[o.i]; return Math.ceil(o.need * (v.poolFactor ?? (v.unique ? 1.2 : 0.8))); }; for (const o of lowFirst) claim([...sources], o.name, wantFor(o)); for (const o of highFirst) claim([...sources].reverse(), o.name, wantFor(o)); // Bound these too. An unbounded claim is only safe while "mid" holds ONE voice: // with a second mid part (a rhythm line alongside the melody) the first one // swallowed every remaining source and the second got an empty pool. The // surplus still goes to mid[0] -- the melody -- in the sweep just below. for (const o of mid) claim([...sources], o.name, wantFor(o)); // anything still unclaimed goes to the melody for (const s of sources) if (!claimed.has(s.v)) { claimed.add(s.v); pools.get(mid[0].name).push(...bySource.get(s.v)); } // A single global transposition, chosen so the ensemble as a whole sits in his // register. Individual voices then move by whole octaves from here. const allMidi = voiceNotes.flat().map((n) => n.midi); const allPool = med([...bySource.values()].flat().map((c) => c.midi)); let GLOBAL_T = 0, bestG = Infinity; for (let t = -48; t <= 0; t += 1) { const cost = allMidi.reduce((a, m) => { const d = m + t - allPool; return a + Math.abs(d - 12 * Math.round(d / 12)); // octave-folded distance }, 0); if (cost < bestG) { bestG = cost; GLOBAL_T = t; } } console.log(`global transpose ${GLOBAL_T} (voices differ by whole octaves only)`); // ---- per-voice matching ----------------------------------------------------- const out = []; VOICES.forEach((v, vi) => { const notes = voiceNotes[vi]; let S = pools.get(v.name).slice() .sort((a, b) => a.date.localeCompare(b.date) || a.video.localeCompare(b.video) || a.start - b.start); // A voice can ask for a particular TOKEN, and to skip deliberate keeps. // A melody that wanders between "um" and "uh" reads as an unsteady vowel // rather than a tune -- the vowel changes on every note, so the line stops // sounding like one instrument. Keeps carry a stray word ("and", "the") which // is worse again on a melody. // This is a PREFERENCE, not a gate: if the preferred clips cannot cover the // notes, the rest top it up, because an unfilled voice is worse than a mixed // vowel. The log says how much of the pool ended up preferred. // With tokenPenalty set, the TOKEN is scored rather than gated: "um" wins a // tie but a much better pitch match can still buy an "uh". A hard filter costs // pitch that cannot be bought back any other way -- on Yoshi it took the melody // from p90 1.59 to 4.47 at full length -- so the penalty is the setting to // reach for when the tune has to be long AND in tune. const TOKEN_PENALTY = Number(v.tokenPenalty ?? 0); const ATTACK_PENALTY = Number(v.attackPenalty ?? 0); const PLOSIVE_BONUS = Number(v.plosiveBonus ?? 0); const tokenWanted = v.tokens ? new Set(v.tokens.map((t) => t.toLowerCase())) : null; const tokenBad = (c) => !!tokenWanted && !tokenWanted.has(wordOf(c).toLowerCase().trim()); if ((v.tokens && !TOKEN_PENALTY) || v.noKeeps) { const allow = v.tokens ? new Set(v.tokens.map((t) => t.toLowerCase())) : null; const ok = (c) => (!allow || TOKEN_PENALTY || allow.has(wordOf(c).toLowerCase().trim())) && (!v.noKeeps || !c.want); const pref = S.filter(ok), rest = S.filter((c) => !ok(c)); const target = Math.ceil(notes.length * (v.poolFactor ?? (v.unique ? 1.2 : 0.8))); S = pref.length >= target ? pref : pref.concat(rest.slice(0, target - pref.length)); console.log(` ${v.name}: prefer ${JSON.stringify(v.tokens ?? "any")}${v.noKeeps ? " +noKeeps" : ""}` + ` -> ${pref.length} preferred of ${pref.length + rest.length}, pool ${S.length}` + (pref.length < target ? ` (topped up with ${S.length - pref.length})` : "")); } if (!S.length) { console.error(`voice ${v.name}: EMPTY pool`); out.push({ ...v, plan: [] }); return; } // ONE transposition for the whole piece, then per-voice OCTAVE offsets only. // Transposing each voice independently put the melody and the pad 6 semitones // apart -- a tritone -- so the parts played in unrelated keys and the tune // dissolved into noise. Octaves preserve harmony; anything else does not. const centre = med(S.map((s) => s.midi)); const oct = Math.round((med(notes.map((n) => n.midi)) + GLOBAL_T - centre) / 12); const T = GLOBAL_T - 12 * oct; const used = new Set(), lastUsed = new Map(); const plan = []; for (let k = 0; k < notes.length; k += 1) { const n = notes[k]; const base = n.midi + T; const p = Math.floor((k * S.length) / notes.length); let pick = -1, pickScore = Infinity, pickOct = 0; for (const phase of [0, 1, 2]) { for (let j = Math.max(0, p - WINDOW); j < Math.min(S.length, p + WINDOW); j += 1) { const isUsed = used.has(j); if (isUsed && (phase < 2 || v.unique)) continue; if (isUsed && (k - (lastUsed.get(j) ?? -1e9)) < v.reuseGap) continue; const s = S[j]; const d = base - s.midi; // Per-voice folding: a bass part may fold freely (its job is the root, not // the shape), the melody must keep its contour. const cmax = v.contourMax ?? CONTOUR_MAX; const kOct = Math.abs(d) > cmax ? Math.round(d / 12) : 0; const shift = Math.abs(d - 12 * kOct); if (phase === 0 && shift > 3) continue; const need2 = Math.min(n.slotDur, n.noteDur * 1.35); // Impure clips are penalised mid-phrase and free at a phrase end. // free where its stray word is harmless, penalised everywhere else. A // clip kept ON PURPOSE for its stray word ("and um") is better than // neutral at the matching boundary -- it is the emphasis you want there. // a word on BOTH sides only works where the note is isolated const okHere = s.both ? (n.phraseStart && n.phraseEnd) : s.head ? n.phraseStart : n.phraseEnd; const impurity = s.impure && !okHere ? IMPURE_PENALTY : (s.want && okHere ? -DELIBERATE_BONUS : 0); // A clip whose pitch glides inside itself cannot hold a steady note: one // with 277 cents of drift sounded audibly off even at zero shift. Push // wobbly clips away from notes, especially long ones where the glide has // time to show. const wobble = Math.max(0, (s.spreadCents ?? 0) - WOBBLE_OK) * WOBBLE_W * (1 + n.slotDur); // Nothing used to penalise a clip much LONGER than its note -- fine for a // plain um, where the tail is expendable, but a KEEP clip is a compound // sound that cannot be cut down without losing its point. Steer those to // notes long enough to hold them. const tooLongForKeep = s.want ? Math.max(0, s.dur - n.slotDur * 1.05) * 14 : 0; // A smooth onset costs; a strong burst earns a discount. Both are scored, // never gated -- see the note where ATTACK is loaded for why the plosive // set in particular can only ever be a preference. const attack = (ATTACK_PENALTY && !isPercussive(s) ? ATTACK_PENALTY : 0) - (PLOSIVE_BONUS && isPlosive(s) ? PLOSIVE_BONUS : 0); const score = shift * SHIFT_W + Math.max(0, need2 - s.dur) * 6 + tooLongForKeep + (TOKEN_PENALTY && tokenBad(s) ? TOKEN_PENALTY : 0) + Math.abs(j - p) * LOCAL + impurity + wobble + attack + (s.prov ? PROV_PENALTY : 0); if (score < pickScore) { pickScore = score; pick = j; pickOct = kOct; } } if (pick >= 0) break; } if (pick < 0) continue; used.add(pick); lastUsed.set(pick, k); const s = S[pick]; const target = base - 12 * pickOct; plan.push({ note: n.i, name: n.name, targetMidi: +target.toFixed(2), slotStart: +n.slotStart.toFixed(4), slotDur: +n.slotDur.toFixed(4), noteDur: +n.noteDur.toFixed(4), // Carried through so anything placing sound OVER the arrangement (a vocal // hook) can find a real phrase start instead of re-deriving the rule and // letting the two definitions drift. phraseStart: !!n.phraseStart, phraseEnd: !!n.phraseEnd, video: s.video, date: s.date, srcStart: s.start, srcEnd: s.end, srcDur: s.dur, // 1 leading word, 2 trailing word, 3 both -- the renderer must not shave // the edge that holds the word, since the word is the point of the clip. keepSide: s.want ? (s.both ? 3 : s.head ? 1 : 2) : 0, // How far into the clip the leading word ends. The UM has to land on the // beat, not the word: measured across the kept clips the word occupies the // first 240ms on average (up to 529ms), so placing the clip at the slot // start would drop the note a quarter-second late. The word becomes a // pickup INTO the beat instead. headOffset: s.want && s.head ? +(HEAD_OFFSET.get(akey(s)) ?? 0).toFixed(4) : 0, srcF0: s.f0, srcMidi: +s.midi.toFixed(2), shift: +(target - s.midi).toFixed(3), spreadCents: s.spreadCents, noTrim: !!s.noTrim }); } const sh = plan.map((x) => Math.abs(x.shift)).sort((a, b) => a - b); const q = (f) => sh[Math.floor(f * (sh.length - 1))] ?? 0; const uniq = new Set(plan.map((x) => `${x.video}@${x.srcStart}`)).size; console.log(`${v.name.padEnd(7)} notes ${String(plan.length).padStart(4)}/${notes.length}` + ` pool ${String(S.length).padStart(3)} from ${new Set(S.map((s) => s.video)).size} src` + ` unique ${uniq} transpose ${T} |shift| p50 ${q(0.5).toFixed(2)} p90 ${q(0.9).toFixed(2)} max ${q(1).toFixed(2)}`); // carry per-voice render flags through: without `fill` the renderer defaulted // every voice to legato and the picture-in-picture never had a gap to blank out.push({ name: v.name, gain: v.gain, role: v.role, fill: v.fill !== false, transpose: T, plan }); }); const total = out.reduce((a, v) => a + v.plan.length, 0); const allKeys = out.flatMap((v) => v.plan.map((p) => `${p.video}@${p.srcStart}`)); console.log(`total notes ${total}; distinct clips ${new Set(allKeys).size}`); writeFileSync(path.join(SONG_DATA, "poly-plan.json"), JSON.stringify({ scale: SCALE, voices: out }, null, 1));