#!/usr/bin/env node // Replace ONE note's clip in a finished plan, without re-arranging. // // Re-arranging to fix a single clip is the wrong tool: the arranger walks the // pool in order, so removing one candidate reshuffles every pick after it and a // render you were otherwise happy with comes back different everywhere. // // node swap-clip.mjs [minDur] // // Picks the best-scoring accepted clip that is NOT already in the plan: closest // in pitch, comfortably longer than the note, steady (low wobble), and plain // (no deliberately-kept stray word, which is what makes a clip stand out). // // It also has to avoid replacing a fault with the same fault. The two things // actually complained about by ear are ANOTHER SPEAKER and STATIC, and neither // was screened here: the pitch window ran to 200Hz (the guest cluster in the // interview sources is 125-218Hz) and nothing looked at noise at all, while // render-poly normalises every clip to TARGET_RMS -- so the hissiest candidate // gets the largest gain and its noise is amplified into the mix. // // F0_MAX= reject candidates above this (default 200; 115 is the gate // arrange-poly uses to separate a guest from Jer) // FLAT_MAX= reject candidates whose spectral flatness exceeds this. // Measured lazily, best-scoring candidate first, because it // is an FFT per clip and the pool is thousands of clips. import { readFileSync, writeFileSync, readdirSync } from "node:fs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; import { readWavMono, spectralFlatness } from "./flatness.mjs"; const DIR = path.resolve(path.dirname(new URL(import.meta.url).pathname)); const [PLANF, VOICE, SLOT] = process.argv.slice(2); const MINDUR = Number(process.argv[5] ?? 0.30); const F0_MAX = Number(process.env.F0_MAX ?? 200); const FLAT_MAX = Number(process.env.FLAT_MAX ?? Infinity); const midiOf = (hz) => 69 + 12 * Math.log2(hz / 440); const plan = JSON.parse(readFileSync(PLANF, "utf8")); const v = plan.voices.find((x) => x.name === VOICE); const idx = v.plan.findIndex((p) => Math.abs(p.slotStart - Number(SLOT)) < 0.02); if (idx < 0) throw new Error(`no ${VOICE} note at ${SLOT}s`); const note = v.plan[idx]; console.log(`replacing ${VOICE} @${note.slotStart}s ${note.video}@${note.srcStart}` + ` target midi ${note.targetMidi} noteDur ${note.noteDur}s slotDur ${note.slotDur}s shift ${note.shift}`); // cand2/ is BULK data and lives under SONG_DATA; the small judged-state files // (accepted/corepitch/impure) are durable and live next to the code. Same split // arrange-poly.mjs makes -- getting it wrong is invisible until a path throws. let all = []; for (const f of readdirSync(path.join(SONG_DATA, "cand2")).filter((f) => f.endsWith(".json"))) all.push(...JSON.parse(readFileSync(path.join(SONG_DATA, "cand2", f), "utf8")).candidates); const acc = new Set(JSON.parse(readFileSync(path.join(DIR, "accepted.json"), "utf8")).accepted); let CORE = {}; try { CORE = JSON.parse(readFileSync(path.join(DIR, "corepitch.json"), "utf8")); } catch {} let IMPURE = new Set(); try { IMPURE = new Set(JSON.parse(readFileSync(path.join(DIR, "impure.json"), "utf8"))); } catch {} // Clips a human has already judged to be ANOTHER SPEAKER, and the sources they // came from. This screen was missing, and its absence did exactly what this // script's header warns against: asked to replace a clip whose window contained // the words "has been", it chose vfMRAdhSSrc@473.60 -- a key sitting in // suspect-sources.json under "a clip here was judged other speaker, not Jer". // One reported fault swapped for another. // // The KEY screen is unconditional. The SOURCE screen is the same rule // arrange-poly applies (a flagged source's clips above SUSPECT_F0 are dropped), // so the two agree about what is admissible. const SUSPECT_F0 = Number(process.env.SUSPECT_F0 ?? 115); let SUSPECT_KEYS = new Set(); let SUSPECT_SRC = new Set(); try { const S = JSON.parse(readFileSync(path.join(DIR, "suspect-sources.json"), "utf8")); SUSPECT_KEYS = new Set((Array.isArray(S) ? S : S.sources ?? []).map((s) => (typeof s === "string" ? s : s.k))); SUSPECT_SRC = new Set([...SUSPECT_KEYS].map((k) => String(k).split("@")[0])); } catch {} const key = (c) => `${c.video}@${(+c.start).toFixed(2)}`; const seen = new Set(); all = all.filter((c) => acc.has(key(c)) && !seen.has(key(c)) && seen.add(key(c))); // every clip already sounding anywhere in this plan const used = new Set(plan.voices.flatMap((x) => x.plan.map((p) => `${p.video}@${p.srcStart}`))); // ---- spectral flatness, for the STATIC case -------------------------------- // The measure itself lives in flatness.mjs, shared with plan-qa.mjs and the // triage app. It was written twice in one afternoon, and two copies of a // MEASURE is worse than two copies of a helper: when they drift, one tool calls // a clip hissy and the other calls it fine with nothing to say which is right. const wavCache = new Map(); function flatnessOf(c) { try { if (!wavCache.has(c.video)) { if (wavCache.size > 2) wavCache.clear(); wavCache.set(c.video, readWavMono(path.join(SONG_DATA, "wav48", `${c.video}.wav`), readFileSync)); } const { x, sr } = wavCache.get(c.video); const a = Math.max(0, Math.round(c.start * sr)), b = Math.min(x.length, Math.round(c.end * sr)); return spectralFlatness(Float32Array.prototype.slice.call(x, a, b), sr); } catch { return 0; } } // ---- a real word inside the window ----------------------------------------- // The other fault reported by ear, and the one nothing screened for. A clip's // stored token can say "uh" while its WINDOW contains the words either side of // it -- 27 of the 615 notes in the Pokemon plan are like this, and the one a // listener picked out ("has been", heard as "then uh") was the melody note at // 11.687s. Without this check a swap can trade an audible word for another. // ALLOW_WORDS=1 turns it off. const ALLOW_WORDS = process.env.ALLOW_WORDS === "1"; /** Room to a neighbouring word past which a clip is not considered cramped. */ const ROOMY = Number(process.env.ROOMY ?? 0.25); // ---- percussive onsets ------------------------------------------------------- // Reported by ear on the caught jingle: the notes "blend together" rather than // each one starting distinctly. attack.mjs already scores exactly this -- a // stopped onset ("d'ummm") against a smooth one ("ummm") -- and nothing had ever // used the file. // // ATTACK_W>0 pushes smooth onsets DOWN the ranking. It is a preference, like // ROOMY, and for the same reason: attack.json covers 700 clips of the 3,959 // accepted (it was scored before the palette grew), so a hard gate would throw // away 82% of the corpus for want of a measurement rather than for a fault. // REQUIRE_ATTACK=1 makes it a filter for the cases where distinctness matters // more than choice -- an 18-note finale can afford that; an 800-note tune cannot. const ATTACK_W = Number(process.env.ATTACK_W ?? 0); const REQUIRE_ATTACK = process.env.REQUIRE_ATTACK === "1"; let ATTACK = {}; if (ATTACK_W > 0 || REQUIRE_ATTACK) { try { ATTACK = JSON.parse(readFileSync(path.join(DIR, "attack.json"), "utf8")); } catch {} console.log(` attack scores loaded for ${Object.keys(ATTACK).length} clips` + (REQUIRE_ATTACK ? " (REQUIRED)" : ` (weight ${ATTACK_W})`)); } /** Higher = more percussive. burst is the onset's energy jump, riseMs its rise time. */ function attackScore(k) { const a = ATTACK[k]; if (!a) return null; return (a.burst ?? 1) - (a.riseMs ?? 20) / 20; } // FORCE names one clip outright, bypassing the ranking and its gates. The finale // is 18 notes and gets to be chosen rather than scored -- e.g. putting an // affirmative on the note that resolves, where the Pokemon is caught. const FORCE = process.env.FORCE ?? ""; // FORCE_F0 supplies a pitch for a clip the palette never measured. The two-syllable // clips ("and um") all carry f0 0 with no corepitch entry, which is the very reason // the arranger can never pick one -- it matches on pitch. Measure the SUSTAINED // second syllable (measure-and.mjs) and pass it here; the first syllable is a // consonant onset and has no pitch to speak of. const FORCE_F0 = Number(process.env.FORCE_F0 ?? 0); const asrCache = new Map(); function wordsOf(video) { let w = asrCache.get(video); if (!w) { try { w = JSON.parse(readFileSync(path.join(SONG_DATA, "asr", `${video}.json`), "utf8")).words ?? []; } catch { w = []; } if (asrCache.size > 24) asrCache.clear(); asrCache.set(video, w); } return w; } function wordsInside(c) { const w = wordsOf(c.video); // The same 30ms tolerance the plan scan uses, so the two agree about which // clips are dirty -- a screen that disagreed with the report naming the fault // would be worse than none. return w.filter((x) => x.end > c.start + 0.03 && x.start < c.end - 0.03).map((x) => x.w); } /** Seconds to the nearest ASR word on either side, or ROOMY if there is none. */ function nearestWordGap(c) { let w; try { w = wordsOf(c.video); } catch { return ROOMY; } let best = ROOMY; for (const x of w) { if (x.end <= c.start) best = Math.min(best, c.start - x.end); else if (x.start >= c.end) best = Math.min(best, x.start - c.end); } return best; } const ranked = []; let skippedWords = 0; let skippedSuspect = 0; for (const c of all) { const k = key(c); if (used.has(`${c.video}@${c.start}`)) continue; if (IMPURE.has(k)) continue; // plain clips only if (!ALLOW_WORDS && wordsInside(c).length) { skippedWords += 1; continue; } if (SUSPECT_KEYS.has(k)) { skippedSuspect += 1; continue; } if (SUSPECT_SRC.has(c.video) && (CORE[k] ? CORE[k].f0 : c.f0) > SUSPECT_F0) { skippedSuspect += 1; continue; } const core = CORE[k]; const dur = core ? core.dur : c.dur; const f0 = core ? core.f0 : c.f0; const midi = midiOf(f0); if (dur < MINDUR) continue; // the whole point: long enough if (!(c.f0 >= 70 && c.f0 <= F0_MAX)) continue; if (f0 > F0_MAX) continue; // measured pitch, not just the candidate's const shift = Math.abs(note.targetMidi - midi); if (shift > 3) continue; const wobble = Math.max(0, (c.spreadCents ?? 0) - 120) * 0.03; // How much ROOM the clip has before the nearest neighbouring word. // // A third fault, distinct from a word inside the window: the melody note at // 1:07 was heard as "and uh" while its window contains no word at all -- // "and" ends 177ms before the clip starts, and parakeet times word ends // early, so its tail bleeds in. // // It is a PREFERENCE, not a filter, and that is a measured decision rather // than caution: 39.5% of the accepted palette sits under 177ms and 23% under // 100ms, so any threshold tight enough to catch this clip would reject most // of the corpus. Nor does the stored marginBefore help -- 96% of accepted // clips are under 0.20s there, because the miner cuts close to the vowel by // construction. So tight clips are pushed DOWN the ranking and still // reachable when nothing roomier fits. const gap = nearestWordGap(c); const cramped = Math.max(0, ROOMY - gap) * 48; // prefer a clip with room to spare over one that only just fits const room = Math.max(0, (note.noteDur * 1.35) - dur) * 8; // A smooth onset costs; a stopped one is free. Unscored clips sit between the // two rather than being rejected, unless REQUIRE_ATTACK says otherwise. const asc = attackScore(k); if (REQUIRE_ATTACK && asc === null) continue; const blunt = ATTACK_W > 0 ? ATTACK_W * (asc === null ? 1.0 : Math.max(0, 1.6 - asc)) : 0; const score = shift * 12 + wobble + room + cramped + blunt; ranked.push({ c, midi, dur, shift, score, attack: asc }); } ranked.sort((a, b) => a.score - b.score); if (skippedWords) console.log(` screened out ${skippedWords} candidate(s) with a real word inside the window`); if (skippedSuspect) console.log(` screened out ${skippedSuspect} candidate(s) from sources judged to be another speaker`); let best = null; if (FORCE) { const c = all.find((x) => `${x.video}@${(+x.start).toFixed(2)}` === FORCE || `${x.video}@${x.start}` === FORCE); if (!c) throw new Error(`FORCE clip ${FORCE} is not in the palette`); const core = CORE[key(c)]; const dur = core ? core.dur : c.dur; const f0 = FORCE_F0 || (core ? core.f0 : c.f0); if (!(f0 > 0)) throw new Error(`${FORCE} has no measured pitch -- pass FORCE_F0= (see measure-and.mjs)`); const midi = midiOf(f0); best = { c, midi, dur, f0, shift: Math.abs(note.targetMidi - midi), attack: attackScore(key(c)) }; const label = (() => { try { return JSON.parse(readFileSync(path.join(DIR, "words.json"), "utf8"))[key(c)]?.word; } catch { return null; } })(); console.log(` FORCED ${FORCE} (token "${label ?? c.token ?? "?"}", ${dur.toFixed(3)}s, midi ${midi.toFixed(2)})`); } for (const cand of best ? [] : ranked) { if (FLAT_MAX === Infinity) { best = cand; break; } const f = flatnessOf(cand.c); if (f <= FLAT_MAX) { best = { ...cand, flat: f }; break; } console.log(` skipping ${cand.c.video}@${cand.c.start}: flatness ${f.toFixed(4)} > ${FLAT_MAX}`); } if (!best) throw new Error("no replacement found"); const { c, midi, dur, shift } = best; console.log(` -> ${c.video}@${c.start} midi ${midi.toFixed(2)} dur ${dur.toFixed(3)}s` + ` shift ${(note.targetMidi - midi).toFixed(2)} (was ${note.srcDur}s)`); v.plan[idx] = { ...note, video: c.video, srcStart: c.start, srcEnd: c.end, srcDur: +dur.toFixed(3), srcF0: best.f0 ?? c.f0, srcMidi: +midi.toFixed(2), shift: +(note.targetMidi - midi).toFixed(3), spreadCents: c.spreadCents, noTrim: !!c.noTrim, keepSide: 0, headOffset: 0 }; writeFileSync(PLANF, JSON.stringify(plan, null, 1)); console.log(`written to ${PLANF}`);