#!/usr/bin/env node // Audit a finished plan for the things that get REPORTED BY EAR, and name the // clip and the song time for each -- so a complaint like "static at 15s" turns // into a clip id that `swap-clip.mjs` can replace. // // The by-ear reports on Yoshi could never be traced, because they were made // against a build whose plan had already been overwritten by the next arrange, // and the arranger reshuffles every pick after any change. Chasing those // timestamps in a new plan is meaningless. What transfers is the KIND of fault, // and each kind has a measurable signature: // // OTHER SPEAKER f0 above SUSPECT_F0. The corpus is median 103Hz / p95 127Hz // and the interviews that carry a guest split into a ~79-93Hz // (Jer) and a 125-218Hz (guest) cluster. arrange-poly already // gates FLAGGED videos this way; this looks at every pick, // including videos nobody has flagged yet. // WRONG WORD the clip is not an um at all ("the", "as", "and"). Every // candidate carries an ASR token, and words.json carries the // human label where one exists. // STATIC / HISS high spectral flatness. Nothing scores this, and // render-poly normalises every clip to TARGET_RMS -- so the // quietest, hissiest clip gets the LARGEST gain and its noise // is amplified into the mix. That is exactly what made one // Yoshi render audibly hiss (flatness 0.10 at source, x4.29). // // node plan-qa.mjs [--top N] import { readFileSync, readdirSync } from "node:fs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; import { readWavMono, spectralFlatness } from "./flatness.mjs"; const DIR = path.resolve(path.dirname(new URL(import.meta.url).pathname)); const PLAN = process.argv[2]; const TOP = Number((process.argv.find((a) => a.startsWith("--top")) ?? "--top=12").split("=")[1] ?? 12); const F0_MAX = Number(process.env.SUSPECT_F0 ?? 115); const FLAT_MAX = Number(process.env.FLAT_MAX ?? 0.055); const plan = JSON.parse(readFileSync(PLAN, "utf8")); const key = (v, s) => `${v}@${(+s).toFixed(2)}`; // candidate metadata, keyed the same way the palette is const cand = new Map(); for (const f of readdirSync(path.join(SONG_DATA, "cand2")).filter((f) => f.endsWith(".json"))) { for (const c of JSON.parse(readFileSync(path.join(SONG_DATA, "cand2", f), "utf8")).candidates) { const k = key(c.video, c.start); if (!cand.has(k)) cand.set(k, c); } } let CORE = {}, WORDS = {}, SUSPECT = new Set(); try { CORE = JSON.parse(readFileSync(path.join(DIR, "corepitch.json"), "utf8")); } catch {} try { WORDS = JSON.parse(readFileSync(path.join(DIR, "words.json"), "utf8")); } catch {} try { const S = JSON.parse(readFileSync(path.join(DIR, "suspect-sources.json"), "utf8")); SUSPECT = new Set((Array.isArray(S) ? S : S.sources ?? []).map((s) => (typeof s === "string" ? s : s.k))); } catch {} // ---- spectral flatness, measured on the clip itself ------------------------- // The measure lives in flatness.mjs, shared with swap-clip.mjs and the triage // app. Keeping one copy is the point: this tool NAMES a clip as hissy and // swap-clip.mjs then has to agree that its replacement is not, and two drifting // copies of the same FFT would make that conversation meaningless. const readWav = (file) => readWavMono(file, readFileSync); const flatness = spectralFlatness; const srcCache = new Map(); const rows = []; for (const v of plan.voices) { for (const p of v.plan) { const k = key(p.video, p.srcStart); const c = cand.get(k); const f0 = CORE[k]?.f0 ?? c?.f0 ?? 0; const token = (WORDS[k]?.word ?? c?.token ?? "").toLowerCase().trim(); rows.push({ voice: v.name, t: p.slotStart, k, f0, token, suspect: SUSPECT.has(k), srcStart: +p.srcStart, srcEnd: +p.srcEnd, video: p.video, keepSide: p.keepSide ?? 0 }); } } rows.sort((a, b) => a.t - b.t); console.log(`${path.basename(PLAN)}: ${rows.length} notes, ${new Set(rows.map((r) => r.k)).size} distinct clips`); // ---- other speaker ---------------------------------------------------------- const hi = rows.filter((r) => r.f0 > F0_MAX).sort((a, b) => b.f0 - a.f0); console.log(`\nOTHER SPEAKER -- picks above ${F0_MAX}Hz: ${hi.length}`); for (const r of hi.slice(0, TOP)) { console.log(` t=${r.t.toFixed(2).padStart(7)}s ${r.voice.padEnd(7)} ${r.k.padEnd(26)} f0 ${r.f0.toFixed(1)}Hz` + (r.suspect ? " [flagged source]" : "")); } // ---- wrong word ------------------------------------------------------------- const OK = new Set(["um", "uh", "hmm", "er", "erm", "mm", "uhh", "umm"]); const bad = rows.filter((r) => r.token && !OK.has(r.token)); const byTok = new Map(); for (const r of bad) byTok.set(r.token, (byTok.get(r.token) ?? 0) + 1); console.log(`\nWRONG WORD -- picks whose token is not a filler: ${bad.length}` + (byTok.size ? ` (${[...byTok.entries()].sort((a, b) => b[1] - a[1]).map(([t, n]) => `"${t}" x${n}`).join(", ")})` : "")); for (const r of bad.slice(0, TOP)) { console.log(` t=${r.t.toFixed(2).padStart(7)}s ${r.voice.padEnd(7)} ${r.k.padEnd(26)} "${r.token}"` + (r.keepSide ? ` [deliberate keep, side ${r.keepSide}]` : "")); } // ---- static / hiss ---------------------------------------------------------- if (process.env.NO_FLATNESS !== "1") { const uniq = [...new Map(rows.map((r) => [r.k, r])).values()]; for (const r of uniq) { try { if (!srcCache.has(r.video)) { if (srcCache.size > 2) srcCache.clear(); srcCache.set(r.video, readWav(path.join(SONG_DATA, "wav48", `${r.video}.wav`))); } const { x, sr } = srcCache.get(r.video); const a = Math.max(0, Math.round(r.srcStart * sr)), b = Math.min(x.length, Math.round(r.srcEnd * sr)); r.flat = flatness(Float32Array.prototype.slice.call(x, a, b), sr); } catch { r.flat = 0; } } uniq.sort((a, b) => (b.flat ?? 0) - (a.flat ?? 0)); const hissy = uniq.filter((r) => (r.flat ?? 0) > FLAT_MAX); console.log(`\nSTATIC / HISS -- spectral flatness above ${FLAT_MAX}: ${hissy.length} of ${uniq.length} clips`); for (const r of uniq.slice(0, TOP)) { console.log(` t=${r.t.toFixed(2).padStart(7)}s ${r.voice.padEnd(7)} ${r.k.padEnd(26)} flatness ${r.flat.toFixed(4)}` + ((r.flat ?? 0) > FLAT_MAX ? " <-- over" : "")); } }