// THE window for a clip: what the review page shows and what the renderer // sounds. These were separate implementations and drifted -- the page cleared // neighbouring words, the renderer did not, so notes played words nobody had // approved (a plain "the" ended one render of Yoshi's Island). One function now, // used by both, so "what you approved" and "what plays" cannot diverge again. import { yinFrame } from "./pitch.mjs"; // Default window = clear of any word parakeet timestamps inside it, then the // steady-pitch core, which is what the renderer actually sounds. // Tunable, because these were fitted to real hand-set windows rather than // guessed: see the sweep in fit-window.mjs. const NUM = (name, dflt) => Number(process.env[name] ?? dflt); export function clipWindow(x, sr, c, words) { let s0 = c.start, e0 = c.end; // How far the window may grow OUTSIDE Whisper's detected boundaries. Clips // tagged "clipped too tight" were keeping 95-99% of the candidate, so the // candidate itself was short -- growing only inside it could never fix them. // The guards are asymmetric because parakeet timestamps a word's start LATE: // every clip tagged "next word bleeds in" already ended before the recorded // start, by 25-290ms, so the audible onset precedes the timestamp. const MAX_EXT = NUM("WIN_MAX_EXT", 0.25), GUARD_PREV = NUM("WIN_GUARD_PREV", 0.04), GUARD_NEXT = NUM("WIN_GUARD_NEXT", 0.07); let sLim = c.start - MAX_EXT, eLim = c.end + MAX_EXT; for (const w of words) { if (w.end <= c.start + 0.02) sLim = Math.max(sLim, w.end + GUARD_PREV); if (w.start >= c.end - 0.02) { eLim = Math.min(eLim, w.start - GUARD_NEXT); break; } } for (const w of words) { if (w.end <= s0 + 0.02 || w.start >= e0 - 0.02) continue; if (w.start <= s0 + 0.02) s0 = Math.max(s0, w.end + 0.015); if (w.end >= e0 - 0.02) e0 = Math.min(e0, w.start - 0.015); // A word lying ENTIRELY inside the window matched neither test above, so it // was left in the clip -- the "a" of "and", the "t" of "the". Nine of the // unclean pile ran up to 245ms past a word's own start because of this. // Keep whichever SIDE of the word is longer: preferring the earlier side // unconditionally isolated the "the" out of "the ummmm" when Whisper's // window opened a hair before the word. if (w.start > s0 + 0.02 && w.end < e0 - 0.02) { const beforeLen = (w.start - 0.015) - s0; const afterLen = e0 - (w.end + 0.015); if (afterLen > beforeLen && afterLen >= 0.14) s0 = Math.max(s0, w.end + 0.015); else if (beforeLen >= 0.14) e0 = Math.min(e0, w.start - 0.015); else if (afterLen >= 0.14) s0 = Math.max(s0, w.end + 0.015); } } if (e0 - s0 < 0.12) { s0 = c.start; e0 = c.end; } const a = Math.max(0, Math.round(s0 * sr)), b = Math.min(x.length, Math.round(e0 * sr)); const seg = x.subarray(a, b); const W = 2048, hop = Math.round(sr * 0.01); const nF = Math.floor((seg.length - W) / hop); if (nF >= 4) { const f = [], cf = []; for (let i = 0; i < nF; i += 1) { const r = yinFrame(seg, i * hop, W, sr, 55, 300); f.push(r.f0); cf.push(r.conf); } const ok = f.filter((v, i) => v > 0 && cf[i] >= 0.5).sort((u, v) => u - v); if (ok.length >= 3) { const mid = ok[ok.length >> 1]; const near = (i) => f[i] > 0 && cf[i] >= 0.5 && Math.abs(1200 * Math.log2(f[i] / mid)) <= 90; let bi = 0, bj = -1, i = 0; while (i < nF) { if (!near(i)) { i += 1; continue; } let j = i; while (j + 1 < nF && near(j + 1)) j += 1; if (j - i > bj - bi) { bi = i; bj = j; } i = j + 1; } if (bj > bi) { let cs = s0 + (bi * hop) / sr, ce = s0 + (bj * hop + W) / sr; if (ce - cs >= 0.12) { // The core alone threw away half the um: YIN's confidence collapses // during the natural onset and decay, so the run ended early at both // edges (median 237ms lost, and on 51% of approved clips under half // the sound survived). Grow the core back out over audio that is still // clearly sounding, stopping at the word-cleared bounds -- so the // attack and release come back but a neighbouring word cannot. // Index the SOURCE, not the windowed slice: growth now runs past the // old window on both sides, which a slice-relative probe cannot see. const env = (t) => { const i0 = Math.max(0, Math.round(t * sr)), n = Math.round(sr * 0.005); let s2 = 0, cnt = 0; for (let k = i0; k < i0 + n && k < x.length; k += 1) { s2 += x[k] * x[k]; cnt += 1; } return cnt ? Math.sqrt(s2 / cnt) : 0; }; let peak = 0; for (let t = cs; t < ce; t += 0.005) peak = Math.max(peak, env(t)); // 0.18 was a guess and it cut the decay short: measured against 127 // hand-set windows the end landed a median 50ms early. Sweeping the // floor against those windows puts the best fit at 0.07 (median error // 35ms); below that it overshoots the sound and gets worse again. const FLOOR = NUM("WIN_FLOOR", 0.07) * peak; // Grow out to the WORD-derived limits, which may lie outside Whisper's // candidate: the detection is frequently narrower than the sound. const limS = Math.max(sLim, Math.min(s0, c.start - MAX_EXT)); const limE = Math.min(eLim, Math.max(e0, c.end + MAX_EXT)); while (cs - 0.005 >= limS && env(cs - 0.005) > FLOOR) cs -= 0.005; while (ce + 0.005 <= limE && env(ce) > FLOOR) ce += 0.005; s0 = cs; e0 = Math.min(limE, ce); } } } } return { s: s0, e: e0 }; }