#!/usr/bin/env node // One review page, two modes, always written to the SAME file (um-review.html). // // MODE=sort (default) fast triage of clips nobody has judged yet. Three // verdicts, NO window editing -- the point is speed. Anything // that is nearly good but carries a neighbouring word goes to // "unclean" instead of being lost. // MODE=salvage work the unclean pile with the trim handles, when a target // needs more clips than the clean ones alone can cover. // // Rounds no longer each get their own page: the manifest is a growing UNION // keyed by clip, so an older paste-back still decodes after a rebuild. import { readFileSync, writeFileSync, mkdirSync, existsSync, rmSync, readdirSync } from "node:fs"; import { execFileSync } from "node:child_process"; import { yinFrame } from "./pitch.mjs"; import { scoreWith } from "./orderfeat.mjs"; import { stripTrailingPop } from "./deplosive.mjs"; import { clipWindow as defaultWindow } from "./clipwindow.mjs"; import { REASONS, OTHER_CODE, DELIBERATE_CODES } from "./reasons.mjs"; import { archiveMomentBase, LEAD_IN } from "./archive-url.mjs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const DIR = path.resolve(path.dirname(new URL(import.meta.url).pathname)); const OUT = process.argv[2] ?? path.join(SONG_DATA, "um-review.html"); // sort = unjudged clips | salvage = the unclean pile | verify = already-accepted // clips that the window/word changes may have altered under you const MODE = ["salvage", "verify", "keeps"].includes(process.env.MODE) ? process.env.MODE : "sort"; // Both salvage and verify hand back a WINDOW, so both need the handles and the // window-carrying export. Only the fast sort pass is verdict-only. // Every mode can adjust the window now -- there is no reason to make someone // defer a trim to a later pass just because this one is meant to be quick. // What still differs is what the verdicts MEAN: in sort, "unclean" parks a clip // for salvage rather than accepting it as deliberately dirty. const EDIT = true; const SR = 48000, PAD = 0.06; function readWav(file) { const b = readFileSync(file); let p = 12, fmt = null, data = null; while (p + 8 <= b.length) { const id = b.toString("latin1", p, p + 4); const sz = b.readUInt32LE(p + 4); const body = p + 8; if (id === "fmt ") fmt = { ch: b.readUInt16LE(body + 2), sr: b.readUInt32LE(body + 4) }; if (id === "data") { data = b.subarray(body, body + sz); break; } p = body + sz + (sz & 1); } const n = Math.floor(data.length / 2 / fmt.ch); const x = new Float32Array(n); for (let i = 0; i < n; i += 1) x[i] = data.readInt16LE(i * fmt.ch * 2) / 32768; return { x, sr: fmt.sr }; } function writeWav(file, x, sr) { const n = x.length, b = Buffer.alloc(44 + n * 2); b.write("RIFF", 0, "latin1"); b.writeUInt32LE(36 + n * 2, 4); b.write("WAVE", 8, "latin1"); b.write("fmt ", 12, "latin1"); b.writeUInt32LE(16, 16); b.writeUInt16LE(1, 20); b.writeUInt16LE(1, 22); b.writeUInt32LE(sr, 24); b.writeUInt32LE(sr * 2, 28); b.writeUInt16LE(2, 32); b.writeUInt16LE(16, 34); b.write("data", 36, "latin1"); b.writeUInt32LE(n * 2, 40); for (let i = 0; i < n; i += 1) b.writeInt16LE(Math.round(Math.max(-1, Math.min(1, x[i])) * 32767), 44 + i * 2); writeFileSync(file, b); } const dates = JSON.parse(readFileSync(path.join(DIR, "dates.json"), "utf8")); const titles = JSON.parse(readFileSync(path.join(DIR, "titles.json"), "utf8")); const acc = JSON.parse(readFileSync(path.join(DIR, "accepted.json"), "utf8")); const accept = new Set(acc.accepted), rejected = new Set(acc.rejected); let unclean = new Set(); try { unclean = new Set(JSON.parse(readFileSync(path.join(DIR, "unclean.json"), "utf8"))); } catch {} let cands = []; for (const f of readdirSync(path.join(SONG_DATA, "cand2")).filter((f) => f.endsWith(".json"))) { cands.push(...JSON.parse(readFileSync(path.join(SONG_DATA, "cand2", f), "utf8")).candidates); } const kOf = (c) => `${c.video}@${(+c.start).toFixed(2)}`; // Overlapping Whisper detections can yield two entries for one vowel; they would // collide in clip-keyed state and attach a verdict to the wrong clip. { const seen = new Set(); cands = cands.filter((c) => { const k = kOf(c); if (seen.has(k)) return false; seen.add(k); return true; }); } let riskWhy = new Map(); if (MODE === "salvage") { cands = cands.filter((c) => unclean.has(kOf(c))); } else if (MODE === "keeps") { // every clip kept ON PURPOSE for a stray word, whichever code it was given let R = {}; try { R = JSON.parse(readFileSync(path.join(DIR, "reasons.json"), "utf8")); } catch {} const want = new Set(Object.entries(R) .filter(([, v]) => v.verdict === 2 && DELIBERATE_CODES.has(v.code)).map(([k]) => k)); let side = {}; try { side = JSON.parse(readFileSync(path.join(DIR, "keepside.json"), "utf8")); } catch {} cands = cands.filter((c) => want.has(kOf(c))); // ones never sided come first; already-sided are still shown, to revise cands.sort((a, b) => (side[kOf(a)] ? 1 : 0) - (side[kOf(b)] ? 1 : 0)); console.log(`keeps mode: ${cands.length} deliberate keeps (${Object.keys(side).length} already sided)`); } else if (MODE === "verify") { // drop anything already settled in an earlier partial export let decided = {}; try { decided = JSON.parse(readFileSync(path.join(DIR, "decided.json"), "utf8")); } catch {} let risky = []; try { risky = JSON.parse(readFileSync(path.join(DIR, "risky.json"), "utf8")); } catch { console.error("no risky.json -- run find-risky.mjs first"); process.exit(1); } const order = new Map(risky.map((r, i) => [r.k, i])); for (const r of risky) riskWhy.set(r.k, r.why); const before = cands.length; cands = cands.filter((c) => order.has(kOf(c)) && !decided[kOf(c)]); cands.sort((a, b) => order.get(kOf(a)) - order.get(kOf(b))); // riskiest first const skipped = Object.keys(decided).length; if (skipped) console.log(`skipping ${skipped} already decided in an earlier export`); } else { // Never judged, and not already parked in the unclean pile. Machine-accepted // (provisional) clips are excluded too -- they are already in the palette. cands = cands.filter((c) => { const k = kOf(c); return !accept.has(k) && !rejected.has(k) && !unclean.has(k) && dates[c.video]; }); } const asrCache = new Map(); const wordsFor = (v) => { if (!asrCache.has(v)) { let w = []; try { w = JSON.parse(readFileSync(path.join(SONG_DATA, "asr", `${v}.json`), "utf8")).words ?? []; } catch {} // MUST be time-ordered: the window limits scan for the nearest word before // and after the clip and stop at the first hit. w.sort((a, b) => a.start - b.start); asrCache.set(v, w); } return asrCache.get(v); }; // Ordering is FITTED to real verdicts (fit-order.mjs), not guessed. The old // min-margin heuristic scored AUC 0.488 against 700 graded clips -- no signal at // all -- which is why a "cleanest-first" page came back only 40% good. The // refitted score holds out a third of the labels and still reaches AUC 0.93. let ORDER_MODEL = null; try { ORDER_MODEL = JSON.parse(readFileSync(path.join(DIR, "order-model.json"), "utf8")); } catch {} if (MODE === "verify") { console.log(`verify mode: ${cands.length} flagged clips, riskiest first`); } else if (process.env.ORDER === "date") { cands.sort((a, b) => dates[a.video].localeCompare(dates[b.video]) || a.start - b.start); } else if (ORDER_MODEL) { const sc = new Map(cands.map((c) => [kOf(c), scoreWith(ORDER_MODEL, c, wordsFor(c.video))])); const dir = process.env.ORDER === "dirty" ? -1 : 1; cands.sort((a, b) => dir * (sc.get(kOf(b)) - sc.get(kOf(a)))); console.log(`ordered by the fitted model (held-out AUC ${ORDER_MODEL.testAuc.toFixed(3)}, trained on ${ORDER_MODEL.n} verdicts)`); } else { console.log("no order-model.json -- falling back to date order; run fit-order.mjs"); cands.sort((a, b) => dates[a.video].localeCompare(dates[b.video]) || a.start - b.start); } const LIMIT = Number(process.env.LIMIT ?? (MODE === "sort" ? 700 : MODE === "verify" ? 600 : 400)); if (cands.length > LIMIT) { console.log(`capping ${cands.length} -> ${LIMIT} (best first; rerun for the next batch)`); cands = cands.slice(0, LIMIT); } // The page points at the SOURCE video and seeks to the clip; the thumbnail size // is just CSS. Cutting a small file per clip meant an ffmpeg run per clip on // every rebuild, to make a copy of footage already sitting on disk. const WANT_VIDEO = process.env.VIDEO !== "0"; const tmp = path.join(SONG_DATA, "umtmp"); if (existsSync(tmp)) rmSync(tmp, { recursive: true, force: true }); mkdirSync(tmp, { recursive: true }); const srcCache = new Map(); const loadSrc = (v) => { if (srcCache.has(v)) return srcCache.get(v); srcCache.clear(); const d = readWav(path.join(SONG_DATA, "wav48", `${v}.wav`)); srcCache.set(v, d); return d; }; const items = [], audio = []; let n = 0; for (const c of cands) { let x; try { ({ x } = loadSrc(c.video)); } catch { continue; } const a = Math.max(0, Math.round((c.start - PAD * 2) * SR)); const b = Math.min(x.length, Math.round((c.end + PAD * 2) * SR)); if (b - a < 0.1 * SR) continue; const dw = defaultWindow(x, SR, c, wordsFor(c.video)); // In salvage the whole point is that the clip needs work, so present it with // the trailing pop / next-word onset already stripped. The handles still move, // so a bad strip costs a drag rather than a lost clip. let strip = { cutMs: 0, why: "" }; if (EDIT) { // Run the stripper on the RAW window, then take whichever end is earlier. // Run on defaultWindow's output it fired on only 2 of 81, because that // function has already pulled the end back -- but it leaves up to one 43ms // analysis window of tail behind, and that residue is the audible pop. const raw = stripTrailingPop(x, SR, c.start, c.end, { words: wordsFor(c.video) }); const end = Math.min(dw.e, raw.cutMs ? raw.end : dw.e); if (end < dw.e - 0.004 && end - dw.s >= 0.14) { strip = { cutMs: Math.round((dw.e - end) * 1000), why: raw.why }; dw.e = end; } } const seg = Float32Array.prototype.slice.call(x, a, b); let peak = 0; for (let i = 0; i < seg.length; i += 1) peak = Math.max(peak, Math.abs(seg[i])); const g = peak > 1e-4 ? 0.85 / peak : 1; for (let i = 0; i < seg.length; i += 1) seg[i] *= g; const wav = path.join(tmp, "s.wav"), mp3 = path.join(tmp, "s.mp3"); writeWav(wav, seg, SR); execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-i", wav, "-c:a", "libmp3lame", "-b:a", "96k", "-ac", "1", mp3]); items.push({ i: n, k: kOf(c), v: c.video, d: dates[c.video], tok: c.token ?? "?", clipStart: +(a / SR).toFixed(3), clipLen: +((b - a) / SR).toFixed(3), selA: +(dw.s - (a / SR)).toFixed(3), selB: +(dw.e - (a / SR)).toFixed(3), f0: Math.round(c.f0), title: titles[c.video] ?? c.video, t: +c.start.toFixed(2), cut: strip.cutMs, why: strip.why, // A word that could not be excluded without dropping under the arranger's // 140ms floor. Nothing can clean these automatically -- say so on the card. wordIn: (wordsFor(c.video).find((w) => w.end > dw.s + 0.02 && w.start < dw.e - 0.02) || {}).w || "", risk: riskWhy.get(kOf(c)) || "", vid: WANT_VIDEO ? `file://${path.join(SONG_DATA, "media", `${c.video}.mp4`)}` : "", adj: (() => { const ws = wordsFor(c.video); const before = ws.filter((w) => w.end <= dw.s + 0.05 && w.end > dw.s - 0.40).slice(-1)[0]; const after = ws.filter((w) => w.start >= dw.e - 0.05 && w.start < dw.e + 0.40)[0]; return { b: before ? before.w : "", a: after ? after.w : "" }; })(), // Every word overlapping the padded clip, in clip-relative time, so the page // can stop an auto-expansion at a neighbouring word instead of swallowing // it. Silence is the usual stopping point; a word butted straight up against // the um is the case silence alone would miss. ws: wordsFor(c.video) .filter((w) => w.end > a / SR && w.start < b / SR) .map((w) => ({ s: +(w.start - a / SR).toFixed(3), e: +(w.end - a / SR).toFixed(3), w: w.w })), }); audio.push(readFileSync(mp3).toString("base64")); n += 1; if (n % 50 === 0) console.log(` encoded ${n}`); } rmSync(tmp, { recursive: true, force: true }); // ONE manifest, merged. Overwriting it per round would strand any paste-back // made against an earlier build of the page. const MF = path.join(DIR, "um-manifest.json"); let merged = new Map(); try { for (const it of JSON.parse(readFileSync(MF, "utf8")).items) merged.set(it.k, it); } catch {} for (const it of items) merged.set(it.k, it); // `vid` stays OUT of the manifest. The page needs it -- a file:// URL into this // machine's SONG_DATA/media, which `items` keeps in memory for META below -- // but this file is tracked, and a tracked file must not carry a machine path. // Nothing reads `vid` back from here (lib/clips.ts has no such field). writeFileSync( MF, JSON.stringify({ version: 1, items: [...merged.values()].map(({ vid: _v, ...rest }) => rest) }, null, 1), ); console.log(`manifest holds ${merged.size} clips (${items.length} on this page)`); // HUD target: Mortal Kombat is the current goal. 1,659 notes for three voices is // the floor where every note can still get its own clip; the higher number is // where there is enough choice that shifts stay small. const TARGET = Number(process.env.TARGET ?? 3930); const FLOOR = Number(process.env.FLOOR ?? 1659); const onPage = new Set(items.map((it) => it.k)); const BASE = acc.accepted.filter((k) => !onPage.has(k)).length; console.log(`HUD: palette ${acc.accepted.length}/${TARGET} (floor ${FLOOR}; base excluding this page ${BASE})`); const html = ` um ${MODE}
um review${MODE}

UM

${EDIT ? '
' : ""}
open this moment in Jeralyzer →
${MODE === "keeps" ? '' : ""}
${MODE !== "keeps" ? `
` : ""}
${EDIT ? '' : ""} ${EDIT ? '' : ""} ${EDIT ? '' : ""} ${EDIT ? 'silence 6% [ ]' : ""}

Paste this back to Claude

`; writeFileSync(OUT, html); console.log(`${OUT} — mode ${MODE}, ${items.length} clips, ${(Buffer.byteLength(html) / 1048576).toFixed(1)} MB`);