commit 6956f44051905f22658ce63bd30f2788c73b4793
parent 07c587f904d9a7359fed4965e4a6b1102ba4dc3c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 8 Oct 2026 23:16:08 -0400
umtool song: asr-via-ops -- word timings around each note, through pnpm ops transcribe
Windows of PAD seconds around every note, merged, each one an ops transcribe
job with "words": true; words cut by a window's edge are dropped. Writes
asr/<id>.json {windows, words} for render-poly's clip-window guards.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
1 file changed, 108 insertions(+), 0 deletions(-)
diff --git a/umtool/song/asr-via-ops.mjs b/umtool/song/asr-via-ops.mjs
@@ -0,0 +1,108 @@
+#!/usr/bin/env node
+// Word timings for a song's sources, through the editor: asr/<id>.json
+// {words: [{w, start, end}]}, on the source's own clock.
+//
+// render-poly's clipWindow keeps every note that is not `noTrim` clear of its
+// neighbouring words -- and grows a window over sounding audio up to 0.25 s past
+// its candidate, which without words runs straight into the next word of
+// continuous speech. A missing asr/<id>.json renders, with every guard gone.
+//
+// Transcription goes through Archilyzer, never a hand-run engine:
+//
+// pnpm ops transcribe {"path", "start", "end", "words": true, "out"}
+//
+// one job at a time, on the editor's worker pool. And only where the words are
+// read: a window around each note (PAD seconds either side, overlapping windows
+// merged), not the whole recording -- the MK plans use 2-3% of their 51 hours.
+// A word cut by a window's edge is dropped (its start or end is unreliable), so
+// PAD must be comfortably more than the 0.25 s a clip can grow.
+//
+// node asr-via-ops.mjs <plan.json>...
+//
+// PAD seconds of context either side of a note (4)
+// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env
+//
+// Resumable: an id whose asr/<id>.json exists is skipped. Each file records the
+// windows it covers.
+import { execFileSync } from "node:child_process";
+import { existsSync, mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { SONG_DATA } from "./paths.mjs";
+
+const plans = process.argv.slice(2).filter((a) => a.endsWith(".json"));
+if (!plans.length || !process.env.WORKER_TOKEN) {
+ console.error("usage: WORKER_TOKEN=... asr-via-ops.mjs <plan.json>...");
+ process.exit(2);
+}
+const PAD = Number(process.env.PAD ?? 4);
+const DIR = path.dirname(new URL(import.meta.url).pathname);
+const OPS = process.env.OPS ?? path.join(DIR, "..", "..", "scripts", "archilyzer-ops.mjs");
+const ASR = path.join(SONG_DATA, "asr");
+mkdirSync(ASR, { recursive: true });
+
+// Note spans per source, from every plan.
+const spans = new Map();
+for (const p of plans) {
+ for (const v of JSON.parse(readFileSync(p, "utf8")).voices) {
+ for (const n of v.plan) {
+ if (!n.video) continue;
+ if (!spans.has(n.video)) spans.set(n.video, []);
+ spans.get(n.video).push([n.srcStart, n.srcEnd]);
+ }
+ }
+}
+const windowsOf = (list) => {
+ const w = list.map(([a, b]) => [Math.max(0, a - PAD), b + PAD]).sort((x, y) => x[0] - y[0]);
+ const out = [];
+ for (const [a, b] of w) {
+ const last = out.at(-1);
+ if (last && a <= last[1]) last[1] = Math.max(last[1], b);
+ else out.push([a, b]);
+ }
+ return out.map(([a, b]) => [+a.toFixed(3), +b.toFixed(3)]);
+};
+
+const tmp = mkdtempSync(path.join(tmpdir(), "asr-via-ops-"));
+const ids = [...spans.keys()].sort();
+const todo = ids.filter((id) => !existsSync(path.join(ASR, `${id}.json`)));
+console.log(`${ids.length} sources; ${todo.length} need word timings`);
+let done = 0, waiting = 0;
+for (const id of todo) {
+ const wav = path.join(SONG_DATA, "wav48", `${id}.wav`);
+ if (!existsSync(wav)) {
+ waiting += 1;
+ continue;
+ }
+ const windows = windowsOf(spans.get(id));
+ const words = [];
+ let ok = true;
+ for (const [start, end] of windows) {
+ const out = path.join(tmp, `${id}-${start}.json`);
+ try {
+ execFileSync("node", [OPS, "transcribe", "--json",
+ JSON.stringify({ path: path.resolve(wav), start, end, words: true, out }), "--wait", "--quiet"],
+ { stdio: ["ignore", "ignore", "pipe"], maxBuffer: 1 << 26 });
+ const r = JSON.parse(readFileSync(out, "utf8"));
+ // Keep only words wholly inside the window and clear of its edges.
+ for (const w of r.words ?? []) {
+ if (start > 0 && w.start < start + 0.3) continue;
+ if (w.end > end - 0.3) continue;
+ words.push({ w: w.w, start: w.start, end: w.end });
+ }
+ } catch (e) {
+ console.log(` ${id} ${start}-${end}s failed: ${String(e.stderr ?? e.message).trim().split("\n").slice(-2).join(" | ")}`);
+ ok = false;
+ break;
+ }
+ }
+ if (!ok) continue;
+ words.sort((a, b) => a.start - b.start);
+ const file = path.join(ASR, `${id}.json`);
+ writeFileSync(`${file}.tmp`, JSON.stringify({ windows, words }));
+ renameSync(`${file}.tmp`, file);
+ done += 1;
+ console.log(`${id}: ${words.length} words in ${windows.length} window(s) [${done}/${todo.length}]`);
+}
+rmSync(tmp, { recursive: true, force: true });
+console.log(`asr: ${done} written, ${waiting} waiting for audio`);