commit cf162d7cbf289f194f87418af2404faad2e663a0
parent 7f91177c75088089fdf8556b05ca9af26d5a5c89
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 01:56:52 -0400
umtool song: asr-via-ops batches every window into one ops transcribe job
One window per job waited ~30 min each for a worker behind the auto lane, so
the MK plans' 841 windows would have taken days. The windows are now cut at
16 kHz and laid end to end with 2 s of silence between them, transcribed in one
"words": true job (ASR_WORKER names the worker), and each word mapped back to
its source by the window it falls in.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
1 file changed, 85 insertions(+), 51 deletions(-)
diff --git a/umtool/song/asr-via-ops.mjs b/umtool/song/asr-via-ops.mjs
@@ -1,32 +1,39 @@
#!/usr/bin/env node
// Word timings for a song's sources, through the editor: asr/<id>.json
-// {words: [{w, start, end}]}, on the source's own clock.
+// {windows, words: [{w, start, end}]}, on each source's own clock.
//
// render-poly's clipWindow keeps every note that is not `noTrim` clear of its
// neighbouring words -- and grows a window over sounding audio up to 0.25 s past
// its candidate, which without words runs straight into the next word of
// continuous speech. A missing asr/<id>.json renders, with every guard gone.
//
-// Transcription goes through Archilyzer, never a hand-run engine:
+// Transcription goes through Archilyzer, never a hand-run engine -- and as ONE
+// job, not one per window:
//
-// pnpm ops transcribe {"path", "start", "end", "words": true, "out"}
+// pnpm ops transcribe {"path": <batch.wav>, "words": true, "out"}
//
-// one job at a time, on the editor's worker pool. And only where the words are
-// read: a window around each note (PAD seconds either side, overlapping windows
-// merged), not the whole recording -- the MK plans use 2-3% of their 51 hours.
-// A word cut by a window's edge is dropped (its start or end is unreliable), so
-// PAD must be comfortably more than the 0.25 s a clip can grow.
+// A one-off job jumps the queue but still waits for a worker to come free, and
+// with the auto lane running that wait is a whole video's transcription: one
+// window per job measured ~30 min each, so the MK plans' 841 windows would have
+// taken days. So every window (PAD seconds either side of a note, overlapping
+// ones merged) is cut at 16 kHz and laid end to end with GAP seconds of silence
+// between them, transcribed once, and each word is mapped back to its source by
+// the window it falls in. A word within EDGE seconds of a window's cut edge is
+// dropped (its timing is unreliable there); PAD is comfortably more than the
+// 0.25 s a clip can grow, so the guards keep everything they read.
//
// node asr-via-ops.mjs <plan.json>...
//
-// PAD seconds of context either side of a note (4)
-// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env
+// PAD seconds of context either side of a note (4)
+// ASR_WORKER a settings worker id to run the batch on (default: the editor's pick);
+// name a GPU worker -- the batch is hours of audio
+// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env
//
-// Resumable: an id whose asr/<id>.json exists is skipped. Each file records the
-// windows it covers.
+// Resumable: an id whose asr/<id>.json exists is left out of the batch. The
+// batch audio, its window map and the job's result are kept in
+// SONG_DATA/asr-batch/ until the next run.
import { execFileSync } from "node:child_process";
-import { existsSync, mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
-import { tmpdir } from "node:os";
+import { closeSync, existsSync, mkdirSync, openSync, readFileSync, renameSync, writeFileSync, writeSync } from "node:fs";
import path from "node:path";
import { SONG_DATA } from "./paths.mjs";
@@ -36,10 +43,15 @@ if (!plans.length || !process.env.WORKER_TOKEN) {
process.exit(2);
}
const PAD = Number(process.env.PAD ?? 4);
+const GAP = 2;
+const EDGE = 0.3;
+const SR = 16000;
const DIR = path.dirname(new URL(import.meta.url).pathname);
const OPS = process.env.OPS ?? path.join(DIR, "..", "..", "scripts", "archilyzer-ops.mjs");
const ASR = path.join(SONG_DATA, "asr");
+const BATCH = path.join(SONG_DATA, "asr-batch");
mkdirSync(ASR, { recursive: true });
+mkdirSync(BATCH, { recursive: true });
// Note spans per source, from every plan.
const spans = new Map();
@@ -63,46 +75,68 @@ const windowsOf = (list) => {
return out.map(([a, b]) => [+a.toFixed(3), +b.toFixed(3)]);
};
-const tmp = mkdtempSync(path.join(tmpdir(), "asr-via-ops-"));
const ids = [...spans.keys()].sort();
const todo = ids.filter((id) => !existsSync(path.join(ASR, `${id}.json`)));
-console.log(`${ids.length} sources; ${todo.length} need word timings`);
-let done = 0, waiting = 0;
-for (const id of todo) {
- const wav = path.join(SONG_DATA, "wav48", `${id}.wav`);
- if (!existsSync(wav)) {
- waiting += 1;
- continue;
- }
- const windows = windowsOf(spans.get(id));
- const words = [];
- let ok = true;
- for (const [start, end] of windows) {
- const out = path.join(tmp, `${id}-${start}.json`);
- try {
- execFileSync("node", [OPS, "transcribe", "--json",
- JSON.stringify({ path: path.resolve(wav), start, end, words: true, out }), "--wait", "--quiet"],
- { stdio: ["ignore", "ignore", "pipe"], maxBuffer: 1 << 26 });
- const r = JSON.parse(readFileSync(out, "utf8"));
- // Keep only words wholly inside the window and clear of its edges.
- for (const w of r.words ?? []) {
- if (start > 0 && w.start < start + 0.3) continue;
- if (w.end > end - 0.3) continue;
- words.push({ w: w.w, start: w.start, end: w.end });
- }
- } catch (e) {
- console.log(` ${id} ${start}-${end}s failed: ${String(e.stderr ?? e.message).trim().split("\n").slice(-2).join(" | ")}`);
- ok = false;
- break;
- }
+const waiting = todo.filter((id) => !existsSync(path.join(SONG_DATA, "wav48", `${id}.wav`)));
+const ready = todo.filter((id) => !waiting.includes(id));
+console.log(`${ids.length} sources; ${todo.length} need word timings; ${ready.length} have audio`);
+if (!ready.length) process.exit(0);
+
+// --- the batch: every window end to end, GAP s of silence between ------------
+const wavPath = path.join(BATCH, "batch.wav");
+const map = []; // {id, start, end, at} -- `at` is the window's offset in the batch
+const pcm = openSync(`${wavPath}.pcm`, "w");
+let samples = 0;
+const silence = Buffer.alloc(GAP * SR * 2);
+for (const id of ready) {
+ const src = path.join(SONG_DATA, "wav48", `${id}.wav`);
+ for (const [start, end] of windowsOf(spans.get(id))) {
+ const raw = execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-ss", String(start), "-to", String(end),
+ "-i", src, "-ac", "1", "-ar", String(SR), "-f", "s16le", "-"], { maxBuffer: 1 << 28 });
+ map.push({ id, start, end, at: +(samples / SR).toFixed(4), dur: +(raw.length / 2 / SR).toFixed(4) });
+ writeSync(pcm, raw);
+ samples += raw.length / 2;
+ writeSync(pcm, silence);
+ samples += silence.length / 2;
}
- if (!ok) continue;
- words.sort((a, b) => a.start - b.start);
+}
+closeSync(pcm);
+const header = Buffer.alloc(44);
+header.write("RIFF", 0, "latin1"); header.writeUInt32LE(36 + samples * 2, 4); header.write("WAVE", 8, "latin1");
+header.write("fmt ", 12, "latin1"); header.writeUInt32LE(16, 16); header.writeUInt16LE(1, 20); header.writeUInt16LE(1, 22);
+header.writeUInt32LE(SR, 24); header.writeUInt32LE(SR * 2, 28); header.writeUInt16LE(2, 32); header.writeUInt16LE(16, 34);
+header.write("data", 36, "latin1"); header.writeUInt32LE(samples * 2, 40);
+writeFileSync(wavPath, Buffer.concat([header, readFileSync(`${wavPath}.pcm`)]));
+execFileSync("rm", ["-f", `${wavPath}.pcm`]);
+writeFileSync(path.join(BATCH, "map.json"), JSON.stringify(map));
+console.log(`batch: ${map.length} windows, ${(samples / SR / 60).toFixed(1)} min -> ${wavPath}`);
+
+// --- one job --------------------------------------------------------------------
+const out = path.join(BATCH, "result.json");
+execFileSync("node", [OPS, "transcribe", "--json",
+ JSON.stringify({ path: wavPath, words: true, out, ...(process.env.ASR_WORKER ? { workerId: process.env.ASR_WORKER } : {}) }), "--wait", "--quiet"],
+{ stdio: ["ignore", "ignore", "inherit"], maxBuffer: 1 << 26 });
+const words = JSON.parse(readFileSync(out, "utf8")).words ?? [];
+console.log(`transcribed: ${words.length} words`);
+
+// --- back onto each source's clock -------------------------------------------
+const perId = new Map(ready.map((id) => [id, []]));
+let wi = 0;
+for (const w of words) {
+ while (wi < map.length && w.start >= map[wi].at + map[wi].dur) wi += 1;
+ const m = map[wi];
+ if (!m || w.start < m.at) continue; // in a silence gap
+ const rel0 = w.start - m.at, rel1 = w.end - m.at;
+ if (rel1 > m.dur - EDGE) continue;
+ if (m.start > 0 && rel0 < EDGE) continue;
+ perId.get(m.id).push({ w: w.w, start: +(m.start + rel0).toFixed(3), end: +(m.start + rel1).toFixed(3) });
+}
+for (const id of ready) {
const file = path.join(ASR, `${id}.json`);
- writeFileSync(`${file}.tmp`, JSON.stringify({ windows, words }));
+ const list = perId.get(id).sort((a, b) => a.start - b.start);
+ writeFileSync(`${file}.tmp`, JSON.stringify({ windows: windowsOf(spans.get(id)), words: list }));
renameSync(`${file}.tmp`, file);
- done += 1;
- console.log(`${id}: ${words.length} words in ${windows.length} window(s) [${done}/${todo.length}]`);
}
-rmSync(tmp, { recursive: true, force: true });
-console.log(`asr: ${done} written, ${waiting} waiting for audio`);
+const counts = ready.map((id) => perId.get(id).length);
+console.log(`asr: ${ready.length} written (${counts.filter((n) => n === 0).length} with no words), ` +
+ `${waiting.length} waiting for audio`);