Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 1c92a17816b8a5e883e47936b3c68fb6d11493fb
parent ea1e503cadc761a7e5788f44689c863480c7b823
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 10 Oct 2026 03:01:18 -0400

Merge main (cf162d7c: umtool song asr-via-ops batching) into r20/integration

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mumtool/song/asr-via-ops.mjs | 136+++++++++++++++++++++++++++++++++++++++++++++++++------------------------------
1 file changed, 85 insertions(+), 51 deletions(-)

diff --git a/umtool/song/asr-via-ops.mjs b/umtool/song/asr-via-ops.mjs @@ -1,32 +1,39 @@ #!/usr/bin/env node // Word timings for a song's sources, through the editor: asr/<id>.json -// {words: [{w, start, end}]}, on the source's own clock. +// {windows, words: [{w, start, end}]}, on each source's own clock. // // render-poly's clipWindow keeps every note that is not `noTrim` clear of its // neighbouring words -- and grows a window over sounding audio up to 0.25 s past // its candidate, which without words runs straight into the next word of // continuous speech. A missing asr/<id>.json renders, with every guard gone. // -// Transcription goes through Archilyzer, never a hand-run engine: +// Transcription goes through Archilyzer, never a hand-run engine -- and as ONE +// job, not one per window: // -// pnpm ops transcribe {"path", "start", "end", "words": true, "out"} +// pnpm ops transcribe {"path": <batch.wav>, "words": true, "out"} // -// one job at a time, on the editor's worker pool. And only where the words are -// read: a window around each note (PAD seconds either side, overlapping windows -// merged), not the whole recording -- the MK plans use 2-3% of their 51 hours. -// A word cut by a window's edge is dropped (its start or end is unreliable), so -// PAD must be comfortably more than the 0.25 s a clip can grow. +// A one-off job jumps the queue but still waits for a worker to come free, and +// with the auto lane running that wait is a whole video's transcription: one +// window per job measured ~30 min each, so the MK plans' 841 windows would have +// taken days. So every window (PAD seconds either side of a note, overlapping +// ones merged) is cut at 16 kHz and laid end to end with GAP seconds of silence +// between them, transcribed once, and each word is mapped back to its source by +// the window it falls in. A word within EDGE seconds of a window's cut edge is +// dropped (its timing is unreliable there); PAD is comfortably more than the +// 0.25 s a clip can grow, so the guards keep everything they read. // // node asr-via-ops.mjs <plan.json>... // -// PAD seconds of context either side of a note (4) -// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env +// PAD seconds of context either side of a note (4) +// ASR_WORKER a settings worker id to run the batch on (default: the editor's pick); +// name a GPU worker -- the batch is hours of audio +// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env // -// Resumable: an id whose asr/<id>.json exists is skipped. Each file records the -// windows it covers. +// Resumable: an id whose asr/<id>.json exists is left out of the batch. The +// batch audio, its window map and the job's result are kept in +// SONG_DATA/asr-batch/ until the next run. import { execFileSync } from "node:child_process"; -import { existsSync, mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs"; -import { tmpdir } from "node:os"; +import { closeSync, existsSync, mkdirSync, openSync, readFileSync, renameSync, writeFileSync, writeSync } from "node:fs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; @@ -36,10 +43,15 @@ if (!plans.length || !process.env.WORKER_TOKEN) { process.exit(2); } const PAD = Number(process.env.PAD ?? 4); +const GAP = 2; +const EDGE = 0.3; +const SR = 16000; const DIR = path.dirname(new URL(import.meta.url).pathname); const OPS = process.env.OPS ?? path.join(DIR, "..", "..", "scripts", "archilyzer-ops.mjs"); const ASR = path.join(SONG_DATA, "asr"); +const BATCH = path.join(SONG_DATA, "asr-batch"); mkdirSync(ASR, { recursive: true }); +mkdirSync(BATCH, { recursive: true }); // Note spans per source, from every plan. const spans = new Map(); @@ -63,46 +75,68 @@ const windowsOf = (list) => { return out.map(([a, b]) => [+a.toFixed(3), +b.toFixed(3)]); }; -const tmp = mkdtempSync(path.join(tmpdir(), "asr-via-ops-")); const ids = [...spans.keys()].sort(); const todo = ids.filter((id) => !existsSync(path.join(ASR, `${id}.json`))); -console.log(`${ids.length} sources; ${todo.length} need word timings`); -let done = 0, waiting = 0; -for (const id of todo) { - const wav = path.join(SONG_DATA, "wav48", `${id}.wav`); - if (!existsSync(wav)) { - waiting += 1; - continue; - } - const windows = windowsOf(spans.get(id)); - const words = []; - let ok = true; - for (const [start, end] of windows) { - const out = path.join(tmp, `${id}-${start}.json`); - try { - execFileSync("node", [OPS, "transcribe", "--json", - JSON.stringify({ path: path.resolve(wav), start, end, words: true, out }), "--wait", "--quiet"], - { stdio: ["ignore", "ignore", "pipe"], maxBuffer: 1 << 26 }); - const r = JSON.parse(readFileSync(out, "utf8")); - // Keep only words wholly inside the window and clear of its edges. - for (const w of r.words ?? []) { - if (start > 0 && w.start < start + 0.3) continue; - if (w.end > end - 0.3) continue; - words.push({ w: w.w, start: w.start, end: w.end }); - } - } catch (e) { - console.log(` ${id} ${start}-${end}s failed: ${String(e.stderr ?? e.message).trim().split("\n").slice(-2).join(" | ")}`); - ok = false; - break; - } +const waiting = todo.filter((id) => !existsSync(path.join(SONG_DATA, "wav48", `${id}.wav`))); +const ready = todo.filter((id) => !waiting.includes(id)); +console.log(`${ids.length} sources; ${todo.length} need word timings; ${ready.length} have audio`); +if (!ready.length) process.exit(0); + +// --- the batch: every window end to end, GAP s of silence between ------------ +const wavPath = path.join(BATCH, "batch.wav"); +const map = []; // {id, start, end, at} -- `at` is the window's offset in the batch +const pcm = openSync(`${wavPath}.pcm`, "w"); +let samples = 0; +const silence = Buffer.alloc(GAP * SR * 2); +for (const id of ready) { + const src = path.join(SONG_DATA, "wav48", `${id}.wav`); + for (const [start, end] of windowsOf(spans.get(id))) { + const raw = execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-ss", String(start), "-to", String(end), + "-i", src, "-ac", "1", "-ar", String(SR), "-f", "s16le", "-"], { maxBuffer: 1 << 28 }); + map.push({ id, start, end, at: +(samples / SR).toFixed(4), dur: +(raw.length / 2 / SR).toFixed(4) }); + writeSync(pcm, raw); + samples += raw.length / 2; + writeSync(pcm, silence); + samples += silence.length / 2; } - if (!ok) continue; - words.sort((a, b) => a.start - b.start); +} +closeSync(pcm); +const header = Buffer.alloc(44); +header.write("RIFF", 0, "latin1"); header.writeUInt32LE(36 + samples * 2, 4); header.write("WAVE", 8, "latin1"); +header.write("fmt ", 12, "latin1"); header.writeUInt32LE(16, 16); header.writeUInt16LE(1, 20); header.writeUInt16LE(1, 22); +header.writeUInt32LE(SR, 24); header.writeUInt32LE(SR * 2, 28); header.writeUInt16LE(2, 32); header.writeUInt16LE(16, 34); +header.write("data", 36, "latin1"); header.writeUInt32LE(samples * 2, 40); +writeFileSync(wavPath, Buffer.concat([header, readFileSync(`${wavPath}.pcm`)])); +execFileSync("rm", ["-f", `${wavPath}.pcm`]); +writeFileSync(path.join(BATCH, "map.json"), JSON.stringify(map)); +console.log(`batch: ${map.length} windows, ${(samples / SR / 60).toFixed(1)} min -> ${wavPath}`); + +// --- one job -------------------------------------------------------------------- +const out = path.join(BATCH, "result.json"); +execFileSync("node", [OPS, "transcribe", "--json", + JSON.stringify({ path: wavPath, words: true, out, ...(process.env.ASR_WORKER ? { workerId: process.env.ASR_WORKER } : {}) }), "--wait", "--quiet"], +{ stdio: ["ignore", "ignore", "inherit"], maxBuffer: 1 << 26 }); +const words = JSON.parse(readFileSync(out, "utf8")).words ?? []; +console.log(`transcribed: ${words.length} words`); + +// --- back onto each source's clock ------------------------------------------- +const perId = new Map(ready.map((id) => [id, []])); +let wi = 0; +for (const w of words) { + while (wi < map.length && w.start >= map[wi].at + map[wi].dur) wi += 1; + const m = map[wi]; + if (!m || w.start < m.at) continue; // in a silence gap + const rel0 = w.start - m.at, rel1 = w.end - m.at; + if (rel1 > m.dur - EDGE) continue; + if (m.start > 0 && rel0 < EDGE) continue; + perId.get(m.id).push({ w: w.w, start: +(m.start + rel0).toFixed(3), end: +(m.start + rel1).toFixed(3) }); +} +for (const id of ready) { const file = path.join(ASR, `${id}.json`); - writeFileSync(`${file}.tmp`, JSON.stringify({ windows, words })); + const list = perId.get(id).sort((a, b) => a.start - b.start); + writeFileSync(`${file}.tmp`, JSON.stringify({ windows: windowsOf(spans.get(id)), words: list })); renameSync(`${file}.tmp`, file); - done += 1; - console.log(`${id}: ${words.length} words in ${windows.length} window(s) [${done}/${todo.length}]`); } -rmSync(tmp, { recursive: true, force: true }); -console.log(`asr: ${done} written, ${waiting} waiting for audio`); +const counts = ready.map((id) => perId.get(id).length); +console.log(`asr: ${ready.length} written (${counts.filter((n) => n === 0).length} with no words), ` + + `${waiting.length} waiting for audio`);