Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 6956f44051905f22658ce63bd30f2788c73b4793
parent 07c587f904d9a7359fed4965e4a6b1102ba4dc3c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  8 Oct 2026 23:16:08 -0400

umtool song: asr-via-ops -- word timings around each note, through pnpm ops transcribe

Windows of PAD seconds around every note, merged, each one an ops transcribe
job with "words": true; words cut by a window's edge are dropped. Writes
asr/<id>.json {windows, words} for render-poly's clip-window guards.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Aumtool/song/asr-via-ops.mjs | 108+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 108 insertions(+), 0 deletions(-)

diff --git a/umtool/song/asr-via-ops.mjs b/umtool/song/asr-via-ops.mjs @@ -0,0 +1,108 @@ +#!/usr/bin/env node +// Word timings for a song's sources, through the editor: asr/<id>.json +// {words: [{w, start, end}]}, on the source's own clock. +// +// render-poly's clipWindow keeps every note that is not `noTrim` clear of its +// neighbouring words -- and grows a window over sounding audio up to 0.25 s past +// its candidate, which without words runs straight into the next word of +// continuous speech. A missing asr/<id>.json renders, with every guard gone. +// +// Transcription goes through Archilyzer, never a hand-run engine: +// +// pnpm ops transcribe {"path", "start", "end", "words": true, "out"} +// +// one job at a time, on the editor's worker pool. And only where the words are +// read: a window around each note (PAD seconds either side, overlapping windows +// merged), not the whole recording -- the MK plans use 2-3% of their 51 hours. +// A word cut by a window's edge is dropped (its start or end is unreliable), so +// PAD must be comfortably more than the 0.25 s a clip can grow. +// +// node asr-via-ops.mjs <plan.json>... +// +// PAD seconds of context either side of a note (4) +// OPS the ops CLI (<repo>/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env +// +// Resumable: an id whose asr/<id>.json exists is skipped. Each file records the +// windows it covers. +import { execFileSync } from "node:child_process"; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { SONG_DATA } from "./paths.mjs"; + +const plans = process.argv.slice(2).filter((a) => a.endsWith(".json")); +if (!plans.length || !process.env.WORKER_TOKEN) { + console.error("usage: WORKER_TOKEN=... asr-via-ops.mjs <plan.json>..."); + process.exit(2); +} +const PAD = Number(process.env.PAD ?? 4); +const DIR = path.dirname(new URL(import.meta.url).pathname); +const OPS = process.env.OPS ?? path.join(DIR, "..", "..", "scripts", "archilyzer-ops.mjs"); +const ASR = path.join(SONG_DATA, "asr"); +mkdirSync(ASR, { recursive: true }); + +// Note spans per source, from every plan. +const spans = new Map(); +for (const p of plans) { + for (const v of JSON.parse(readFileSync(p, "utf8")).voices) { + for (const n of v.plan) { + if (!n.video) continue; + if (!spans.has(n.video)) spans.set(n.video, []); + spans.get(n.video).push([n.srcStart, n.srcEnd]); + } + } +} +const windowsOf = (list) => { + const w = list.map(([a, b]) => [Math.max(0, a - PAD), b + PAD]).sort((x, y) => x[0] - y[0]); + const out = []; + for (const [a, b] of w) { + const last = out.at(-1); + if (last && a <= last[1]) last[1] = Math.max(last[1], b); + else out.push([a, b]); + } + return out.map(([a, b]) => [+a.toFixed(3), +b.toFixed(3)]); +}; + +const tmp = mkdtempSync(path.join(tmpdir(), "asr-via-ops-")); +const ids = [...spans.keys()].sort(); +const todo = ids.filter((id) => !existsSync(path.join(ASR, `${id}.json`))); +console.log(`${ids.length} sources; ${todo.length} need word timings`); +let done = 0, waiting = 0; +for (const id of todo) { + const wav = path.join(SONG_DATA, "wav48", `${id}.wav`); + if (!existsSync(wav)) { + waiting += 1; + continue; + } + const windows = windowsOf(spans.get(id)); + const words = []; + let ok = true; + for (const [start, end] of windows) { + const out = path.join(tmp, `${id}-${start}.json`); + try { + execFileSync("node", [OPS, "transcribe", "--json", + JSON.stringify({ path: path.resolve(wav), start, end, words: true, out }), "--wait", "--quiet"], + { stdio: ["ignore", "ignore", "pipe"], maxBuffer: 1 << 26 }); + const r = JSON.parse(readFileSync(out, "utf8")); + // Keep only words wholly inside the window and clear of its edges. + for (const w of r.words ?? []) { + if (start > 0 && w.start < start + 0.3) continue; + if (w.end > end - 0.3) continue; + words.push({ w: w.w, start: w.start, end: w.end }); + } + } catch (e) { + console.log(` ${id} ${start}-${end}s failed: ${String(e.stderr ?? e.message).trim().split("\n").slice(-2).join(" | ")}`); + ok = false; + break; + } + } + if (!ok) continue; + words.sort((a, b) => a.start - b.start); + const file = path.join(ASR, `${id}.json`); + writeFileSync(`${file}.tmp`, JSON.stringify({ windows, words })); + renameSync(`${file}.tmp`, file); + done += 1; + console.log(`${id}: ${words.length} words in ${windows.length} window(s) [${done}/${todo.length}]`); +} +rmSync(tmp, { recursive: true, force: true }); +console.log(`asr: ${done} written, ${waiting} waiting for audio`);