#!/usr/bin/env node // Word timings for a song's sources, through the editor: asr/.json // {windows, words: [{w, start, end}]}, on each source's own clock. // // render-poly's clipWindow keeps every note that is not `noTrim` clear of its // neighbouring words -- and grows a window over sounding audio up to 0.25 s past // its candidate, which without words runs straight into the next word of // continuous speech. A missing asr/.json renders, with every guard gone. // // Transcription goes through Archilyzer, never a hand-run engine -- and as ONE // job, not one per window: // // pnpm ops transcribe {"path": , "words": true, "out"} // // A one-off job jumps the queue but still waits for a worker to come free, and // with the auto lane running that wait is a whole video's transcription: one // window per job measured ~30 min each, so the MK plans' 841 windows would have // taken days. So every window (PAD seconds either side of a note, overlapping // ones merged) is cut at 16 kHz and laid end to end with GAP seconds of silence // between them, transcribed once, and each word is mapped back to its source by // the window it falls in. A word within EDGE seconds of a window's cut edge is // dropped (its timing is unreliable there); PAD is comfortably more than the // 0.25 s a clip can grow, so the guards keep everything they read. // // node asr-via-ops.mjs ... // // PAD seconds of context either side of a note (4) // ASR_WORKER a settings worker id to run the batch on (default: the editor's pick); // name a GPU worker -- the batch is hours of audio // OPS the ops CLI (/scripts/archilyzer-ops.mjs); WORKER_TOKEN from the env // // Resumable: an id whose asr/.json exists is left out of the batch. The // batch audio, its window map and the job's result are kept in // SONG_DATA/asr-batch/ until the next run. import { execFileSync } from "node:child_process"; import { closeSync, existsSync, mkdirSync, openSync, readFileSync, renameSync, writeFileSync, writeSync } from "node:fs"; import path from "node:path"; import { SONG_DATA } from "./paths.mjs"; const plans = process.argv.slice(2).filter((a) => a.endsWith(".json")); if (!plans.length || !process.env.WORKER_TOKEN) { console.error("usage: WORKER_TOKEN=... asr-via-ops.mjs ..."); process.exit(2); } const PAD = Number(process.env.PAD ?? 4); const GAP = 2; const EDGE = 0.3; const SR = 16000; const DIR = path.dirname(new URL(import.meta.url).pathname); const OPS = process.env.OPS ?? path.join(DIR, "..", "..", "scripts", "archilyzer-ops.mjs"); const ASR = path.join(SONG_DATA, "asr"); const BATCH = path.join(SONG_DATA, "asr-batch"); mkdirSync(ASR, { recursive: true }); mkdirSync(BATCH, { recursive: true }); // Note spans per source, from every plan. const spans = new Map(); for (const p of plans) { for (const v of JSON.parse(readFileSync(p, "utf8")).voices) { for (const n of v.plan) { if (!n.video) continue; if (!spans.has(n.video)) spans.set(n.video, []); spans.get(n.video).push([n.srcStart, n.srcEnd]); } } } const windowsOf = (list) => { const w = list.map(([a, b]) => [Math.max(0, a - PAD), b + PAD]).sort((x, y) => x[0] - y[0]); const out = []; for (const [a, b] of w) { const last = out.at(-1); if (last && a <= last[1]) last[1] = Math.max(last[1], b); else out.push([a, b]); } return out.map(([a, b]) => [+a.toFixed(3), +b.toFixed(3)]); }; const ids = [...spans.keys()].sort(); const todo = ids.filter((id) => !existsSync(path.join(ASR, `${id}.json`))); const waiting = todo.filter((id) => !existsSync(path.join(SONG_DATA, "wav48", `${id}.wav`))); const ready = todo.filter((id) => !waiting.includes(id)); console.log(`${ids.length} sources; ${todo.length} need word timings; ${ready.length} have audio`); if (!ready.length) process.exit(0); // --- the batch: every window end to end, GAP s of silence between ------------ const wavPath = path.join(BATCH, "batch.wav"); const map = []; // {id, start, end, at} -- `at` is the window's offset in the batch const pcm = openSync(`${wavPath}.pcm`, "w"); let samples = 0; const silence = Buffer.alloc(GAP * SR * 2); for (const id of ready) { const src = path.join(SONG_DATA, "wav48", `${id}.wav`); for (const [start, end] of windowsOf(spans.get(id))) { const raw = execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-ss", String(start), "-to", String(end), "-i", src, "-ac", "1", "-ar", String(SR), "-f", "s16le", "-"], { maxBuffer: 1 << 28 }); map.push({ id, start, end, at: +(samples / SR).toFixed(4), dur: +(raw.length / 2 / SR).toFixed(4) }); writeSync(pcm, raw); samples += raw.length / 2; writeSync(pcm, silence); samples += silence.length / 2; } } closeSync(pcm); const header = Buffer.alloc(44); header.write("RIFF", 0, "latin1"); header.writeUInt32LE(36 + samples * 2, 4); header.write("WAVE", 8, "latin1"); header.write("fmt ", 12, "latin1"); header.writeUInt32LE(16, 16); header.writeUInt16LE(1, 20); header.writeUInt16LE(1, 22); header.writeUInt32LE(SR, 24); header.writeUInt32LE(SR * 2, 28); header.writeUInt16LE(2, 32); header.writeUInt16LE(16, 34); header.write("data", 36, "latin1"); header.writeUInt32LE(samples * 2, 40); writeFileSync(wavPath, Buffer.concat([header, readFileSync(`${wavPath}.pcm`)])); execFileSync("rm", ["-f", `${wavPath}.pcm`]); writeFileSync(path.join(BATCH, "map.json"), JSON.stringify(map)); console.log(`batch: ${map.length} windows, ${(samples / SR / 60).toFixed(1)} min -> ${wavPath}`); // --- one job -------------------------------------------------------------------- const out = path.join(BATCH, "result.json"); execFileSync("node", [OPS, "transcribe", "--json", JSON.stringify({ path: wavPath, words: true, out, ...(process.env.ASR_WORKER ? { workerId: process.env.ASR_WORKER } : {}) }), "--wait", "--quiet"], { stdio: ["ignore", "ignore", "inherit"], maxBuffer: 1 << 26 }); const words = JSON.parse(readFileSync(out, "utf8")).words ?? []; console.log(`transcribed: ${words.length} words`); // --- back onto each source's clock ------------------------------------------- const perId = new Map(ready.map((id) => [id, []])); let wi = 0; for (const w of words) { while (wi < map.length && w.start >= map[wi].at + map[wi].dur) wi += 1; const m = map[wi]; if (!m || w.start < m.at) continue; // in a silence gap const rel0 = w.start - m.at, rel1 = w.end - m.at; if (rel1 > m.dur - EDGE) continue; if (m.start > 0 && rel0 < EDGE) continue; perId.get(m.id).push({ w: w.w, start: +(m.start + rel0).toFixed(3), end: +(m.start + rel1).toFixed(3) }); } for (const id of ready) { const file = path.join(ASR, `${id}.json`); const list = perId.get(id).sort((a, b) => a.start - b.start); writeFileSync(`${file}.tmp`, JSON.stringify({ windows: windowsOf(spans.get(id)), words: list })); renameSync(`${file}.tmp`, file); } const counts = ready.map((id) => perId.get(id).length); console.log(`asr: ${ready.length} written (${counts.filter((n) => n === 0).length} with no words), ` + `${waiting.length} waiting for audio`);