Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit add79a2fc87ea8949bc3178c6b1d3f66807458c0
parent 2898c1f6816cb53b882bdac851e106af7420dc4e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 13 Jun 2026 13:11:55 -0400

parakeet stitch test

Diffstat:
Mcommon/jobs/progressParsers.ts | 19+++++++++++++++++++
Mcommon/lib/paths.ts | 11+++++++++++
Mcommon/lib/transcriptionApps.ts | 36++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/settings/components/SettingsForm.tsx | 8++++++--
Aeditor/e2e/fixtures/bin/fake-parakeet-stitch.mjs | 41+++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/parakeet.spec.ts | 83+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/package.json | 4++--
Ascripts/parakeet-stitch.mjs | 293+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
9 files changed, 492 insertions(+), 4 deletions(-)

diff --git a/common/jobs/progressParsers.ts b/common/jobs/progressParsers.ts @@ -200,3 +200,22 @@ export function createChoughProgressParser(): { }, }; } + +// parakeet-stitch.mjs (the overlapping-segment wrapper) prints one stderr line +// per window, e.g. +// parakeet-stitch: segment 3/12 @360s — transcribing +// Progress = completed segment index / total segments. +export function createParakeetProgressParser(): { + feed: (line: string) => ProgressUpdate | null; +} { + return { + feed(line: string): ProgressUpdate | null { + const m = line.match(/segment\s+(\d+)\s*\/\s*(\d+)/i); + if (!m) return null; + const i = Number.parseInt(m[1], 10); + const n = Number.parseInt(m[2], 10); + if (!Number.isFinite(n) || n <= 0) return null; + return { fraction: clamp01(i / n), detail: `segment ${i}/${n}` }; + }, + }; +} diff --git a/common/lib/paths.ts b/common/lib/paths.ts @@ -37,6 +37,12 @@ export type Paths = { whisperBin: string; whisperModel: string; ffmpegBin: string; + // parakeet (overlapping-segment stitching) app. parakeetBin is the standalone + // wrapper script invoked as the app binary; parakeetCliBin is the underlying + // parakeet-cli it drives; parakeetModel is the default .gguf model. + parakeetBin: string; + parakeetCliBin: string; + parakeetModel: string; }; let cached: Paths | null = null; @@ -91,6 +97,11 @@ export function getPaths(): Paths { "ggml-base.en.bin", ), ffmpegBin: process.env.FFMPEG_BIN ?? "ffmpeg", + parakeetBin: + process.env.PARAKEET_STITCH_BIN ?? + path.join(monorepoRoot, "scripts", "parakeet-stitch.mjs"), + parakeetCliBin: process.env.PARAKEET_CLI ?? "parakeet-cli", + parakeetModel: process.env.PARAKEET_MODEL ?? "", }; return cached; } diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts @@ -15,6 +15,7 @@ import { type ProgressUpdate, createTranscribeProgressParser, createChoughProgressParser, + createParakeetProgressParser, } from "../jobs/progressParsers"; export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt"; @@ -197,9 +198,44 @@ const chough: TranscriptionApp = { makeProgressParser: createChoughProgressParser, }; +// parakeet.cpp via the overlapping-segment wrapper (scripts/parakeet-stitch.mjs). +// The wrapper IS the binary here: it slices the audio into overlapping 16kHz-mono +// windows, runs parakeet-cli per window, and stitches the word timestamps into a +// single chough-native JSON document written to the exact `-o` path (no extension +// appended — same contract as chough). `model` is the .gguf; `chunkSize` is the +// per-window length in seconds (overlap is wrapper-defaulted, tunable via the +// PARAKEET_OVERLAP_SEC env). +const parakeet: TranscriptionApp = { + id: "parakeet", + label: "parakeet.cpp (overlapping segments)", + fields: { model: true, chunkSize: true }, + defaultBin: () => getPaths().parakeetBin, + build({ audioFile, outputBase, config }) { + const paths = getPaths(); + const model = config.model?.trim() || paths.parakeetModel; + const argv = ["--model", model, "--output", outputBase]; + if (typeof config.chunkSize === "number" && config.chunkSize > 0) { + argv.push("--segment", String(Math.floor(config.chunkSize))); + } + argv.push(audioFile); + return { + argv, + env: { + PARAKEET_CLI: paths.parakeetCliBin, + FFMPEG_BIN: paths.ffmpegBin, + }, + // The wrapper writes EXACTLY the --output path (like chough's -o). + outputFile: outputBase, + outputFormat: "chough-json", + }; + }, + makeProgressParser: createParakeetProgressParser, +}; + export const TRANSCRIPTION_APPS: Record<string, TranscriptionApp> = { [whisperCpp.id]: whisperCpp, [chough.id]: chough, + [parakeet.id]: parakeet, }; export const DEFAULT_TRANSCRIPTION_APP_ID = "whisper-cpp"; diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **New transcription app: parakeet.cpp with overlapping-segment stitching.** A third **App** option (alongside whisper.cpp and chough) in **Settings → Transcription** transcribes via `parakeet-cli`, working around its two long-audio limitations: it only accepts a single WAV, and a >4GB PCM stream silently yields an empty transcript (dr_wav's data-chunk size is 32-bit). A standalone wrapper (`scripts/parakeet-stitch.mjs`) — usable both as the app's binary and directly on the CLI — slices the source into **overlapping** 16kHz-mono windows (the format parakeet-cli resamples to internally anyway), runs `parakeet-cli --json --timestamps` on each, then stitches the per-window word timestamps back into one transcript. The overlap guarantees a word clipped at one window's boundary is captured whole by the neighbour; a cut point in the middle of each overlap region decides which window owns each word, so nothing is dropped or duplicated across seams. Output is chough-native JSON (seconds-based `chunk_data`), so it flows through the existing format-aware normalize/index path with no downstream changes and is tagged `chough-json` in `transcript.cues.json`. The settings fields are **Model** (the `.gguf` path) and **Chunk size** (per-window length, default 480s); the underlying `parakeet-cli` binary and default model resolve from `PARAKEET_CLI` / `PARAKEET_MODEL` (overlap is tunable via `PARAKEET_OVERLAP_SEC`). Run on the CLI with `scripts/parakeet-stitch.mjs --model <gguf> <audio> [out.json]` (prints JSON to stdout when no output path is given). Per-window progress drives the job progress bars. - **Batch jobs estimate the time remaining.** A running batch's overall progress bar on `/jobs/active` now shows an estimate of how long is left (e.g. `~4:30 left`), alongside the existing `Transcripts: 12 / 50` count. The estimate is **remaining tasks × the measured average time per task**: each download/transcription's real wall-clock duration is folded into per-job running totals as it finishes, and that average is converted to a wall-clock figure using the parallelism observed so far (so a 4-way parallel transcribe batch isn't estimated as if it ran one at a time). It reads as **estimating…** until the first sub-operation completes (no average yet), and disappears once no work remains. Jobs without per-task tracking (e.g. storing a playlist) show no estimate. - **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged. - **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template). diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -86,7 +86,7 @@ export function SettingsForm({ initial, apps }: Props) { label="Model" name={`app.${app.id}.model`} defaultValue={cfg.model ?? ""} - hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. Leave blank for the app default." + hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. parakeet: the .gguf model path. Leave blank for the app default." /> )} {app.fields.remoteUrl && ( @@ -105,7 +105,11 @@ export function SettingsForm({ initial, apps }: Props) { cfg.chunkSize !== undefined ? String(cfg.chunkSize) : "" } type="number" - hint="chough only (-c). Leave blank for the app default (60s)." + hint={ + app.id === "parakeet" + ? "parakeet: per-window length in seconds for overlapping splitting. Leave blank for the default (480s)." + : "chough only (-c). Leave blank for the app default (60s)." + } /> )} {app.fields.customArgs && ( diff --git a/editor/e2e/fixtures/bin/fake-parakeet-stitch.mjs b/editor/e2e/fixtures/bin/fake-parakeet-stitch.mjs @@ -0,0 +1,41 @@ +#!/usr/bin/env node +// E2E fake for scripts/parakeet-stitch.mjs (the parakeet overlapping-segment +// wrapper). Mimics the args the parakeet app builds (transcriptionApps.ts): +// parakeet-stitch.mjs --model <m> --output <tmpBase> [--segment <n>] <audioFile> +// and the wrapper's output contract: write chough-native JSON to EXACTLY the +// --output path (no extension appended), and print "segment X/Y" progress lines +// to stderr (consumed by createParakeetProgressParser). Run from cwd == the +// video dir. Stays instant — it never shells out to ffmpeg/parakeet-cli. +import { writeFile } from "node:fs/promises"; + +const argv = process.argv.slice(2); +function arg(flag) { + const i = argv.indexOf(flag); + return i < 0 ? undefined : argv[i + 1]; +} + +const out = arg("--output") ?? arg("-o"); +const audio = argv[argv.length - 1]; +if (!out) { + process.stderr.write(`[fake-parakeet-stitch] missing --output\n`); + process.exit(2); +} + +// Two synthetic windows so the stitched output and progress lines are exercised. +process.stderr.write(`parakeet-stitch: ${audio}: 10s -> 2 window(s)\n`); +process.stderr.write(`parakeet-stitch: segment 1/2 @0s — transcribing\n`); +process.stderr.write(`parakeet-stitch: segment 2/2 @5s — transcribing\n`); + +const doc = { + duration_seconds: 10, + chunks: 2, + text: `Synthetic parakeet output for ${audio} #1 Synthetic parakeet output for ${audio} #2`, + chunk_data: [ + { start_time: 0, end_time: 5, text: `Synthetic parakeet output for ${audio} #1` }, + { start_time: 5, end_time: 10, text: `Synthetic parakeet output for ${audio} #2` }, + ], +}; + +// The wrapper writes EXACTLY the --output path (same as chough's -o). +await writeFile(out, JSON.stringify(doc)); +process.stderr.write(`parakeet-stitch: wrote 2 cues -> ${out}\n`); diff --git a/editor/e2e/parakeet.spec.ts b/editor/e2e/parakeet.spec.ts @@ -0,0 +1,83 @@ +import { readFile, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath, writeSettings } from "./helpers"; + +const CHANNEL = "test-transcribe"; +const DATA = `test-transcripts/channels/${CHANNEL}/data`; + +// Minimal metadata so transcribeOne's normalize pass runs (it needs +// metadata.info.json) and writes transcript.cues.json. +const META = JSON.stringify({ + id: "vidA", + title: "Synthetic parakeet video", + upload_date: "20240101", + duration: 10, +}); + +async function selectParakeet() { + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + transcriptionApp: "parakeet", + transcriptionApps: { parakeet: {} }, + }); +} + +test("parakeet app transcribes audio and writes a transcript.json", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await selectParakeet(); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Transcribe missing" }).click(); + await expect(page.getByLabel("Transcribe missing output")).toContainText( + "3 succeeded", + { timeout: 30_000 }, + ); + + for (const id of ["vidA", "vidB", "vidC"]) { + expect(await pathExists(`${DATA}/${id}/transcript.json`)).toBe(true); + } + + // The wrapper emits chough-native JSON (chunk_data in seconds). Reaching this + // shape proves the parakeet app's argv ran AND that transcribeOne correctly + // renamed the wrapper's exact `--output` path (no ".json" appended) to + // transcript.json. + const raw = JSON.parse( + await readFile(resolvePath(`${DATA}/vidA/transcript.json`), "utf8"), + ); + expect(Array.isArray(raw.chunk_data)).toBe(true); + expect(raw.transcription).toBeUndefined(); +}); + +test("parakeet output normalizes to chough-tagged cues", async ({ page }) => { + await resetData("one-transcribe-channel-with-audio"); + await selectParakeet(); + // Give vidA metadata so the normalize pass produces transcript.cues.json. + await writeFile(resolvePath(`${DATA}/vidA/metadata.info.json`), META); + + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Transcribe missing" }).click(); + await expect(page.getByLabel("Transcribe missing output")).toContainText( + "3 succeeded", + { timeout: 30_000 }, + ); + + await expect + .poll(() => pathExists(`${DATA}/vidA/transcript.cues.json`), { + timeout: 15_000, + }) + .toBe(true); + + const cues = JSON.parse( + await readFile(resolvePath(`${DATA}/vidA/transcript.cues.json`), "utf8"), + ); + // The stitched output is chough-shaped, so it carries the chough-json tag… + expect(cues.transcriptFormat).toBe("chough-json"); + // …and the cues come from seconds-based chunk_data (start 0, end 5), not + // misread as whisper milliseconds. + expect(cues.cues.length).toBeGreaterThan(0); + expect(cues.cues[0].start).toBe(0); + expect(cues.cues[0].end).toBe(5); +}); diff --git a/editor/package.json b/editor/package.json @@ -5,8 +5,8 @@ "type": "module", "scripts": { "dev": "next dev --port 3001", - "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011", - "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011", + "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011", + "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011", "build": "next build", "start": "next start --port 3001", "lint": "eslint", diff --git a/scripts/parakeet-stitch.mjs b/scripts/parakeet-stitch.mjs @@ -0,0 +1,293 @@ +#!/usr/bin/env node +// Standalone parakeet.cpp transcription with overlapped splitting + stitching. +// +// parakeet-cli only accepts a single WAV and chokes on very long audio (dr_wav's +// data-chunk size is 32-bit, so >4GB PCM overflows and yields an empty +// transcript). This script works around both: it slices the source into +// OVERLAPPING 16kHz-mono WAV windows (the format parakeet-cli resamples to +// internally anyway), transcribes each with `--json --timestamps`, then stitches +// the per-window word timestamps back into one transcript. The overlap means a +// word clipped at one window boundary is always captured whole by the +// neighbouring window; a per-boundary cut point in the middle of each overlap +// region decides which window owns each word, so nothing is dropped or doubled. +// +// Dual-use, by design: +// * In the app: common/lib/transcriptionApps.ts registers a `parakeet` app +// whose binary IS this script. transcribeOne runs it with cwd == the video +// dir and an explicit output path, then renames + normalizes the result. +// * On the CLI: run it directly. With no output path it prints JSON to stdout. +// +// Output is chough-native JSON ({ chunk_data: [{ start_time, end_time, text }] }, +// seconds) so it flows through the existing normalizeTranscript("chough-json") +// path with zero downstream changes. +// +// Usage: +// parakeet-stitch.mjs --model <gguf> [options] <audioFile> [outputFile] +// +// Options (env fallback in parens): +// --model <path> parakeet .gguf model (PARAKEET_MODEL) [required] +// --cli <path> parakeet-cli binary (PARAKEET_CLI, default "parakeet-cli") +// --ffmpeg <path> ffmpeg binary (FFMPEG_BIN, default "ffmpeg") +// --ffprobe <path> ffprobe binary (FFPROBE_BIN, default derived from ffmpeg) +// --segment <sec> window length (PARAKEET_SEGMENT_SEC, default 480) +// --overlap <sec> window overlap (PARAKEET_OVERLAP_SEC, default 6) +// --decoder <ctc|tdt> passed through to parakeet-cli (PARAKEET_DECODER) +// --lang <locale> passed through to parakeet-cli (PARAKEET_LANG) +// --gap <sec> start a new cue when the inter-word gap exceeds this (default 0.8) +// --max-cue <sec> cap a single cue's duration (default 8) +// -o, --output <f> output file (alternative to the positional arg) +// --keep-temp keep the temp working dir (for debugging) +// -h, --help show this help + +import { execFile } from "node:child_process"; +import { promisify } from "node:util"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { existsSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const execFileP = promisify(execFile); + +function fail(msg, code = 2) { + process.stderr.write(`parakeet-stitch: ${msg}\n`); + process.exit(code); +} + +function progress(msg) { + // stderr so it never contaminates stdout JSON, and so the app's progress + // parser (createParakeetProgressParser) can read it off the merged stream. + process.stderr.write(`parakeet-stitch: ${msg}\n`); +} + +const HELP = `parakeet-stitch.mjs --model <gguf> [options] <audioFile> [outputFile] + +Slices audio into overlapping 16kHz-mono windows, transcribes each with +parakeet-cli --json --timestamps, and stitches the word timestamps into one +chough-native JSON transcript. With no outputFile, prints JSON to stdout. + +See the header of this file for the full option list.`; + +function parseArgs(argv) { + const opts = { + model: process.env.PARAKEET_MODEL ?? "", + cli: process.env.PARAKEET_CLI ?? "parakeet-cli", + ffmpeg: process.env.FFMPEG_BIN ?? "ffmpeg", + ffprobe: process.env.FFPROBE_BIN ?? "", + segment: numEnv(process.env.PARAKEET_SEGMENT_SEC, 480), + overlap: numEnv(process.env.PARAKEET_OVERLAP_SEC, 6), + decoder: process.env.PARAKEET_DECODER ?? "", + lang: process.env.PARAKEET_LANG ?? "", + gap: 0.8, + maxCue: 8, + keepTemp: false, + output: undefined, + audio: undefined, + }; + const positional = []; + for (let i = 0; i < argv.length; i += 1) { + const a = argv[i]; + const next = () => { + const v = argv[++i]; + if (v === undefined) fail(`missing value for ${a}`); + return v; + }; + switch (a) { + case "-h": case "--help": process.stdout.write(HELP + "\n"); process.exit(0); + case "--model": opts.model = next(); break; + case "--cli": opts.cli = next(); break; + case "--ffmpeg": opts.ffmpeg = next(); break; + case "--ffprobe": opts.ffprobe = next(); break; + case "--segment": opts.segment = Number(next()); break; + case "--overlap": opts.overlap = Number(next()); break; + case "--decoder": opts.decoder = next(); break; + case "--lang": opts.lang = next(); break; + case "--gap": opts.gap = Number(next()); break; + case "--max-cue": opts.maxCue = Number(next()); break; + case "-o": case "--output": opts.output = next(); break; + case "--keep-temp": opts.keepTemp = true; break; + default: + if (a.startsWith("-")) fail(`unknown option ${a}`); + positional.push(a); + } + } + opts.audio = positional[0]; + if (opts.output === undefined) opts.output = positional[1]; + return opts; +} + +function numEnv(v, dflt) { + if (v === undefined || v === "") return dflt; + const n = Number(v); + return Number.isFinite(n) ? n : dflt; +} + +const round3 = (n) => Math.round(n * 1000) / 1000; + +// Locate ffprobe: explicit flag/env wins, else swap a trailing "ffmpeg" in the +// ffmpeg path for "ffprobe" (handles absolute paths), else fall back to PATH. +function resolveFfprobe(opts) { + if (opts.ffprobe) return opts.ffprobe; + if (/ffmpeg$/.test(opts.ffmpeg)) return opts.ffmpeg.replace(/ffmpeg$/, "ffprobe"); + return "ffprobe"; +} + +async function probeDuration(ffprobe, audio) { + const { stdout } = await execFileP(ffprobe, [ + "-v", "error", + "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", + audio, + ]); + const dur = Number.parseFloat(stdout.trim()); + if (!Number.isFinite(dur) || dur <= 0) { + throw new Error(`could not determine duration of ${audio} (got "${stdout.trim()}")`); + } + return dur; +} + +// Slice [start, start+len) of the source into a 16kHz-mono s16 WAV. Input +// seeking (-ss before -i) is fast and accurate enough given the overlap. +async function sliceWav(ffmpeg, audio, start, len, outWav) { + const args = ["-nostdin", "-v", "error", "-y"]; + if (start > 0) args.push("-ss", String(start)); + args.push("-i", audio); + if (len !== undefined) args.push("-t", String(len)); + args.push("-ac", "1", "-ar", "16000", "-c:a", "pcm_s16le", "-f", "wav", outWav); + await execFileP(ffmpeg, args); +} + +// Run parakeet-cli on one window and return its parsed JSON document. +async function transcribeWindow(opts, wav) { + const args = ["transcribe", "--model", opts.model, "--timestamps", "--json", "--input", wav]; + if (opts.decoder) args.push("--decoder", opts.decoder); + if (opts.lang) args.push("--lang", opts.lang); + const { stdout } = await execFileP(opts.cli, args, { maxBuffer: 256 * 1024 * 1024 }); + return parseCliJson(stdout); +} + +// parakeet-cli prints the JSON document to stdout; model-load chatter goes to +// stderr. Be defensive anyway and carve out the outermost { ... }. +function parseCliJson(stdout) { + const s = stdout.indexOf("{"); + const e = stdout.lastIndexOf("}"); + if (s < 0 || e <= s) throw new Error(`no JSON in parakeet-cli output: ${stdout.slice(0, 200)}`); + return JSON.parse(stdout.slice(s, e + 1)); +} + +// Group absolute-timestamped words into cues: break on a large inter-word gap or +// when a cue would exceed maxCue seconds. +function groupCues(words, { gap, maxCue }) { + const cues = []; + let cur = null; + for (const w of words) { + const text = (w.w ?? w.text ?? "").trim(); + if (!text) continue; + if (cur && (w.start - cur.end > gap || w.end - cur.start > maxCue)) { + cues.push(cur); + cur = null; + } + if (!cur) cur = { start: w.start, end: w.end, words: [text] }; + else { cur.end = w.end; cur.words.push(text); } + } + if (cur) cues.push(cur); + return cues.map((c) => ({ + start_time: round3(c.start), + end_time: round3(c.end), + text: c.words.join(" "), + })); +} + +async function main() { + const opts = parseArgs(process.argv.slice(2)); + if (!opts.audio) fail("missing <audioFile>\n\n" + HELP); + if (!opts.model) fail("missing --model (or PARAKEET_MODEL)"); + if (!existsSync(opts.audio)) fail(`audio file not found: ${opts.audio}`); + if (!existsSync(opts.model)) fail(`model not found: ${opts.model}`); + if (!(opts.segment > 0)) fail(`--segment must be > 0 (got ${opts.segment})`); + if (!(opts.overlap >= 0)) fail(`--overlap must be >= 0 (got ${opts.overlap})`); + // Overlap must be strictly less than the window or the windows never advance. + const overlap = Math.min(opts.overlap, opts.segment - 1); + const step = opts.segment - overlap; + + const ffprobe = resolveFfprobe(opts); + const duration = await probeDuration(ffprobe, opts.audio); + + // Window offsets: 0, step, 2*step, ... while there is audio left to cover. + const offsets = []; + for (let t = 0; t < duration; t += step) { + offsets.push(t); + if (t + opts.segment >= duration) break; // this window reaches the end + } + const N = offsets.length; + progress(`${path.basename(opts.audio)}: ${round3(duration)}s -> ${N} window(s) of ${opts.segment}s (overlap ${overlap}s)`); + + const tmp = await mkdtemp(path.join(os.tmpdir(), "parakeet-stitch-")); + // Per-window absolute-timestamped word lists. + const windowWords = []; + try { + for (let i = 0; i < N; i += 1) { + const start = offsets[i]; + const isLast = i === N - 1; + // Last window runs to EOF (no -t) so we never miss a trailing fragment. + const len = isLast ? undefined : opts.segment; + const wav = path.join(tmp, `win-${String(i).padStart(4, "0")}.wav`); + progress(`segment ${i + 1}/${N} @${round3(start)}s — slicing`); + await sliceWav(opts.ffmpeg, opts.audio, start, len, wav); + progress(`segment ${i + 1}/${N} @${round3(start)}s — transcribing`); + const doc = await transcribeWindow(opts, wav); + const words = Array.isArray(doc.words) ? doc.words : []; + // Shift window-relative times into absolute timeline. + windowWords.push( + words.map((w) => ({ + w: w.w ?? w.text ?? "", + start: (Number(w.start) || 0) + start, + end: (Number(w.end) || 0) + start, + conf: w.conf, + })), + ); + if (!opts.keepTemp) await rm(wav, { force: true }); + } + } finally { + if (!opts.keepTemp) await rm(tmp, { recursive: true, force: true }); + } + + // Boundary cut points: the midpoint of each overlap region. Window i owns + // words whose start is in [cut[i-1], cut[i]); the first/last windows are + // open-ended. This keeps exactly one copy of every word across the seams. + const cuts = []; + for (let i = 0; i < N - 1; i += 1) { + const nextStart = offsets[i + 1]; + cuts.push(nextStart + overlap / 2); + } + const stitched = []; + for (let i = 0; i < N; i += 1) { + const lo = i > 0 ? cuts[i - 1] : -Infinity; + const hi = i < N - 1 ? cuts[i] : Infinity; + for (const w of windowWords[i]) { + if (w.start >= lo && w.start < hi) stitched.push(w); + } + } + stitched.sort((a, b) => a.start - b.start); + + const chunkData = groupCues(stitched, { gap: opts.gap, maxCue: opts.maxCue }); + const text = chunkData.map((c) => c.text).join(" "); + const doc = { + duration_seconds: round3(duration), + chunks: chunkData.length, + text, + chunk_data: chunkData, + }; + + const json = JSON.stringify(doc); + if (opts.output) { + await writeFile(opts.output, json); + progress(`wrote ${chunkData.length} cues (${stitched.length} words) -> ${opts.output}`); + } else { + process.stdout.write(json + "\n"); + progress(`emitted ${chunkData.length} cues (${stitched.length} words) to stdout`); + } +} + +main().catch((err) => { + fail(err?.stack || String(err), 1); +});