commit add79a2fc87ea8949bc3178c6b1d3f66807458c0
parent 2898c1f6816cb53b882bdac851e106af7420dc4e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 13 Jun 2026 13:11:55 -0400
parakeet stitch test
Diffstat:
9 files changed, 492 insertions(+), 4 deletions(-)
diff --git a/common/jobs/progressParsers.ts b/common/jobs/progressParsers.ts
@@ -200,3 +200,22 @@ export function createChoughProgressParser(): {
},
};
}
+
+// parakeet-stitch.mjs (the overlapping-segment wrapper) prints one stderr line
+// per window, e.g.
+// parakeet-stitch: segment 3/12 @360s — transcribing
+// Progress = completed segment index / total segments.
+export function createParakeetProgressParser(): {
+ feed: (line: string) => ProgressUpdate | null;
+} {
+ return {
+ feed(line: string): ProgressUpdate | null {
+ const m = line.match(/segment\s+(\d+)\s*\/\s*(\d+)/i);
+ if (!m) return null;
+ const i = Number.parseInt(m[1], 10);
+ const n = Number.parseInt(m[2], 10);
+ if (!Number.isFinite(n) || n <= 0) return null;
+ return { fraction: clamp01(i / n), detail: `segment ${i}/${n}` };
+ },
+ };
+}
diff --git a/common/lib/paths.ts b/common/lib/paths.ts
@@ -37,6 +37,12 @@ export type Paths = {
whisperBin: string;
whisperModel: string;
ffmpegBin: string;
+ // parakeet (overlapping-segment stitching) app. parakeetBin is the standalone
+ // wrapper script invoked as the app binary; parakeetCliBin is the underlying
+ // parakeet-cli it drives; parakeetModel is the default .gguf model.
+ parakeetBin: string;
+ parakeetCliBin: string;
+ parakeetModel: string;
};
let cached: Paths | null = null;
@@ -91,6 +97,11 @@ export function getPaths(): Paths {
"ggml-base.en.bin",
),
ffmpegBin: process.env.FFMPEG_BIN ?? "ffmpeg",
+ parakeetBin:
+ process.env.PARAKEET_STITCH_BIN ??
+ path.join(monorepoRoot, "scripts", "parakeet-stitch.mjs"),
+ parakeetCliBin: process.env.PARAKEET_CLI ?? "parakeet-cli",
+ parakeetModel: process.env.PARAKEET_MODEL ?? "",
};
return cached;
}
diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts
@@ -15,6 +15,7 @@ import {
type ProgressUpdate,
createTranscribeProgressParser,
createChoughProgressParser,
+ createParakeetProgressParser,
} from "../jobs/progressParsers";
export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt";
@@ -197,9 +198,44 @@ const chough: TranscriptionApp = {
makeProgressParser: createChoughProgressParser,
};
+// parakeet.cpp via the overlapping-segment wrapper (scripts/parakeet-stitch.mjs).
+// The wrapper IS the binary here: it slices the audio into overlapping 16kHz-mono
+// windows, runs parakeet-cli per window, and stitches the word timestamps into a
+// single chough-native JSON document written to the exact `-o` path (no extension
+// appended — same contract as chough). `model` is the .gguf; `chunkSize` is the
+// per-window length in seconds (overlap is wrapper-defaulted, tunable via the
+// PARAKEET_OVERLAP_SEC env).
+const parakeet: TranscriptionApp = {
+ id: "parakeet",
+ label: "parakeet.cpp (overlapping segments)",
+ fields: { model: true, chunkSize: true },
+ defaultBin: () => getPaths().parakeetBin,
+ build({ audioFile, outputBase, config }) {
+ const paths = getPaths();
+ const model = config.model?.trim() || paths.parakeetModel;
+ const argv = ["--model", model, "--output", outputBase];
+ if (typeof config.chunkSize === "number" && config.chunkSize > 0) {
+ argv.push("--segment", String(Math.floor(config.chunkSize)));
+ }
+ argv.push(audioFile);
+ return {
+ argv,
+ env: {
+ PARAKEET_CLI: paths.parakeetCliBin,
+ FFMPEG_BIN: paths.ffmpegBin,
+ },
+ // The wrapper writes EXACTLY the --output path (like chough's -o).
+ outputFile: outputBase,
+ outputFormat: "chough-json",
+ };
+ },
+ makeProgressParser: createParakeetProgressParser,
+};
+
export const TRANSCRIPTION_APPS: Record<string, TranscriptionApp> = {
[whisperCpp.id]: whisperCpp,
[chough.id]: chough,
+ [parakeet.id]: parakeet,
};
export const DEFAULT_TRANSCRIPTION_APP_ID = "whisper-cpp";
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **New transcription app: parakeet.cpp with overlapping-segment stitching.** A third **App** option (alongside whisper.cpp and chough) in **Settings → Transcription** transcribes via `parakeet-cli`, working around its two long-audio limitations: it only accepts a single WAV, and a >4GB PCM stream silently yields an empty transcript (dr_wav's data-chunk size is 32-bit). A standalone wrapper (`scripts/parakeet-stitch.mjs`) — usable both as the app's binary and directly on the CLI — slices the source into **overlapping** 16kHz-mono windows (the format parakeet-cli resamples to internally anyway), runs `parakeet-cli --json --timestamps` on each, then stitches the per-window word timestamps back into one transcript. The overlap guarantees a word clipped at one window's boundary is captured whole by the neighbour; a cut point in the middle of each overlap region decides which window owns each word, so nothing is dropped or duplicated across seams. Output is chough-native JSON (seconds-based `chunk_data`), so it flows through the existing format-aware normalize/index path with no downstream changes and is tagged `chough-json` in `transcript.cues.json`. The settings fields are **Model** (the `.gguf` path) and **Chunk size** (per-window length, default 480s); the underlying `parakeet-cli` binary and default model resolve from `PARAKEET_CLI` / `PARAKEET_MODEL` (overlap is tunable via `PARAKEET_OVERLAP_SEC`). Run on the CLI with `scripts/parakeet-stitch.mjs --model <gguf> <audio> [out.json]` (prints JSON to stdout when no output path is given). Per-window progress drives the job progress bars.
- **Batch jobs estimate the time remaining.** A running batch's overall progress bar on `/jobs/active` now shows an estimate of how long is left (e.g. `~4:30 left`), alongside the existing `Transcripts: 12 / 50` count. The estimate is **remaining tasks × the measured average time per task**: each download/transcription's real wall-clock duration is folded into per-job running totals as it finishes, and that average is converted to a wall-clock figure using the parallelism observed so far (so a 4-way parallel transcribe batch isn't estimated as if it ran one at a time). It reads as **estimating…** until the first sub-operation completes (no average yet), and disappears once no work remains. Jobs without per-task tracking (e.g. storing a playlist) show no estimate.
- **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged.
- **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template).
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -86,7 +86,7 @@ export function SettingsForm({ initial, apps }: Props) {
label="Model"
name={`app.${app.id}.model`}
defaultValue={cfg.model ?? ""}
- hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. Leave blank for the app default."
+ hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. parakeet: the .gguf model path. Leave blank for the app default."
/>
)}
{app.fields.remoteUrl && (
@@ -105,7 +105,11 @@ export function SettingsForm({ initial, apps }: Props) {
cfg.chunkSize !== undefined ? String(cfg.chunkSize) : ""
}
type="number"
- hint="chough only (-c). Leave blank for the app default (60s)."
+ hint={
+ app.id === "parakeet"
+ ? "parakeet: per-window length in seconds for overlapping splitting. Leave blank for the default (480s)."
+ : "chough only (-c). Leave blank for the app default (60s)."
+ }
/>
)}
{app.fields.customArgs && (
diff --git a/editor/e2e/fixtures/bin/fake-parakeet-stitch.mjs b/editor/e2e/fixtures/bin/fake-parakeet-stitch.mjs
@@ -0,0 +1,41 @@
+#!/usr/bin/env node
+// E2E fake for scripts/parakeet-stitch.mjs (the parakeet overlapping-segment
+// wrapper). Mimics the args the parakeet app builds (transcriptionApps.ts):
+// parakeet-stitch.mjs --model <m> --output <tmpBase> [--segment <n>] <audioFile>
+// and the wrapper's output contract: write chough-native JSON to EXACTLY the
+// --output path (no extension appended), and print "segment X/Y" progress lines
+// to stderr (consumed by createParakeetProgressParser). Run from cwd == the
+// video dir. Stays instant — it never shells out to ffmpeg/parakeet-cli.
+import { writeFile } from "node:fs/promises";
+
+const argv = process.argv.slice(2);
+function arg(flag) {
+ const i = argv.indexOf(flag);
+ return i < 0 ? undefined : argv[i + 1];
+}
+
+const out = arg("--output") ?? arg("-o");
+const audio = argv[argv.length - 1];
+if (!out) {
+ process.stderr.write(`[fake-parakeet-stitch] missing --output\n`);
+ process.exit(2);
+}
+
+// Two synthetic windows so the stitched output and progress lines are exercised.
+process.stderr.write(`parakeet-stitch: ${audio}: 10s -> 2 window(s)\n`);
+process.stderr.write(`parakeet-stitch: segment 1/2 @0s — transcribing\n`);
+process.stderr.write(`parakeet-stitch: segment 2/2 @5s — transcribing\n`);
+
+const doc = {
+ duration_seconds: 10,
+ chunks: 2,
+ text: `Synthetic parakeet output for ${audio} #1 Synthetic parakeet output for ${audio} #2`,
+ chunk_data: [
+ { start_time: 0, end_time: 5, text: `Synthetic parakeet output for ${audio} #1` },
+ { start_time: 5, end_time: 10, text: `Synthetic parakeet output for ${audio} #2` },
+ ],
+};
+
+// The wrapper writes EXACTLY the --output path (same as chough's -o).
+await writeFile(out, JSON.stringify(doc));
+process.stderr.write(`parakeet-stitch: wrote 2 cues -> ${out}\n`);
diff --git a/editor/e2e/parakeet.spec.ts b/editor/e2e/parakeet.spec.ts
@@ -0,0 +1,83 @@
+import { readFile, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { pathExists, resetData, resolvePath, writeSettings } from "./helpers";
+
+const CHANNEL = "test-transcribe";
+const DATA = `test-transcripts/channels/${CHANNEL}/data`;
+
+// Minimal metadata so transcribeOne's normalize pass runs (it needs
+// metadata.info.json) and writes transcript.cues.json.
+const META = JSON.stringify({
+ id: "vidA",
+ title: "Synthetic parakeet video",
+ upload_date: "20240101",
+ duration: 10,
+});
+
+async function selectParakeet() {
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ transcriptionApp: "parakeet",
+ transcriptionApps: { parakeet: {} },
+ });
+}
+
+test("parakeet app transcribes audio and writes a transcript.json", async ({
+ page,
+}) => {
+ await resetData("one-transcribe-channel-with-audio");
+ await selectParakeet();
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+ await expect(page.getByLabel("Transcribe missing output")).toContainText(
+ "3 succeeded",
+ { timeout: 30_000 },
+ );
+
+ for (const id of ["vidA", "vidB", "vidC"]) {
+ expect(await pathExists(`${DATA}/${id}/transcript.json`)).toBe(true);
+ }
+
+ // The wrapper emits chough-native JSON (chunk_data in seconds). Reaching this
+ // shape proves the parakeet app's argv ran AND that transcribeOne correctly
+ // renamed the wrapper's exact `--output` path (no ".json" appended) to
+ // transcript.json.
+ const raw = JSON.parse(
+ await readFile(resolvePath(`${DATA}/vidA/transcript.json`), "utf8"),
+ );
+ expect(Array.isArray(raw.chunk_data)).toBe(true);
+ expect(raw.transcription).toBeUndefined();
+});
+
+test("parakeet output normalizes to chough-tagged cues", async ({ page }) => {
+ await resetData("one-transcribe-channel-with-audio");
+ await selectParakeet();
+ // Give vidA metadata so the normalize pass produces transcript.cues.json.
+ await writeFile(resolvePath(`${DATA}/vidA/metadata.info.json`), META);
+
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+ await expect(page.getByLabel("Transcribe missing output")).toContainText(
+ "3 succeeded",
+ { timeout: 30_000 },
+ );
+
+ await expect
+ .poll(() => pathExists(`${DATA}/vidA/transcript.cues.json`), {
+ timeout: 15_000,
+ })
+ .toBe(true);
+
+ const cues = JSON.parse(
+ await readFile(resolvePath(`${DATA}/vidA/transcript.cues.json`), "utf8"),
+ );
+ // The stitched output is chough-shaped, so it carries the chough-json tag…
+ expect(cues.transcriptFormat).toBe("chough-json");
+ // …and the cues come from seconds-based chunk_data (start 0, end 5), not
+ // misread as whisper milliseconds.
+ expect(cues.cues.length).toBeGreaterThan(0);
+ expect(cues.cues[0].start).toBe(0);
+ expect(cues.cues[0].end).toBe(5);
+});
diff --git a/editor/package.json b/editor/package.json
@@ -5,8 +5,8 @@
"type": "module",
"scripts": {
"dev": "next dev --port 3001",
- "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011",
- "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011",
+ "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011",
+ "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null PARAKEET_STITCH_BIN=$(pwd)/e2e/fixtures/bin/fake-parakeet-stitch.mjs PARAKEET_CLI=/dev/null PARAKEET_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011",
"build": "next build",
"start": "next start --port 3001",
"lint": "eslint",
diff --git a/scripts/parakeet-stitch.mjs b/scripts/parakeet-stitch.mjs
@@ -0,0 +1,293 @@
+#!/usr/bin/env node
+// Standalone parakeet.cpp transcription with overlapped splitting + stitching.
+//
+// parakeet-cli only accepts a single WAV and chokes on very long audio (dr_wav's
+// data-chunk size is 32-bit, so >4GB PCM overflows and yields an empty
+// transcript). This script works around both: it slices the source into
+// OVERLAPPING 16kHz-mono WAV windows (the format parakeet-cli resamples to
+// internally anyway), transcribes each with `--json --timestamps`, then stitches
+// the per-window word timestamps back into one transcript. The overlap means a
+// word clipped at one window boundary is always captured whole by the
+// neighbouring window; a per-boundary cut point in the middle of each overlap
+// region decides which window owns each word, so nothing is dropped or doubled.
+//
+// Dual-use, by design:
+// * In the app: common/lib/transcriptionApps.ts registers a `parakeet` app
+// whose binary IS this script. transcribeOne runs it with cwd == the video
+// dir and an explicit output path, then renames + normalizes the result.
+// * On the CLI: run it directly. With no output path it prints JSON to stdout.
+//
+// Output is chough-native JSON ({ chunk_data: [{ start_time, end_time, text }] },
+// seconds) so it flows through the existing normalizeTranscript("chough-json")
+// path with zero downstream changes.
+//
+// Usage:
+// parakeet-stitch.mjs --model <gguf> [options] <audioFile> [outputFile]
+//
+// Options (env fallback in parens):
+// --model <path> parakeet .gguf model (PARAKEET_MODEL) [required]
+// --cli <path> parakeet-cli binary (PARAKEET_CLI, default "parakeet-cli")
+// --ffmpeg <path> ffmpeg binary (FFMPEG_BIN, default "ffmpeg")
+// --ffprobe <path> ffprobe binary (FFPROBE_BIN, default derived from ffmpeg)
+// --segment <sec> window length (PARAKEET_SEGMENT_SEC, default 480)
+// --overlap <sec> window overlap (PARAKEET_OVERLAP_SEC, default 6)
+// --decoder <ctc|tdt> passed through to parakeet-cli (PARAKEET_DECODER)
+// --lang <locale> passed through to parakeet-cli (PARAKEET_LANG)
+// --gap <sec> start a new cue when the inter-word gap exceeds this (default 0.8)
+// --max-cue <sec> cap a single cue's duration (default 8)
+// -o, --output <f> output file (alternative to the positional arg)
+// --keep-temp keep the temp working dir (for debugging)
+// -h, --help show this help
+
+import { execFile } from "node:child_process";
+import { promisify } from "node:util";
+import { mkdtemp, rm, writeFile } from "node:fs/promises";
+import { existsSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+
+const execFileP = promisify(execFile);
+
+function fail(msg, code = 2) {
+ process.stderr.write(`parakeet-stitch: ${msg}\n`);
+ process.exit(code);
+}
+
+function progress(msg) {
+ // stderr so it never contaminates stdout JSON, and so the app's progress
+ // parser (createParakeetProgressParser) can read it off the merged stream.
+ process.stderr.write(`parakeet-stitch: ${msg}\n`);
+}
+
+const HELP = `parakeet-stitch.mjs --model <gguf> [options] <audioFile> [outputFile]
+
+Slices audio into overlapping 16kHz-mono windows, transcribes each with
+parakeet-cli --json --timestamps, and stitches the word timestamps into one
+chough-native JSON transcript. With no outputFile, prints JSON to stdout.
+
+See the header of this file for the full option list.`;
+
+function parseArgs(argv) {
+ const opts = {
+ model: process.env.PARAKEET_MODEL ?? "",
+ cli: process.env.PARAKEET_CLI ?? "parakeet-cli",
+ ffmpeg: process.env.FFMPEG_BIN ?? "ffmpeg",
+ ffprobe: process.env.FFPROBE_BIN ?? "",
+ segment: numEnv(process.env.PARAKEET_SEGMENT_SEC, 480),
+ overlap: numEnv(process.env.PARAKEET_OVERLAP_SEC, 6),
+ decoder: process.env.PARAKEET_DECODER ?? "",
+ lang: process.env.PARAKEET_LANG ?? "",
+ gap: 0.8,
+ maxCue: 8,
+ keepTemp: false,
+ output: undefined,
+ audio: undefined,
+ };
+ const positional = [];
+ for (let i = 0; i < argv.length; i += 1) {
+ const a = argv[i];
+ const next = () => {
+ const v = argv[++i];
+ if (v === undefined) fail(`missing value for ${a}`);
+ return v;
+ };
+ switch (a) {
+ case "-h": case "--help": process.stdout.write(HELP + "\n"); process.exit(0);
+ case "--model": opts.model = next(); break;
+ case "--cli": opts.cli = next(); break;
+ case "--ffmpeg": opts.ffmpeg = next(); break;
+ case "--ffprobe": opts.ffprobe = next(); break;
+ case "--segment": opts.segment = Number(next()); break;
+ case "--overlap": opts.overlap = Number(next()); break;
+ case "--decoder": opts.decoder = next(); break;
+ case "--lang": opts.lang = next(); break;
+ case "--gap": opts.gap = Number(next()); break;
+ case "--max-cue": opts.maxCue = Number(next()); break;
+ case "-o": case "--output": opts.output = next(); break;
+ case "--keep-temp": opts.keepTemp = true; break;
+ default:
+ if (a.startsWith("-")) fail(`unknown option ${a}`);
+ positional.push(a);
+ }
+ }
+ opts.audio = positional[0];
+ if (opts.output === undefined) opts.output = positional[1];
+ return opts;
+}
+
+function numEnv(v, dflt) {
+ if (v === undefined || v === "") return dflt;
+ const n = Number(v);
+ return Number.isFinite(n) ? n : dflt;
+}
+
+const round3 = (n) => Math.round(n * 1000) / 1000;
+
+// Locate ffprobe: explicit flag/env wins, else swap a trailing "ffmpeg" in the
+// ffmpeg path for "ffprobe" (handles absolute paths), else fall back to PATH.
+function resolveFfprobe(opts) {
+ if (opts.ffprobe) return opts.ffprobe;
+ if (/ffmpeg$/.test(opts.ffmpeg)) return opts.ffmpeg.replace(/ffmpeg$/, "ffprobe");
+ return "ffprobe";
+}
+
+async function probeDuration(ffprobe, audio) {
+ const { stdout } = await execFileP(ffprobe, [
+ "-v", "error",
+ "-show_entries", "format=duration",
+ "-of", "default=noprint_wrappers=1:nokey=1",
+ audio,
+ ]);
+ const dur = Number.parseFloat(stdout.trim());
+ if (!Number.isFinite(dur) || dur <= 0) {
+ throw new Error(`could not determine duration of ${audio} (got "${stdout.trim()}")`);
+ }
+ return dur;
+}
+
+// Slice [start, start+len) of the source into a 16kHz-mono s16 WAV. Input
+// seeking (-ss before -i) is fast and accurate enough given the overlap.
+async function sliceWav(ffmpeg, audio, start, len, outWav) {
+ const args = ["-nostdin", "-v", "error", "-y"];
+ if (start > 0) args.push("-ss", String(start));
+ args.push("-i", audio);
+ if (len !== undefined) args.push("-t", String(len));
+ args.push("-ac", "1", "-ar", "16000", "-c:a", "pcm_s16le", "-f", "wav", outWav);
+ await execFileP(ffmpeg, args);
+}
+
+// Run parakeet-cli on one window and return its parsed JSON document.
+async function transcribeWindow(opts, wav) {
+ const args = ["transcribe", "--model", opts.model, "--timestamps", "--json", "--input", wav];
+ if (opts.decoder) args.push("--decoder", opts.decoder);
+ if (opts.lang) args.push("--lang", opts.lang);
+ const { stdout } = await execFileP(opts.cli, args, { maxBuffer: 256 * 1024 * 1024 });
+ return parseCliJson(stdout);
+}
+
+// parakeet-cli prints the JSON document to stdout; model-load chatter goes to
+// stderr. Be defensive anyway and carve out the outermost { ... }.
+function parseCliJson(stdout) {
+ const s = stdout.indexOf("{");
+ const e = stdout.lastIndexOf("}");
+ if (s < 0 || e <= s) throw new Error(`no JSON in parakeet-cli output: ${stdout.slice(0, 200)}`);
+ return JSON.parse(stdout.slice(s, e + 1));
+}
+
+// Group absolute-timestamped words into cues: break on a large inter-word gap or
+// when a cue would exceed maxCue seconds.
+function groupCues(words, { gap, maxCue }) {
+ const cues = [];
+ let cur = null;
+ for (const w of words) {
+ const text = (w.w ?? w.text ?? "").trim();
+ if (!text) continue;
+ if (cur && (w.start - cur.end > gap || w.end - cur.start > maxCue)) {
+ cues.push(cur);
+ cur = null;
+ }
+ if (!cur) cur = { start: w.start, end: w.end, words: [text] };
+ else { cur.end = w.end; cur.words.push(text); }
+ }
+ if (cur) cues.push(cur);
+ return cues.map((c) => ({
+ start_time: round3(c.start),
+ end_time: round3(c.end),
+ text: c.words.join(" "),
+ }));
+}
+
+async function main() {
+ const opts = parseArgs(process.argv.slice(2));
+ if (!opts.audio) fail("missing <audioFile>\n\n" + HELP);
+ if (!opts.model) fail("missing --model (or PARAKEET_MODEL)");
+ if (!existsSync(opts.audio)) fail(`audio file not found: ${opts.audio}`);
+ if (!existsSync(opts.model)) fail(`model not found: ${opts.model}`);
+ if (!(opts.segment > 0)) fail(`--segment must be > 0 (got ${opts.segment})`);
+ if (!(opts.overlap >= 0)) fail(`--overlap must be >= 0 (got ${opts.overlap})`);
+ // Overlap must be strictly less than the window or the windows never advance.
+ const overlap = Math.min(opts.overlap, opts.segment - 1);
+ const step = opts.segment - overlap;
+
+ const ffprobe = resolveFfprobe(opts);
+ const duration = await probeDuration(ffprobe, opts.audio);
+
+ // Window offsets: 0, step, 2*step, ... while there is audio left to cover.
+ const offsets = [];
+ for (let t = 0; t < duration; t += step) {
+ offsets.push(t);
+ if (t + opts.segment >= duration) break; // this window reaches the end
+ }
+ const N = offsets.length;
+ progress(`${path.basename(opts.audio)}: ${round3(duration)}s -> ${N} window(s) of ${opts.segment}s (overlap ${overlap}s)`);
+
+ const tmp = await mkdtemp(path.join(os.tmpdir(), "parakeet-stitch-"));
+ // Per-window absolute-timestamped word lists.
+ const windowWords = [];
+ try {
+ for (let i = 0; i < N; i += 1) {
+ const start = offsets[i];
+ const isLast = i === N - 1;
+ // Last window runs to EOF (no -t) so we never miss a trailing fragment.
+ const len = isLast ? undefined : opts.segment;
+ const wav = path.join(tmp, `win-${String(i).padStart(4, "0")}.wav`);
+ progress(`segment ${i + 1}/${N} @${round3(start)}s — slicing`);
+ await sliceWav(opts.ffmpeg, opts.audio, start, len, wav);
+ progress(`segment ${i + 1}/${N} @${round3(start)}s — transcribing`);
+ const doc = await transcribeWindow(opts, wav);
+ const words = Array.isArray(doc.words) ? doc.words : [];
+ // Shift window-relative times into absolute timeline.
+ windowWords.push(
+ words.map((w) => ({
+ w: w.w ?? w.text ?? "",
+ start: (Number(w.start) || 0) + start,
+ end: (Number(w.end) || 0) + start,
+ conf: w.conf,
+ })),
+ );
+ if (!opts.keepTemp) await rm(wav, { force: true });
+ }
+ } finally {
+ if (!opts.keepTemp) await rm(tmp, { recursive: true, force: true });
+ }
+
+ // Boundary cut points: the midpoint of each overlap region. Window i owns
+ // words whose start is in [cut[i-1], cut[i]); the first/last windows are
+ // open-ended. This keeps exactly one copy of every word across the seams.
+ const cuts = [];
+ for (let i = 0; i < N - 1; i += 1) {
+ const nextStart = offsets[i + 1];
+ cuts.push(nextStart + overlap / 2);
+ }
+ const stitched = [];
+ for (let i = 0; i < N; i += 1) {
+ const lo = i > 0 ? cuts[i - 1] : -Infinity;
+ const hi = i < N - 1 ? cuts[i] : Infinity;
+ for (const w of windowWords[i]) {
+ if (w.start >= lo && w.start < hi) stitched.push(w);
+ }
+ }
+ stitched.sort((a, b) => a.start - b.start);
+
+ const chunkData = groupCues(stitched, { gap: opts.gap, maxCue: opts.maxCue });
+ const text = chunkData.map((c) => c.text).join(" ");
+ const doc = {
+ duration_seconds: round3(duration),
+ chunks: chunkData.length,
+ text,
+ chunk_data: chunkData,
+ };
+
+ const json = JSON.stringify(doc);
+ if (opts.output) {
+ await writeFile(opts.output, json);
+ progress(`wrote ${chunkData.length} cues (${stitched.length} words) -> ${opts.output}`);
+ } else {
+ process.stdout.write(json + "\n");
+ progress(`emitted ${chunkData.length} cues (${stitched.length} words) to stdout`);
+ }
+}
+
+main().catch((err) => {
+ fail(err?.stack || String(err), 1);
+});