commit 7a97f79fe0f89827e37d6ede12c566abb14778ef
parent 69edae233f4744e52d7391eb8fdd0c3d8b40fd7c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 8 Oct 2026 23:08:18 -0400
ops transcribe: "words": true returns the engine's word timings
parakeet stitches word timestamps from its windows and then kept only the
grouped cues. Its wrapper takes --words to write them beside chunk_data; a
build input's `words` asks for it (parakeet only; the other engines ignore it),
transcribeWithWorker and transcribeOneVideo pass it through, and a file
transcription with "words": true returns words: [{w, start, end, conf?}] on the
source file's clock. [] from an engine without them. Corpus transcriptions do
not ask, so their transcript.json is unchanged.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 137 insertions(+), 1 deletion(-)
diff --git a/common/controller/transcribeFile.test.ts b/common/controller/transcribeFile.test.ts
@@ -133,6 +133,7 @@ const {
transcribeWorkerFilter,
windowOf,
windowWavArgs,
+ wordsFromTranscript,
} = await import("./transcribeFile");
const { getPaths } = await import("../lib/paths");
const paths = getPaths();
@@ -184,6 +185,34 @@ test("workerId and out are checked for shape", () => {
assert.match(err({ out: 1 })!, /"out" must be a non-empty string/);
});
+test("words is a boolean, kept only when true", () => {
+ const parsed = (b: Record<string, unknown>) => parseTranscribeFileBody({ path: MEDIA, ...b });
+ assert.match((parsed({ words: "yes" }) as { error: string }).error, /"words" must be true or false/);
+ const on = parsed({ words: true });
+ assert.ok(on.ok && on.value.words === true);
+ const off = parsed({ words: false });
+ assert.ok(off.ok && !("words" in off.value));
+});
+
+test("wordsFromTranscript shifts the engine's words onto the file's clock", () => {
+ const raw = JSON.stringify({
+ chunk_data: [{ start_time: 0, end_time: 1, text: "um so" }],
+ words: [
+ { w: " um", start: 0.12, end: 0.4, conf: 0.9 },
+ { w: "so", start: 0.5, end: 0.7 },
+ { w: " ", start: 0.8, end: 0.9 },
+ { w: "bad", start: "x", end: 1 },
+ ],
+ });
+ assert.deepEqual(wordsFromTranscript(raw, 120), [
+ { w: "um", start: 120.12, end: 120.4, conf: 0.9 },
+ { w: "so", start: 120.5, end: 120.7 },
+ ]);
+ // Another engine's document, or an older wrapper: no words, not a failure.
+ assert.deepEqual(wordsFromTranscript(JSON.stringify({ chunk_data: [] }), 0), []);
+ assert.deepEqual(wordsFromTranscript("not json", 0), []);
+});
+
// --- the disk checks --------------------------------------------------------
const ctx = { paths, workers: WORKERS };
diff --git a/common/controller/transcribeFile.ts b/common/controller/transcribeFile.ts
@@ -74,6 +74,7 @@ export const TRANSCRIBE_FILE_BODY_KEYS = [
"end",
"workerId",
"out",
+ "words",
] as const;
const AUDIO_NAME = "audio.wav";
@@ -87,8 +88,14 @@ export type TranscribeFileRequest = {
end?: number;
workerId?: string;
out?: string;
+ // True returns the engine's word timestamps as well as the cues. Only an
+ // engine that keeps them (parakeet) answers with any; the rest give none.
+ words?: boolean;
};
+// One word as the engine timed it, on the source file's clock (seconds).
+export type TranscribedWord = { w: string; start: number; end: number; conf?: number };
+
export type TranscribeWorkerInfo = {
id: string;
name: string;
@@ -108,6 +115,8 @@ export type TranscribeFileResult = {
durationMs: number;
cues: Cue[];
text: string;
+ // Present only when the request asked for words: [] when the engine has none.
+ words?: TranscribedWord[];
};
type Check<T> = { ok: true; value: T } | { ok: false; error: string };
@@ -157,6 +166,10 @@ export function parseTranscribeFileBody(
return { ok: false, error: `"out" must be an absolute path (got "${out}")` };
}
}
+ const words = body.words;
+ if (words !== undefined && typeof words !== "boolean") {
+ return { ok: false, error: '"words" must be true or false' };
+ }
return {
ok: true,
value: {
@@ -165,6 +178,7 @@ export function parseTranscribeFileBody(
...(end.value !== undefined ? { end: end.value } : {}),
...(typeof workerId === "string" ? { workerId: workerId.trim() } : {}),
...(typeof out === "string" ? { out: path.resolve(out) } : {}),
+ ...(words === true ? { words: true } : {}),
},
};
}
@@ -364,6 +378,35 @@ export function windowWavArgs(
const ms = (n: number) => Math.round(n * 1000) / 1000;
// Cues from a window start at zero; shift them back onto the source's clock.
+// The top-level `words` an engine asked with `words` wrote (parakeet's
+// wrapper, --words), shifted by `offset` onto the source file's clock. A
+// document without them -- another engine, or an older wrapper -- gives [].
+export function wordsFromTranscript(raw: string, offset: number): TranscribedWord[] {
+ let doc: unknown;
+ try {
+ doc = JSON.parse(raw);
+ } catch {
+ return [];
+ }
+ const list = (doc as { words?: unknown } | null)?.words;
+ if (!Array.isArray(list)) return [];
+ const out: TranscribedWord[] = [];
+ for (const w of list) {
+ if (!w || typeof w !== "object") continue;
+ const { w: text, start, end, conf } = w as Record<string, unknown>;
+ if (typeof text !== "string" || !text.trim()) continue;
+ if (typeof start !== "number" || typeof end !== "number") continue;
+ if (!Number.isFinite(start) || !Number.isFinite(end)) continue;
+ out.push({
+ w: text.trim(),
+ start: ms(start + offset),
+ end: ms(end + offset),
+ ...(typeof conf === "number" && Number.isFinite(conf) ? { conf } : {}),
+ });
+ }
+ return out;
+}
+
export function offsetCues(cues: readonly Cue[], offset: number): Cue[] {
return cues.map((c) => ({
start: ms(c.start + offset),
@@ -445,6 +488,7 @@ export async function runTranscribeFile(
used.worker = w;
},
skipInlineDiarization: true,
+ ...(req.words ? { words: true } : {}),
});
const worker = used.worker;
if (outcome !== "transcribed" || !worker) {
@@ -467,6 +511,7 @@ export async function runTranscribeFile(
durationMs: Date.now() - started,
cues,
text: cues.map((c) => c.text.trim()).filter(Boolean).join(" "),
+ ...(req.words ? { words: wordsFromTranscript(raw, req.start ?? 0) } : {}),
};
} finally {
await rm(scratch, { recursive: true, force: true }).catch(() => {});
diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts
@@ -99,6 +99,9 @@ export type TranscribeOneOptions = {
// one-off file transcription (controller/transcribeFile.ts) runs in a scratch
// dir that is deleted afterwards, so a diarization there is work thrown away.
skipInlineDiarization?: boolean;
+ // True asks the engine to keep word timestamps in transcript.json (see
+ // TranscribeBuildInput.words). Only the one-off file transcription asks.
+ words?: boolean;
};
export type TranscribeOneOutcome = "transcribed" | "already-exists" | "paused";
@@ -197,6 +200,7 @@ export async function transcribeOneVideo(
audioFile: resolvedAudio,
outputBase: tmpBase,
config: appConfig,
+ ...(opts.words ? { words: true } : {}),
});
const child = execa(bin, build.argv, {
cwd: opts.videoDir,
@@ -380,6 +384,8 @@ export type TranscribeWithWorkerOptions = {
onWorker?: (worker: Worker) => void;
// See TranscribeOneOptions.skipInlineDiarization.
skipInlineDiarization?: boolean;
+ // See TranscribeOneOptions.words.
+ words?: boolean;
};
// Acquire a worker from the global pool and transcribe one video through it,
@@ -444,6 +450,7 @@ export async function transcribeWithWorker(
signal: opts.signal,
partialSignal: partialController?.signal,
skipInlineDiarization: opts.skipInlineDiarization,
+ words: opts.words,
});
pool.markSuccess(worker.id);
return outcome;
diff --git a/common/lib/transcriptionApps.test.ts b/common/lib/transcriptionApps.test.ts
@@ -0,0 +1,27 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { getTranscriptionApp } from "./transcriptionApps";
+
+const input = { audioFile: "audio.wav", outputBase: "transcript.tmp", config: { model: "/m.gguf" } };
+
+test("parakeet keeps its words only when asked", () => {
+ const app = getTranscriptionApp("parakeet");
+ assert.ok(!app.build(input).argv.includes("--words"));
+ const asked = app.build({ ...input, words: true }).argv;
+ assert.ok(asked.includes("--words"));
+ // The audio file stays the last argument: the wrapper reads it positionally.
+ assert.equal(asked.at(-1), "audio.wav");
+});
+
+test("an engine with no words to give ignores the ask", () => {
+ for (const id of ["whisper.cpp", "chough"]) {
+ let app;
+ try {
+ app = getTranscriptionApp(id);
+ } catch {
+ continue;
+ }
+ if (app.id !== id) continue;
+ assert.deepEqual(app.build({ ...input, words: true }).argv, app.build(input).argv);
+ }
+});
diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts
@@ -73,6 +73,11 @@ export type TranscribeBuildInput = {
audioFile: string; // basename relative to videoDir
outputBase: string; // e.g. "transcript.tmp-<pid>" (NO extension)
config: AppInstanceConfig;
+ // Ask the engine to keep its word timestamps in the output document (a
+ // top-level `words` array) as well as the grouped cues. Only parakeet has
+ // them to give; the others ignore it. Off for corpus transcriptions, whose
+ // transcript.json would otherwise carry every word twice.
+ words?: boolean;
};
export type TranscriptionApp = {
@@ -244,10 +249,11 @@ const parakeet: TranscriptionApp = {
supportsPartialStop: true,
defaultBin: () => getPaths().parakeetBin,
resolveModel: (config) => config.model?.trim() || getPaths().parakeetModel,
- build({ audioFile, outputBase, config }) {
+ build({ audioFile, outputBase, config, words }) {
const paths = getPaths();
const model = parakeet.resolveModel(config) as string;
const argv = ["--model", model, "--output", outputBase];
+ if (words) argv.push("--words");
if (typeof config.chunkSize === "number" && config.chunkSize > 0) {
argv.push("--segment", String(Math.floor(config.chunkSize)));
}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A file transcription can return word timings.** `pnpm ops transcribe` takes `"words": true` and adds `words: [{w, start, end, conf?}]` to its result, on the source file's clock like the cues. parakeet keeps the word timestamps it stitches from its windows (its wrapper's new `--words` writes them beside the cues); an engine without them answers `[]`. Corpus transcriptions do not ask, so their `transcript.json` is unchanged.
- **A single file can be transcribed through the editor.** `pnpm ops transcribe` (`POST /api/ops/transcribe`) takes `{"path"}` — an absolute path to a local audio or video file — with `"start"`/`"end"` (seconds) for a window, `"workerId"` (a configured local worker; default: the one auto-transcribe would get) and `"out"` (an absolute path for the result). It runs as a "Transcribe file" job on `/jobs`, through the worker pool and the same engine, model and command line the corpus is transcribed with. A window is cut by ffmpeg to a temporary 16 kHz mono WAV, removed afterwards. The result is `{path, window, worker: {id, name, appId, model, device}, transcriptFormat, cues: [{start, end, text}], text, …}`, cue times on the source file's clock; it ends the job's log, is written to `out` when given, and `--wait` prints it on stdout. Refused before any job: a relative or unreadable path, an `end` not after `start`, an unknown or remote worker, a named worker switched off, an `out` inside the corpus (resolved through symlinks, storage locations included). Nothing is written to the corpus or to settings.
- **Page titles no longer carry a paragraph of explanation.** /channels used to open with three lines of prose about where its numbers come from, and at a narrow window with the sidebar open its four buttons took the row and squeezed that prose to one word a line. It is now one status line — how many channels, how old the oldest report is, how many have none — and when the buttons do not fit beside the title they drop below it. Workers, Sites, Review and Operations lose their lead paragraph; Tags, Storage and Saved videos keep one line each, the instruction (rule edits apply at the next index build; media moves from a channel's Storage panel; the keep-latest window is in a channel's settings); the Monitor widget builder loses its own, which repeated the board's. Needs a restart of the editor.
- **Clip windows can be fetched as a list, paced, in one job per platform.** `pnpm ops fetch-windows` (`POST /api/ops/fetch-windows`) takes `{"siteId"}` — every window a site's published reports cite and the disk does not hold — or `{"items": [{"slug", "id", "from", "to", "clipId"?, "reason"?}], "requestedBy", "manifest"?}`, with `"maxHeight"` and `"dryRun"`. A window already on disk is answered at once and joins no job; the rest are grouped by platform queue and fetched as one `fetch-windows` job per platform, so YouTube and Rumble run side by side, each window through the same managed fetch as a single one (cookie policy, auth and HLS retries, provenance). Between two fetches on a platform the job waits that platform's batch gap, at least 30 s and up to half again at random; before each it checks the platform's cooldown and hold and stops when either is set. A 429 backs the platform off and stops the job; one 403 is that window's failure, two in a row back the platform off and stop it. A platform cooling down or held is refused for its group before anything starts. The job is drainable, shows its progress as clip windows, and Retry or running the same body again fetches only what is still missing. A dry run lists the windows per platform, the ones on disk, and the spans no window can fill (a video the source says is deleted, private or members-only, a channel off the site, an unmounted drive). A site's **Reports** tab has **Fetch missing evidence** and **Preview missing evidence** beside **Prepare evidence media**, and umtool's `fetch-via-editor.mjs --all` sends a manifest's whole timeline as one request. A window fetch whose cookie retry runs into a 429 now records the cooldown too. Needs a restart of the editor.
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -529,6 +529,8 @@ export function usage() {
' would get), "out" (an absolute path for the result JSON, never inside the',
" corpus). The result is {path, window, worker: {id, appId, model, device},",
" cues: [{start, end, text}], text, ...}, cue times on the file's own clock.",
+ ' "words": true adds words: [{w, start, end, conf?}] on the same clock -- from',
+ " parakeet, which keeps its word timestamps; [] from an engine that does not.",
" With --wait it is printed on stdout (the response and the log go to",
" stderr), so `pnpm ops transcribe ... --wait | jq -r .text` works.",
"",
diff --git a/scripts/parakeet-stitch.mjs b/scripts/parakeet-stitch.mjs
@@ -48,6 +48,9 @@
// --max-cue <sec> cap a single cue's duration (default 8)
// -o, --output <f> output file (alternative to the positional arg)
// --keep-temp keep the temp working dir (for debugging)
+// --words also write the stitched words, as a top-level
+// `words: [{w, start, end, conf}]` beside chunk_data
+// (seconds; readers of chunk_data ignore it)
// -h, --help show this help
import { execFile } from "node:child_process";
@@ -92,6 +95,7 @@ function parseArgs(argv) {
gap: 0.8,
maxCue: 8,
keepTemp: false,
+ words: false,
output: undefined,
audio: undefined,
};
@@ -118,6 +122,7 @@ function parseArgs(argv) {
case "--max-cue": opts.maxCue = Number(next()); break;
case "-o": case "--output": opts.output = next(); break;
case "--keep-temp": opts.keepTemp = true; break;
+ case "--words": opts.words = true; break;
default:
if (a.startsWith("-")) fail(`unknown option ${a}`);
positional.push(a);
@@ -136,6 +141,19 @@ function numEnv(v, dflt) {
const round3 = (n) => Math.round(n * 1000) / 1000;
+// The stitched words as --words writes them: absolute seconds, empty words
+// dropped, conf kept only when the engine gave one.
+function wordsOut(stitched) {
+ return stitched
+ .filter((w) => String(w.w ?? "").trim() !== "")
+ .map((w) => ({
+ w: String(w.w).trim(),
+ start: round3(w.start),
+ end: round3(w.end),
+ ...(typeof w.conf === "number" ? { conf: w.conf } : {}),
+ }));
+}
+
// Seconds -> m:ss (or h:mm:ss). Used for the per-video ETA in progress lines.
function formatClock(totalSeconds) {
const s = Math.max(0, Math.floor(totalSeconds));
@@ -402,6 +420,7 @@ async function main() {
chunks: chunkData.length,
text,
chunk_data: chunkData,
+ ...(opts.words ? { words: wordsOut(stitched) } : {}),
};
const json = JSON.stringify(doc);