commit c199f78397fc06f359cd903394e52519d34f4d67
parent 3c6332662dd4ae49d5af7fa2301b9afbd499fccf
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 9 Jun 2026 12:12:14 -0400
support multi-app / chough transcription
Diffstat:
17 files changed, 878 insertions(+), 172 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -25,7 +25,7 @@ import type { Dirent, WriteStream } from "node:fs";
import { once } from "node:events";
import { open } from "lmdb";
import { parseVtt, type Cue } from "../lib/vtt";
-import { parseWhisper } from "../lib/whisper";
+import { parseTranscriptJson } from "../lib/whisper";
import { parseLiveChat } from "../lib/liveChat";
import {
summarize,
@@ -494,7 +494,7 @@ export async function buildIndex({
cueList =
s.transcriptKind === "vtt"
? parseVtt(raw)
- : parseWhisper(raw);
+ : parseTranscriptJson(raw);
} catch {
cueList = undefined;
}
diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts
@@ -27,7 +27,7 @@ import type { Paths } from "../lib/paths";
import type { VideoStat } from "../lib/stats";
import { STATS_SCHEMA_VERSION } from "../lib/stats";
import type { Cue } from "../lib/vtt";
-import { parseWhisper } from "../lib/whisper";
+import { parseTranscriptJson } from "../lib/whisper";
import { readVideoFiles, pickIndexTranscript } from "../lib/videoStatus";
import {
DEFAULT_CONTAINMENT_THRESHOLD,
@@ -407,7 +407,7 @@ async function readRawCues(
} catch {
return null;
}
- return picked.kind === "whisper" ? parseWhisper(raw) : lenientVttCues(raw);
+ return picked.kind === "whisper" ? parseTranscriptJson(raw) : lenientVttCues(raw);
}
// Lenient VTT text extraction: unlike parseVtt (which only keeps YouTube
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -6,7 +6,11 @@
import path from "node:path";
import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises";
import { parseVtt, type Cue } from "../lib/vtt";
-import { parseWhisper } from "../lib/whisper";
+import {
+ detectTranscriptFormat,
+ parseTranscriptJson,
+} from "../lib/whisper";
+import type { TranscriptOutputFormat } from "../lib/transcriptionApps";
import { summarize, type RawMetadata } from "../lib/transcripts-server";
import type { TranscriptDetail } from "../lib/transcripts";
import {
@@ -25,6 +29,9 @@ export const CUES_FILE_VERSION = 1;
export type NormalizedTranscript = TranscriptDetail & {
version: number;
source: IndexTranscript["kind"] | "live_chat";
+ // The raw transcript format this was parsed from. Durable per-video record so
+ // a later re-normalize knows how to read the raw file without re-sniffing.
+ transcriptFormat?: TranscriptOutputFormat;
};
export type NormalizedLiveChat = NormalizedTranscript & { source: "live_chat" };
@@ -33,6 +40,10 @@ export type NormalizeOptions = {
videoDir: string;
channelSlug: string;
configName?: string;
+ // Authoritative raw-transcript format, passed by transcribeOne right after a
+ // run (it knows which app produced the file). When omitted, the format is
+ // resolved from the prior cues.json tag, then by content sniff.
+ formatHint?: TranscriptOutputFormat;
log?: (msg: string) => void;
force?: boolean;
};
@@ -90,8 +101,21 @@ export async function normalizeTranscript(
const rawTranscript = await readFile(transcriptPath, "utf8");
let cues: Cue[];
+ let transcriptFormat: TranscriptOutputFormat;
try {
- cues = picked.kind === "vtt" ? parseVtt(rawTranscript) : parseWhisper(rawTranscript);
+ if (picked.kind === "vtt") {
+ cues = parseVtt(rawTranscript);
+ transcriptFormat = "vtt";
+ } else {
+ // Resolve the JSON format: authoritative hint -> recorded per-video tag
+ // -> content sniff -> whisper fallback (the only app before chough).
+ transcriptFormat =
+ opts.formatHint ??
+ (await readRecordedFormat(cuesPath)) ??
+ detectTranscriptFormat(rawTranscript) ??
+ "whisper-json";
+ cues = parseTranscriptJson(rawTranscript, transcriptFormat);
+ }
} catch (err) {
throw new Error(
`Failed to parse ${picked.filename} for ${opts.channelSlug}/${path.basename(opts.videoDir)}: ${(err as Error).message}`,
@@ -101,6 +125,7 @@ export async function normalizeTranscript(
const out: NormalizedTranscript = {
version: CUES_FILE_VERSION,
source: picked.kind,
+ transcriptFormat,
...summary,
cues,
};
@@ -109,11 +134,24 @@ export async function normalizeTranscript(
await writeFile(tmp, JSON.stringify(out));
await rename(tmp, cuesPath);
opts.log?.(
- `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${picked.kind}, ${cues.length} cues)`,
+ `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${transcriptFormat}, ${cues.length} cues)`,
);
return { status: "wrote", cuesPath };
}
+// Read the format recorded in a prior transcript.cues.json, if any. This is the
+// durable per-video tag that lets a re-normalize parse the raw file the same way
+// it was first parsed, without re-sniffing.
+async function readRecordedFormat(
+ cuesPath: string,
+): Promise<TranscriptOutputFormat | undefined> {
+ const prior = await readNormalizedTranscript(cuesPath);
+ const f = prior?.transcriptFormat;
+ return f === "whisper-json" || f === "chough-json" || f === "vtt"
+ ? f
+ : undefined;
+}
+
export async function readNormalizedTranscript(
cuesPath: string,
): Promise<NormalizedTranscript | null> {
diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts
@@ -2,13 +2,8 @@ import path from "node:path";
import fs from "fs-extra";
import { execa } from "execa";
import type { Paths } from "../lib/paths";
-import {
- TRANSCRIBE_PLACEHOLDER_AUDIO,
- TRANSCRIBE_PLACEHOLDER_MODEL,
- TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
- TRANSCRIBE_KNOWN_PLACEHOLDERS,
- getSettings,
-} from "../lib/settings";
+import { getSettings } from "../lib/settings";
+import { getTranscriptionApp } from "../lib/transcriptionApps";
import { normalizeTranscript } from "./normalizeTranscript";
import { isRealAudioFile } from "../lib/videoStatus";
@@ -32,25 +27,6 @@ async function resolveAudioFile(
return [...candidates].sort()[0];
}
-function substitutePlaceholders(
- args: string[],
- values: { audioFile: string; outputBase: string; model: string },
-): string[] {
- return args.map((arg) => {
- return arg.replace(/\{[^}]+\}/g, (token) => {
- if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) {
- throw new Error(
- `Unknown placeholder ${token} in transcribeArgs; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`,
- );
- }
- if (token === TRANSCRIBE_PLACEHOLDER_AUDIO) return values.audioFile;
- if (token === TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE) return values.outputBase;
- if (token === TRANSCRIBE_PLACEHOLDER_MODEL) return values.model;
- return token;
- });
- });
-}
-
export type TranscribeOneOptions = {
paths: Paths;
videoDir: string;
@@ -92,26 +68,37 @@ export async function transcribeOneVideo(
}
const settings = getSettings();
- log(`Transcribe ${opts.videoId} start (${resolvedAudio})`);
+ const app = getTranscriptionApp(settings.transcriptionApp);
+ const appConfig = settings.transcriptionApps[app.id] ?? {};
+ const bin = appConfig.bin?.trim() || app.defaultBin();
+ log(`Transcribe ${opts.videoId} start (${resolvedAudio}) via ${app.id}`);
const start = Date.now();
- // Whisper writes "<outputBase>.json" relative to cwd. Use a tmp name and
- // rename on success so a SIGTERM mid-write can't leave a half-baked file.
+ // Each app writes "<outputBase><ext>" relative to cwd (the ext differs: whisper
+ // appends ".json", chough writes the exact -o path). Use a tmp base and rename
+ // the app-declared output file on success so a SIGTERM mid-write can't leave a
+ // half-baked transcript.json.
const tmpBase = `transcript.tmp-${process.pid}`;
- const args = substitutePlaceholders(settings.transcribeArgs, {
+ const build = app.build({
audioFile: resolvedAudio,
outputBase: tmpBase,
- model: settings.transcribeModel,
+ config: appConfig,
});
- const child = execa(settings.transcribeBin, args, {
+ const child = execa(bin, build.argv, {
cwd: opts.videoDir,
cancelSignal: opts.signal,
all: true,
buffer: false,
+ env: build.env ? { ...process.env, ...build.env } : undefined,
});
child.all?.on("data", (c: Buffer) => log(c.toString("utf8")));
await child;
- const tmpPath = path.join(opts.videoDir, `${tmpBase}.json`);
- await rename(tmpPath, transcriptPath);
+ const producedPath = path.join(opts.videoDir, build.outputFile);
+ if (!(await pathExists(producedPath))) {
+ throw new Error(
+ `transcription with ${app.id} produced no ${build.outputFile} in ${opts.videoDir}`,
+ );
+ }
+ await rename(producedPath, transcriptPath);
log(
`Transcribe ${opts.videoId} done in ${((Date.now() - start) / 1000).toFixed(2)}s`,
);
@@ -122,6 +109,7 @@ export async function transcribeOneVideo(
await normalizeTranscript({
videoDir: opts.videoDir,
channelSlug,
+ formatHint: build.outputFormat,
log,
});
} catch (err) {
diff --git a/common/jobs/progressParsers.ts b/common/jobs/progressParsers.ts
@@ -173,3 +173,30 @@ export function createTranscribeProgressParser(): {
},
};
}
+
+// chough output is bar-based, not segment-timestamp based. It prints a header
+// audio: 28595.7s • chunks: 60s • format: json
+// then repeatedly overwrites a progress bar of full (█) and light (░) blocks
+// with a trailing ETA, e.g.
+// ████░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░ ETA 38m 57s
+// Progress = filled blocks / total blocks; the ETA text is surfaced as detail.
+export function createChoughProgressParser(): {
+ feed: (line: string) => ProgressUpdate | null;
+} {
+ return {
+ feed(line: string): ProgressUpdate | null {
+ // Consume (and ignore) the audio header line.
+ if (/\baudio:\s*[\d.]+s\b/.test(line)) return null;
+ const filled = (line.match(/█/g) ?? []).length;
+ const empty = (line.match(/░/g) ?? []).length;
+ const total = filled + empty;
+ if (total === 0) return null;
+ const out: ProgressUpdate = { fraction: clamp01(filled / total) };
+ // ETA text up to an ANSI clear ("[K") or end of line.
+ const etaMatch = line.match(/ETA\s+([0-9hdms\s]+?)\s*(?:\x1b?\[K|$)/);
+ const eta = etaMatch?.[1]?.trim();
+ if (eta) out.detail = `ETA ${eta}`;
+ return out;
+ },
+ };
+}
diff --git a/common/jobs/taskHooks.ts b/common/jobs/taskHooks.ts
@@ -1,9 +1,8 @@
import type { JobTaskKind } from "./registry";
import type { JobRunContext } from "./streamCommand";
-import {
- parseDownloadProgress,
- createTranscribeProgressParser,
-} from "./progressParsers";
+import { parseDownloadProgress } from "./progressParsers";
+import { getSettings } from "../lib/settings";
+import { getTranscriptionApp } from "../lib/transcriptionApps";
// A handle for one in-flight sub-operation. `onLog` is a drop-in replacement
// for the controller's existing `onLog`: it forwards every line to the shared
@@ -34,8 +33,12 @@ export function makeTaskTracker(
return { onLog: forwardLog, end: () => {} };
}
ctx.addTask({ id, label, kind, startedAt: Date.now() });
+ // Transcription progress output is app-specific (whisper's segment
+ // timestamps vs chough's ETA bars), so pick the active app's parser.
const transcribeParser =
- kind === "transcribe" ? createTranscribeProgressParser() : null;
+ kind === "transcribe"
+ ? getTranscriptionApp(getSettings().transcriptionApp).makeProgressParser()
+ : null;
let ended = false;
const onLog = (line: string) => {
forwardLog(line);
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -1,5 +1,26 @@
import fs from "node:fs";
+import path from "node:path";
import { getPaths } from "./paths";
+import {
+ type AppInstanceConfig,
+ DEFAULT_TRANSCRIBE_ARGS,
+ DEFAULT_TRANSCRIPTION_APP_ID,
+ getTranscriptionApp,
+ TRANSCRIPTION_APPS,
+ validateTranscribeArgs,
+} from "./transcriptionApps";
+
+// Transcribe placeholder/arg helpers now live with the whisper-cpp app in
+// transcriptionApps.ts. Re-exported here so existing import sites keep working.
+export {
+ type AppInstanceConfig,
+ TRANSCRIBE_PLACEHOLDER_AUDIO,
+ TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
+ TRANSCRIBE_PLACEHOLDER_MODEL,
+ TRANSCRIBE_KNOWN_PLACEHOLDERS,
+ DEFAULT_TRANSCRIBE_ARGS,
+ validateTranscribeArgs,
+} from "./transcriptionApps";
// Global, OPERATIONAL settings shared across every site this editor powers.
// Per-site presentation (branding, social links, channel groups, membership)
@@ -10,9 +31,12 @@ export type SiteSettings = {
// from site.json.
adminTitle: string;
maxTranscriptPageBytes: number;
- transcribeBin: string;
- transcribeArgs: string[];
- transcribeModel: string;
+ // Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp"
+ // or "chough"). Selected globally; see common/lib/transcriptionApps.ts.
+ transcriptionApp: string;
+ // Per-app configuration, keyed by app id. Each app reads only its own block;
+ // a missing block means "use the app's defaults".
+ transcriptionApps: Record<string, AppInstanceConfig>;
// Browser spec (e.g. "firefox", "chrome:Default") passed to
// `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the
// managed per-video download flow. Empty string = disabled.
@@ -62,36 +86,14 @@ export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024;
export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024;
export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024;
-export const TRANSCRIBE_PLACEHOLDER_AUDIO = "{audioFile}";
-export const TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE = "{outputBase}";
-export const TRANSCRIBE_PLACEHOLDER_MODEL = "{model}";
-export const TRANSCRIBE_KNOWN_PLACEHOLDERS: ReadonlyArray<string> = [
- TRANSCRIBE_PLACEHOLDER_AUDIO,
- TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
- TRANSCRIBE_PLACEHOLDER_MODEL,
-];
-
-export const DEFAULT_TRANSCRIBE_ARGS: ReadonlyArray<string> = [
- "-ojf",
- "-l",
- "en",
- "-m",
- TRANSCRIBE_PLACEHOLDER_MODEL,
- "-of",
- TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
- TRANSCRIBE_PLACEHOLDER_AUDIO,
-];
-
export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin";
function defaults(): SiteSettings {
- const paths = getPaths();
return {
adminTitle: DEFAULT_ADMIN_TITLE,
maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES,
- transcribeBin: paths.whisperBin,
- transcribeArgs: [...DEFAULT_TRANSCRIBE_ARGS],
- transcribeModel: paths.whisperModel,
+ transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID,
+ transcriptionApps: {},
cookiesFromBrowser: "",
sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS,
parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT,
@@ -196,14 +198,17 @@ export function getSettings(): SiteSettings {
merged.adminTitle = DEFAULT_ADMIN_TITLE;
}
merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes);
- if (!Array.isArray(merged.transcribeArgs) || merged.transcribeArgs.length === 0) {
- merged.transcribeArgs = [...DEFAULT_TRANSCRIBE_ARGS];
+ merged.transcriptionApps = sanitizeTranscriptionApps(merged.transcriptionApps);
+ // Migrate a pre-multi-app settings.json (transcribeBin/transcribeArgs/
+ // transcribeModel, with no transcriptionApp key) onto the app registry.
+ if (parsed.transcriptionApp === undefined) {
+ migrateLegacyTranscription(parsed as LegacyTranscribeFields, merged);
}
- if (typeof merged.transcribeBin !== "string" || !merged.transcribeBin.trim()) {
- merged.transcribeBin = defaults().transcribeBin;
- }
- if (typeof merged.transcribeModel !== "string") {
- merged.transcribeModel = defaults().transcribeModel;
+ if (
+ typeof merged.transcriptionApp !== "string" ||
+ !TRANSCRIPTION_APPS[merged.transcriptionApp]
+ ) {
+ merged.transcriptionApp = DEFAULT_TRANSCRIPTION_APP_ID;
}
if (typeof merged.cookiesFromBrowser !== "string") {
merged.cookiesFromBrowser = "";
@@ -234,33 +239,98 @@ function clampPageBytes(value: unknown): number {
return Math.floor(n);
}
-export function validateTranscribeArgs(args: string[]): string | null {
- if (!Array.isArray(args) || args.length === 0) {
- return "Transcribe args must contain at least one entry";
- }
- const joined = args.join(" ");
- if (!joined.includes(TRANSCRIBE_PLACEHOLDER_AUDIO)) {
- return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_AUDIO} placeholder`;
- }
- if (!joined.includes(TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE)) {
- return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE} placeholder`;
- }
- for (const arg of args) {
- const tokens = arg.match(/\{[^}]+\}/g) ?? [];
- for (const token of tokens) {
- if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) {
- return `Unknown placeholder ${token}; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`;
- }
+type LegacyTranscribeFields = {
+ transcribeBin?: unknown;
+ transcribeArgs?: unknown;
+ transcribeModel?: unknown;
+};
+
+// Coerce a raw settings.transcriptionApps value into a clean keyed map of
+// AppInstanceConfig, dropping unknown/ill-typed fields.
+export function sanitizeTranscriptionApps(
+ value: unknown,
+): Record<string, AppInstanceConfig> {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return {};
+ const out: Record<string, AppInstanceConfig> = {};
+ for (const [id, raw] of Object.entries(value as Record<string, unknown>)) {
+ if (!raw || typeof raw !== "object") continue;
+ const r = raw as Record<string, unknown>;
+ const cfg: AppInstanceConfig = {};
+ if (typeof r.bin === "string") cfg.bin = r.bin;
+ if (typeof r.model === "string") cfg.model = r.model;
+ if (typeof r.remoteUrl === "string") cfg.remoteUrl = r.remoteUrl;
+ if (typeof r.chunkSize === "number" && Number.isFinite(r.chunkSize)) {
+ cfg.chunkSize = Math.floor(r.chunkSize);
}
+ if (Array.isArray(r.customArgs)) cfg.customArgs = r.customArgs.map(String);
+ out[id] = cfg;
+ }
+ return out;
+}
+
+function argsAreDefault(args: string[]): boolean {
+ return (
+ args.length === DEFAULT_TRANSCRIBE_ARGS.length &&
+ args.every((a, i) => a === DEFAULT_TRANSCRIBE_ARGS[i])
+ );
+}
+
+function migrateLegacyTranscription(
+ raw: LegacyTranscribeFields,
+ merged: SiteSettings,
+): void {
+ const bin = typeof raw.transcribeBin === "string" ? raw.transcribeBin.trim() : "";
+ const model = typeof raw.transcribeModel === "string" ? raw.transcribeModel : "";
+ const legacyArgs = Array.isArray(raw.transcribeArgs)
+ ? raw.transcribeArgs.map(String)
+ : [];
+ if (!bin && !model && legacyArgs.length === 0) return; // nothing to migrate
+
+ // If the legacy binary names a known app (e.g. "chough"), adopt that app so
+ // its native arg/output handling applies; otherwise fall back to whisper-cpp,
+ // which preserves the old placeholder-template behavior.
+ const binBase = bin ? path.basename(bin).toLowerCase() : "";
+ const appId =
+ binBase && TRANSCRIPTION_APPS[binBase]
+ ? binBase
+ : DEFAULT_TRANSCRIPTION_APP_ID;
+ const cfg: AppInstanceConfig = { ...merged.transcriptionApps[appId] };
+ if (bin) cfg.bin = bin;
+ if (model) cfg.model = model;
+ // Carry a customArgs template only for whisper-cpp and only when it differs
+ // from the default — a default install migrates to a clean config. The chough
+ // app builds its own argv, so legacy args are dropped for it.
+ if (
+ appId === DEFAULT_TRANSCRIPTION_APP_ID &&
+ legacyArgs.length > 0 &&
+ !argsAreDefault(legacyArgs)
+ ) {
+ cfg.customArgs = legacyArgs;
}
- return null;
+ merged.transcriptionApp = appId;
+ merged.transcriptionApps = { ...merged.transcriptionApps, [appId]: cfg };
}
export async function writeSettings(next: SiteSettings): Promise<void> {
- const argsErr = validateTranscribeArgs(next.transcribeArgs);
- if (argsErr) throw new Error(argsErr);
- if (!next.transcribeBin.trim()) {
- throw new Error("Transcribe binary is required");
+ const appId =
+ typeof next.transcriptionApp === "string" &&
+ TRANSCRIPTION_APPS[next.transcriptionApp]
+ ? next.transcriptionApp
+ : DEFAULT_TRANSCRIPTION_APP_ID;
+ const app = getTranscriptionApp(appId);
+ const transcriptionApps = sanitizeTranscriptionApps(next.transcriptionApps);
+ const activeCfg = transcriptionApps[appId] ?? {};
+ const resolvedBin = activeCfg.bin?.trim() || app.defaultBin();
+ if (!resolvedBin) {
+ throw new Error(`Transcribe binary is required for ${app.label}`);
+ }
+ if (
+ app.fields.customArgs &&
+ activeCfg.customArgs &&
+ activeCfg.customArgs.length > 0
+ ) {
+ const argsErr = validateTranscribeArgs(activeCfg.customArgs);
+ if (argsErr) throw new Error(argsErr);
}
const socialLinks: SocialLink[] = [];
for (const link of parseSocialLinks(next.socialLinks)) {
@@ -281,9 +351,8 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
? next.adminTitle.trim()
: DEFAULT_ADMIN_TITLE,
maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes),
- transcribeBin: next.transcribeBin,
- transcribeModel: next.transcribeModel,
- transcribeArgs: next.transcribeArgs.map((a) => String(a)),
+ transcriptionApp: appId,
+ transcriptionApps,
cookiesFromBrowser:
typeof next.cookiesFromBrowser === "string"
? next.cookiesFromBrowser.trim()
diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts
@@ -0,0 +1,229 @@
+// First-class registry of transcription apps. Each app owns how it builds its
+// argv/env from a resolved per-app config, what file it actually writes (the
+// `-o`/`-of` naming differs between tools), how its raw output is shaped, and
+// how its live progress output is parsed. The editor selects ONE active app
+// globally (settings.transcriptionApp) and stores a small per-app config block
+// (settings.transcriptionApps[id]); transcribeOne resolves the app and runs it.
+//
+// Pure definitions only (no I/O beyond reading process.env / getPaths inside
+// builders) so both the `common` controllers and the editor UI can import this.
+// Dependency direction is one-way: settings.ts -> transcriptionApps.ts ->
+// { paths, jobs/progressParsers }. Keep it that way to avoid import cycles.
+
+import { getPaths } from "./paths";
+import {
+ type ProgressUpdate,
+ createTranscribeProgressParser,
+ createChoughProgressParser,
+} from "../jobs/progressParsers";
+
+export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt";
+
+// Per-app configuration persisted under settings.transcriptionApps[id]. Every
+// field is optional; an app falls back to its own defaults (defaultBin, env).
+export type AppInstanceConfig = {
+ // Binary path/name override. Empty/undefined falls back to app.defaultBin().
+ bin?: string;
+ // whisper.cpp model path (substituted for {model}); for chough this is the
+ // optional CHOUGH_MODEL env (chough auto-downloads a model when unset).
+ model?: string;
+ // chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription.
+ remoteUrl?: string;
+ // chough chunk size in seconds (-c). Undefined = chough's own default.
+ chunkSize?: number;
+ // whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model}
+ // placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS.
+ customArgs?: string[];
+};
+
+export type TranscribeBuild = {
+ // Args passed after the binary.
+ argv: string[];
+ // Extra environment variables merged over process.env for this run.
+ env?: Record<string, string>;
+ // The file the app actually writes, relative to the video dir (cwd).
+ // transcribeOne renames THIS to transcript.json — this is where the
+ // chough-vs-whisper `-o`/`-of` naming difference is absorbed.
+ outputFile: string;
+ // Shape of the raw output, used as the authoritative parse hint downstream.
+ outputFormat: TranscriptOutputFormat;
+};
+
+export type TranscribeBuildInput = {
+ audioFile: string; // basename relative to videoDir
+ outputBase: string; // e.g. "transcript.tmp-<pid>" (NO extension)
+ config: AppInstanceConfig;
+};
+
+export type TranscriptionApp = {
+ id: string;
+ label: string;
+ // Which config fields this app surfaces in the settings UI / consumes.
+ fields: {
+ model?: boolean;
+ remoteUrl?: boolean;
+ chunkSize?: boolean;
+ customArgs?: boolean;
+ };
+ // Default binary when the per-app `bin` override is empty.
+ defaultBin: () => string;
+ build: (input: TranscribeBuildInput) => TranscribeBuild;
+ makeProgressParser: () => { feed: (line: string) => ProgressUpdate | null };
+};
+
+// ---------------------------------------------------------------------------
+// whisper.cpp argv template + placeholders (formerly in settings.ts; kept here
+// so the whisper-cpp app owns its own arg construction). settings.ts re-exports
+// these for backward compatibility with existing import sites.
+// ---------------------------------------------------------------------------
+
+export const TRANSCRIBE_PLACEHOLDER_AUDIO = "{audioFile}";
+export const TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE = "{outputBase}";
+export const TRANSCRIBE_PLACEHOLDER_MODEL = "{model}";
+export const TRANSCRIBE_KNOWN_PLACEHOLDERS: ReadonlyArray<string> = [
+ TRANSCRIBE_PLACEHOLDER_AUDIO,
+ TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
+ TRANSCRIBE_PLACEHOLDER_MODEL,
+];
+
+export const DEFAULT_TRANSCRIBE_ARGS: ReadonlyArray<string> = [
+ "-ojf",
+ "-l",
+ "en",
+ "-m",
+ TRANSCRIBE_PLACEHOLDER_MODEL,
+ "-of",
+ TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE,
+ TRANSCRIBE_PLACEHOLDER_AUDIO,
+];
+
+// Validate a whisper.cpp customArgs template: must reference {audioFile} and
+// {outputBase}, may reference {model}, and must contain no unknown placeholders.
+// Returns an error string or null when valid.
+export function validateTranscribeArgs(args: string[]): string | null {
+ if (!Array.isArray(args) || args.length === 0) {
+ return "Transcribe args must contain at least one entry";
+ }
+ const joined = args.join(" ");
+ if (!joined.includes(TRANSCRIBE_PLACEHOLDER_AUDIO)) {
+ return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_AUDIO} placeholder`;
+ }
+ if (!joined.includes(TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE)) {
+ return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE} placeholder`;
+ }
+ for (const arg of args) {
+ const tokens = arg.match(/\{[^}]+\}/g) ?? [];
+ for (const token of tokens) {
+ if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) {
+ return `Unknown placeholder ${token}; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`;
+ }
+ }
+ }
+ return null;
+}
+
+function substitutePlaceholders(
+ args: ReadonlyArray<string>,
+ values: { audioFile: string; outputBase: string; model: string },
+): string[] {
+ return args.map((arg) =>
+ arg.replace(/\{[^}]+\}/g, (token) => {
+ if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) {
+ throw new Error(
+ `Unknown placeholder ${token} in transcribeArgs; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`,
+ );
+ }
+ if (token === TRANSCRIBE_PLACEHOLDER_AUDIO) return values.audioFile;
+ if (token === TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE) return values.outputBase;
+ if (token === TRANSCRIBE_PLACEHOLDER_MODEL) return values.model;
+ return token;
+ }),
+ );
+}
+
+// ---------------------------------------------------------------------------
+// App registry
+// ---------------------------------------------------------------------------
+
+const whisperCpp: TranscriptionApp = {
+ id: "whisper-cpp",
+ label: "whisper.cpp (whisper-cli)",
+ fields: { model: true, customArgs: true },
+ defaultBin: () => getPaths().whisperBin,
+ build({ audioFile, outputBase, config }) {
+ const template =
+ config.customArgs && config.customArgs.length > 0
+ ? config.customArgs
+ : DEFAULT_TRANSCRIBE_ARGS;
+ const model = config.model ?? getPaths().whisperModel;
+ const argv = substitutePlaceholders(template, {
+ audioFile,
+ outputBase,
+ model,
+ });
+ // whisper-cli's `-of <base>` writes "<base>.json".
+ return { argv, outputFile: `${outputBase}.json`, outputFormat: "whisper-json" };
+ },
+ makeProgressParser: createTranscribeProgressParser,
+};
+
+const chough: TranscriptionApp = {
+ id: "chough",
+ label: "chough",
+ fields: { model: true, remoteUrl: true, chunkSize: true },
+ defaultBin: () => process.env.CHOUGH_BIN ?? "chough",
+ build({ audioFile, outputBase, config }) {
+ const argv = ["-f", "json", "-o", outputBase];
+ if (typeof config.chunkSize === "number" && config.chunkSize > 0) {
+ argv.push("-c", String(Math.floor(config.chunkSize)));
+ }
+ const env: Record<string, string> = {};
+ const remoteUrl = config.remoteUrl?.trim();
+ if (remoteUrl) {
+ argv.push("-r");
+ env.CHOUGH_URL = remoteUrl;
+ }
+ const model = config.model?.trim();
+ if (model) env.CHOUGH_MODEL = model;
+ argv.push(audioFile);
+ // chough writes EXACTLY the `-o` path — no ".json" is appended.
+ return {
+ argv,
+ env: Object.keys(env).length > 0 ? env : undefined,
+ outputFile: outputBase,
+ outputFormat: "chough-json",
+ };
+ },
+ makeProgressParser: createChoughProgressParser,
+};
+
+export const TRANSCRIPTION_APPS: Record<string, TranscriptionApp> = {
+ [whisperCpp.id]: whisperCpp,
+ [chough.id]: chough,
+};
+
+export const DEFAULT_TRANSCRIPTION_APP_ID = "whisper-cpp";
+
+export function getTranscriptionApp(id: string | undefined): TranscriptionApp {
+ return (
+ (id ? TRANSCRIPTION_APPS[id] : undefined) ??
+ TRANSCRIPTION_APPS[DEFAULT_TRANSCRIPTION_APP_ID]
+ );
+}
+
+// A client-safe view of an app (id/label/fields only — no functions). Build it
+// on the server and pass it to the settings form so the registry module (which
+// reaches getPaths/process.env) never ends up in the client bundle.
+export type TranscriptionAppDescriptor = {
+ id: string;
+ label: string;
+ fields: TranscriptionApp["fields"];
+};
+
+export function listTranscriptionApps(): TranscriptionAppDescriptor[] {
+ return Object.values(TRANSCRIPTION_APPS).map((a) => ({
+ id: a.id,
+ label: a.label,
+ fields: a.fields,
+ }));
+}
diff --git a/common/lib/whisper.ts b/common/lib/whisper.ts
@@ -1,4 +1,5 @@
import type { Cue } from "./vtt";
+import type { TranscriptOutputFormat } from "./transcriptionApps";
type WhisperSegment = {
offsets?: { from?: number; to?: number };
@@ -9,24 +10,95 @@ type WhisperDoc = {
transcription?: WhisperSegment[];
};
-export function parseWhisper(src: string): Cue[] {
- const doc = JSON.parse(src) as WhisperDoc;
+type ChoughChunk = {
+ start_time?: number;
+ end_time?: number;
+ text?: string;
+};
+
+type ChoughDoc = {
+ chunk_data?: ChoughChunk[];
+};
+
+// Append a cue, merging into the previous one when the text is identical
+// (whisper/chough both repeat a line across adjacent chunks). Shared by both
+// parsers so the merge behavior stays consistent.
+function pushCue(cues: Cue[], start: number, end: number, text: string): void {
+ const trimmed = text.trim();
+ if (!trimmed) return;
+ const prev = cues[cues.length - 1];
+ if (prev && prev.text === trimmed) {
+ prev.end = end;
+ return;
+ }
+ cues.push({ start, end, text: trimmed });
+}
+
+function whisperCuesFromDoc(doc: WhisperDoc): Cue[] {
const segments = doc.transcription ?? [];
const cues: Cue[] = [];
for (const seg of segments) {
const fromMs = seg.offsets?.from;
const toMs = seg.offsets?.to;
if (typeof fromMs !== "number" || typeof toMs !== "number") continue;
- const text = (seg.text ?? "").trim();
- if (!text) continue;
- const start = fromMs / 1000;
- const end = toMs / 1000;
- const prev = cues[cues.length - 1];
- if (prev && prev.text === text) {
- prev.end = end;
- continue;
- }
- cues.push({ start, end, text });
+ pushCue(cues, fromMs / 1000, toMs / 1000, seg.text ?? "");
}
return cues;
}
+
+function choughCuesFromDoc(doc: ChoughDoc): Cue[] {
+ const chunks = doc.chunk_data ?? [];
+ const cues: Cue[] = [];
+ for (const chunk of chunks) {
+ const start = chunk.start_time;
+ const end = chunk.end_time;
+ if (typeof start !== "number" || typeof end !== "number") continue;
+ // chough times are already in seconds.
+ pushCue(cues, start, end, chunk.text ?? "");
+ }
+ return cues;
+}
+
+// whisper.cpp JSON: { transcription: [{ offsets: { from, to (ms) }, text }] }.
+export function parseWhisper(src: string): Cue[] {
+ return whisperCuesFromDoc(JSON.parse(src) as WhisperDoc);
+}
+
+// chough native JSON: { chunk_data: [{ start_time, end_time (sec), text }], ... }.
+export function parseChough(src: string): Cue[] {
+ return choughCuesFromDoc(JSON.parse(src) as ChoughDoc);
+}
+
+// Sniff a transcript.json shape. Returns null for an unrecognized shape so the
+// caller can fall back to whisper (the only app before chough).
+export function detectTranscriptFormat(
+ src: string,
+): "whisper-json" | "chough-json" | null {
+ let doc: unknown;
+ try {
+ doc = JSON.parse(src);
+ } catch {
+ return null;
+ }
+ if (doc && typeof doc === "object") {
+ const d = doc as Record<string, unknown>;
+ if (Array.isArray(d.chunk_data)) return "chough-json";
+ if (Array.isArray(d.transcription)) return "whisper-json";
+ }
+ return null;
+}
+
+// Parse a transcript.json into cues. `hint` is the authoritative format when
+// known (e.g. the app that just produced the file, or a recorded per-video
+// tag); when absent or "vtt", the shape is sniffed, and anything unrecognized
+// falls back to whisper — the only transcription format in use before chough.
+export function parseTranscriptJson(
+ src: string,
+ hint?: TranscriptOutputFormat | null,
+): Cue[] {
+ const fmt =
+ hint === "whisper-json" || hint === "chough-json"
+ ? hint
+ : detectTranscriptFormat(src) ?? "whisper-json";
+ return fmt === "chough-json" ? parseChough(src) : parseWhisper(src);
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template).
- **Global default for parallel transcriptions.** A new **Settings → Parallel transcriptions** field sets how many videos a "Transcribe missing" / bucket run transcribes at once when its per-run Concurrency input is left blank (default **2**, clamped 1–16). This replaces the old `PARALLEL_TRANSCRIBE_LIMIT` env-var default of 4 as the source of the default: the value is now persisted in `settings.json`, shown as the placeholder in each channel's Concurrency input, and used as the server-side fallback. The per-run Concurrency input still overrides it for a single run.
- **Managed downloads skip live and upcoming videos by default.** Before downloading each video, the managed downloader now runs a quick metadata-only pass, then evaluates an app-level filter: videos that are currently live or scheduled/upcoming are skipped (their finished VODs still download normally — a skip isn't archived, so the next **Sync** / **Download missing** retries the video once the stream ends). A skip is recorded in the video's `download-outcome.json` (`status: "skipped-filtered"` with the filter name and reason) and logged to `download.log`; it never counts as a failure or aborts the batch, and the run's summary line reports how many were skipped. Skipped-live videos are surfaced in a new **Skipped: live or upcoming** bucket in the channel's Diagnostics (with a **Retry** to force an attempt now). On by default for every channel via a new **Settings → Skip live and upcoming videos** toggle, with a per-channel override (`config.json` `skipLiveDownloads`). To support the filter (and as a reusable building block for future per-video decisions), each managed download is now split into a metadata fetch followed by the real download, which reuses that metadata via `--load-info-json` so the second pass doesn't re-extract; the audio-integrity-checked download path re-extracts as before. Turn the toggle off for a channel that streams nothing to allow live captures.
- **Global default social links, overridable per site.** Footer social links can now be defined once in **Settings → Social links** and apply to every site by default. Each site's form gained a **Use global default social links** checkbox (on by default): leave it checked to inherit the global list, or uncheck it to give that site its own links (an empty list shows none). Existing sites keep their current footer — sites that already had links stay as explicit overrides until you flip the toggle. Validation and SVG sanitization are unchanged and shared by both the global and per-site editors.
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -13,6 +13,10 @@ import {
type SiteSettings,
type SocialLink,
} from "yt-dlp-transcript-common/lib/settings";
+import {
+ TRANSCRIPTION_APPS,
+ type AppInstanceConfig,
+} from "yt-dlp-transcript-common/lib/transcriptionApps";
export type SaveResult = { ok: true } | { ok: false; error: string };
@@ -22,8 +26,6 @@ export async function saveSettingsAction(
): Promise<SaveResult> {
const adminTitle = String(formData.get("adminTitle") ?? "").trim();
const maxBytesRaw = String(formData.get("maxTranscriptPageBytes") ?? "").trim();
- const transcribeBin = String(formData.get("transcribeBin") ?? "").trim();
- const transcribeModel = String(formData.get("transcribeModel") ?? "").trim();
const cookiesFromBrowser = String(
formData.get("cookiesFromBrowser") ?? "",
).trim();
@@ -36,19 +38,67 @@ export async function saveSettingsAction(
const inlineTranscribeOnFallback =
formData.get("inlineTranscribeOnFallback") === "on";
const skipLiveDownloads = formData.get("skipLiveDownloads") === "on";
- const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? "");
- const transcribeArgs = transcribeArgsRaw
- .split("\n")
- .map((s) => s.trim())
- .filter(Boolean);
if (!adminTitle) return { ok: false, error: "Admin title is required" };
- if (!transcribeBin) {
- return { ok: false, error: "Transcribe binary is required" };
+
+ const transcriptionApp = String(formData.get("transcriptionApp") ?? "").trim();
+ const activeApp = TRANSCRIPTION_APPS[transcriptionApp];
+ if (!activeApp) {
+ return { ok: false, error: "Unknown transcription app" };
}
- const argsErr = validateTranscribeArgs(transcribeArgs);
- if (argsErr) return { ok: false, error: argsErr };
+ // Collect per-app config from namespaced fields (app.<id>.<field>). Only the
+ // selected app's customArgs are validated; writeSettings re-checks the
+ // resolved binary.
+ const transcriptionApps: Record<string, AppInstanceConfig> = {};
+ for (const app of Object.values(TRANSCRIPTION_APPS)) {
+ const cfg: AppInstanceConfig = {};
+ const bin = String(formData.get(`app.${app.id}.bin`) ?? "").trim();
+ if (bin) cfg.bin = bin;
+ if (app.fields.model) {
+ const model = String(formData.get(`app.${app.id}.model`) ?? "").trim();
+ if (model) cfg.model = model;
+ }
+ if (app.fields.remoteUrl) {
+ const remoteUrl = String(
+ formData.get(`app.${app.id}.remoteUrl`) ?? "",
+ ).trim();
+ if (remoteUrl) cfg.remoteUrl = remoteUrl;
+ }
+ if (app.fields.chunkSize) {
+ const chunkRaw = String(
+ formData.get(`app.${app.id}.chunkSize`) ?? "",
+ ).trim();
+ if (chunkRaw) {
+ const n = Number.parseInt(chunkRaw, 10);
+ if (!Number.isFinite(n) || n <= 0) {
+ return {
+ ok: false,
+ error: `${app.label} chunk size must be a positive number`,
+ };
+ }
+ cfg.chunkSize = n;
+ }
+ }
+ if (app.fields.customArgs) {
+ const customArgs = String(formData.get(`app.${app.id}.customArgs`) ?? "")
+ .split("\n")
+ .map((s) => s.trim())
+ .filter(Boolean);
+ if (customArgs.length > 0) cfg.customArgs = customArgs;
+ }
+ if (Object.keys(cfg).length > 0) transcriptionApps[app.id] = cfg;
+ }
+
+ const activeCfg = transcriptionApps[transcriptionApp];
+ if (
+ activeApp.fields.customArgs &&
+ activeCfg?.customArgs &&
+ activeCfg.customArgs.length > 0
+ ) {
+ const argsErr = validateTranscribeArgs(activeCfg.customArgs);
+ if (argsErr) return { ok: false, error: argsErr };
+ }
const parsed = Number.parseInt(maxBytesRaw, 10);
if (!Number.isFinite(parsed)) {
@@ -121,9 +171,8 @@ export async function saveSettingsAction(
const next: SiteSettings = {
adminTitle,
maxTranscriptPageBytes: parsed,
- transcribeBin,
- transcribeModel,
- transcribeArgs,
+ transcriptionApp,
+ transcriptionApps,
cookiesFromBrowser,
sleepBetweenDownloadsSeconds: sleepParsed,
parallelTranscriptions: parallelParsed,
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -6,6 +6,7 @@ import {
type SaveResult,
} from "../actions";
import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings";
+import type { TranscriptionAppDescriptor } from "yt-dlp-transcript-common/lib/transcriptionApps";
import {
SocialLinksField,
toSocialRow,
@@ -14,9 +15,10 @@ import {
type Props = {
initial: SiteSettings;
+ apps: TranscriptionAppDescriptor[];
};
-export function SettingsForm({ initial }: Props) {
+export function SettingsForm({ initial, apps }: Props) {
const [state, formAction] = useActionState<SaveResult | undefined, FormData>(
saveSettingsAction,
undefined,
@@ -24,6 +26,7 @@ export function SettingsForm({ initial }: Props) {
const [social, setSocial] = useState<SocialRow[]>(() =>
initial.socialLinks.map(toSocialRow),
);
+ const [appId, setAppId] = useState<string>(initial.transcriptionApp);
return (
<form action={formAction} className="flex flex-col gap-4 max-w-xl">
@@ -42,44 +45,93 @@ export function SettingsForm({ initial }: Props) {
type="number"
/>
<fieldset className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded p-3">
- <legend className="px-1 text-sm font-medium">Transcription command</legend>
+ <legend className="px-1 text-sm font-medium">Transcription</legend>
<p className="text-xs text-zinc-500">
- Used by the channel "Transcribe missing" / "Retry failures" actions.
- The command must write <code>{`<outputBase>.json`}</code> in
- whisper.cpp's <code>{`{ transcription: [{ offsets, text }] }`}</code>{" "}
- format. Available placeholders:{" "}
- <code>{"{audioFile}"}</code> (required),{" "}
- <code>{"{outputBase}"}</code> (required),{" "}
- <code>{"{model}"}</code> (optional).
+ The app used by the channel “Transcribe missing” /
+ “Retry failures” actions. Each app handles its own
+ arguments and output format. Per-app fields below; only the selected
+ app's settings are used.
</p>
- <Field
- label="Binary"
- name="transcribeBin"
- defaultValue={initial.transcribeBin}
- required
- hint="e.g., whisper-cli, /usr/local/bin/whisper, or your own wrapper."
- />
- <Field
- label="Model"
- name="transcribeModel"
- defaultValue={initial.transcribeModel}
- hint="Substituted for {model}. Leave blank if your binary doesn't take a model."
- />
<label className="flex flex-col gap-1 text-sm">
- <span className="font-medium">Args (one per line)</span>
- <textarea
- name="transcribeArgs"
- defaultValue={initial.transcribeArgs.join("\n")}
- rows={Math.max(6, initial.transcribeArgs.length + 1)}
- className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono"
- />
- <span className="text-xs text-zinc-500">
- Default whisper.cpp call:{" "}
- <code>
- -ojf -l en -m {"{model}"} -of {"{outputBase}"} {"{audioFile}"}
- </code>
- </span>
+ <span className="font-medium">App</span>
+ <select
+ name="transcriptionApp"
+ value={appId}
+ onChange={(e) => setAppId(e.target.value)}
+ className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm"
+ >
+ {apps.map((app) => (
+ <option key={app.id} value={app.id}>
+ {app.label}
+ </option>
+ ))}
+ </select>
</label>
+ {apps.map((app) => {
+ const cfg = initial.transcriptionApps[app.id] ?? {};
+ return (
+ <div
+ key={app.id}
+ hidden={app.id !== appId}
+ className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-3"
+ >
+ <Field
+ label="Binary"
+ name={`app.${app.id}.bin`}
+ defaultValue={cfg.bin ?? ""}
+ hint="Path or name of the executable. Leave blank to use the default for this app."
+ />
+ {app.fields.model && (
+ <Field
+ label="Model"
+ name={`app.${app.id}.model`}
+ defaultValue={cfg.model ?? ""}
+ hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. Leave blank for the app default."
+ />
+ )}
+ {app.fields.remoteUrl && (
+ <Field
+ label="Remote URL"
+ name={`app.${app.id}.remoteUrl`}
+ defaultValue={cfg.remoteUrl ?? ""}
+ hint="chough only: transcribe via a remote chough --server (CHOUGH_URL). Leave blank for local transcription."
+ />
+ )}
+ {app.fields.chunkSize && (
+ <Field
+ label="Chunk size (seconds)"
+ name={`app.${app.id}.chunkSize`}
+ defaultValue={
+ cfg.chunkSize !== undefined ? String(cfg.chunkSize) : ""
+ }
+ type="number"
+ hint="chough only (-c). Leave blank for the app default (60s)."
+ />
+ )}
+ {app.fields.customArgs && (
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Custom args (one per line)</span>
+ <textarea
+ name={`app.${app.id}.customArgs`}
+ defaultValue={(cfg.customArgs ?? []).join("\n")}
+ rows={6}
+ className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono"
+ />
+ <span className="text-xs text-zinc-500">
+ Leave blank for the default whisper.cpp call:{" "}
+ <code>
+ -ojf -l en -m {"{model}"} -of {"{outputBase}"}{" "}
+ {"{audioFile}"}
+ </code>
+ . Placeholders: <code>{"{audioFile}"}</code> (required),{" "}
+ <code>{"{outputBase}"}</code> (required),{" "}
+ <code>{"{model}"}</code> (optional).
+ </span>
+ </label>
+ )}
+ </div>
+ );
+ })}
</fieldset>
<Field
label="Cookies from browser (retry only)"
diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx
@@ -1,6 +1,7 @@
import type { Metadata } from "next";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { listTranscriptionApps } from "yt-dlp-transcript-common/lib/transcriptionApps";
import { SettingsForm } from "./components/SettingsForm";
export const dynamic = "force-dynamic";
@@ -35,7 +36,7 @@ export default function SettingsPage() {
Used by the export build to title the static site.
</p>
</div>
- <SettingsForm initial={settings} />
+ <SettingsForm initial={settings} apps={listTranscriptionApps()} />
</section>
<section className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6">
@@ -46,9 +47,9 @@ export default function SettingsPage() {
<p className="text-sm text-zinc-500 mt-2">
Read-only view of <code>getPaths()</code>. Set the matching env
vars (TRANSCRIPTS_DIR, SETTINGS_FILE, EXPORT_PUBLIC_DIR, YTDLP_BIN,
- WHISPER_BIN, WHISPER_MODEL) before launching to override. The
- transcription command above takes
- precedence over <code>whisperBin</code> / <code>whisperModel</code>{" "}
+ WHISPER_BIN, WHISPER_MODEL, CHOUGH_BIN, CHOUGH_URL, CHOUGH_MODEL)
+ before launching to override. The per-app Binary / Model fields in
+ the transcription settings above take precedence over these defaults
when running channel transcribe actions.
</p>
<table className="text-sm mt-3 w-full">
diff --git a/editor/e2e/chough.spec.ts b/editor/e2e/chough.spec.ts
@@ -0,0 +1,82 @@
+import { readFile, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { pathExists, resetData, resolvePath, writeSettings } from "./helpers";
+
+const CHANNEL = "test-transcribe";
+const DATA = `test-transcripts/channels/${CHANNEL}/data`;
+
+// Minimal metadata so transcribeOne's normalize pass runs (it needs
+// metadata.info.json) and writes transcript.cues.json.
+const META = JSON.stringify({
+ id: "vidA",
+ title: "Synthetic chough video",
+ upload_date: "20240101",
+ duration: 10,
+});
+
+async function selectChough() {
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ transcriptionApp: "chough",
+ transcriptionApps: { chough: {} },
+ });
+}
+
+test("chough app transcribes audio and writes a transcript.json", async ({
+ page,
+}) => {
+ await resetData("one-transcribe-channel-with-audio");
+ await selectChough();
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+ await expect(page.getByLabel("Transcribe missing output")).toContainText(
+ "3 succeeded",
+ { timeout: 30_000 },
+ );
+
+ for (const id of ["vidA", "vidB", "vidC"]) {
+ expect(await pathExists(`${DATA}/${id}/transcript.json`)).toBe(true);
+ }
+
+ // The raw transcript is chough-native (chunk_data with seconds). Reaching this
+ // shape proves the chough app's argv ran AND that transcribeOne correctly
+ // renamed chough's exact `-o` output (no ".json" appended) to transcript.json.
+ const raw = JSON.parse(
+ await readFile(resolvePath(`${DATA}/vidA/transcript.json`), "utf8"),
+ );
+ expect(Array.isArray(raw.chunk_data)).toBe(true);
+ expect(raw.transcription).toBeUndefined();
+});
+
+test("chough output normalizes to chough-tagged cues", async ({ page }) => {
+ await resetData("one-transcribe-channel-with-audio");
+ await selectChough();
+ // Give vidA metadata so the normalize pass produces transcript.cues.json.
+ await writeFile(resolvePath(`${DATA}/vidA/metadata.info.json`), META);
+
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Transcribe missing" }).click();
+ await expect(page.getByLabel("Transcribe missing output")).toContainText(
+ "3 succeeded",
+ { timeout: 30_000 },
+ );
+
+ await expect
+ .poll(() => pathExists(`${DATA}/vidA/transcript.cues.json`), {
+ timeout: 15_000,
+ })
+ .toBe(true);
+
+ const cues = JSON.parse(
+ await readFile(resolvePath(`${DATA}/vidA/transcript.cues.json`), "utf8"),
+ );
+ // The durable per-video format tag is recorded…
+ expect(cues.transcriptFormat).toBe("chough-json");
+ // …and the cues come from chough's seconds-based chunk_data (start 0, end 5),
+ // not misread as whisper milliseconds.
+ expect(cues.cues.length).toBeGreaterThan(0);
+ expect(cues.cues[0].start).toBe(0);
+ expect(cues.cues[0].end).toBe(5);
+});
diff --git a/editor/e2e/fixtures/bin/fake-chough.mjs b/editor/e2e/fixtures/bin/fake-chough.mjs
@@ -0,0 +1,57 @@
+#!/usr/bin/env node
+// E2E fake chough. Mimics the args the chough app builds (transcriptionApps.ts):
+// chough -f json -o <tmpBase> [-c <n>] [-r] <audioFilename>
+// Writes the output to EXACTLY <tmpBase> (no ".json" appended — this is the key
+// behavioural difference from whisper-cli's -of), in chough's native JSON shape
+// { duration_seconds, chunks, text, chunk_data: [{ start_time, end_time, text }] }
+// with times in SECONDS. Run from cwd == the video dir (set by runWhisperBatch).
+import { writeFile } from "node:fs/promises";
+
+const argv = process.argv.slice(2);
+function arg(flag) {
+ const i = argv.indexOf(flag);
+ return i < 0 ? undefined : argv[i + 1];
+}
+
+const out = arg("-o");
+const audio = argv[argv.length - 1];
+if (!out) {
+ process.stderr.write(`[fake-chough] missing -o\n`);
+ process.exit(2);
+}
+
+const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
+
+// Videos whose dir (== video id) contains the SLOWOP marker emit chough-style
+// progress output (audio header + █/░ bars + ETA) with real delays, so the
+// per-operation progress bars and drain behaviour can be observed. Videos
+// without the marker stay instant so the rest of the suite is fast.
+const totalSec = 50;
+if (process.cwd().toLowerCase().includes("slowop")) {
+ process.stdout.write(`mode: local\n`);
+ process.stdout.write(`audio: ${totalSec}.0s • chunks: 60s • format: json\n`);
+ const width = 40;
+ for (let step = 1; step <= 10; step += 1) {
+ const filled = Math.round((step / 10) * width);
+ const bar = "█".repeat(filled) + "░".repeat(width - filled);
+ const etaSec = Math.max(0, Math.round(((10 - step) / 10) * 30));
+ process.stdout.write(
+ `${bar} ETA ${Math.floor(etaSec / 60)}m ${etaSec % 60}s\x1b[K\n`,
+ );
+ await sleep(700);
+ }
+}
+
+const doc = {
+ duration_seconds: totalSec,
+ chunks: 2,
+ text: `Synthetic chough output for ${audio} #1 Synthetic chough output for ${audio} #2`,
+ chunk_data: [
+ { start_time: 0, end_time: 5, text: `Synthetic chough output for ${audio} #1` },
+ { start_time: 5, end_time: 10, text: `Synthetic chough output for ${audio} #2` },
+ ],
+};
+
+// chough writes EXACTLY the -o path (no extension munging).
+await writeFile(out, JSON.stringify(doc));
+process.stdout.write(`Output: ${out}\n`);
diff --git a/editor/e2e/transcription-app-migration.spec.ts b/editor/e2e/transcription-app-migration.spec.ts
@@ -0,0 +1,38 @@
+import { test, expect } from "@playwright/test";
+import { resetData, writeSettings } from "./helpers";
+
+// A pre-multi-app settings.json (transcribeBin/transcribeArgs/transcribeModel,
+// no transcriptionApp key) must migrate onto the app registry when read, so the
+// Settings page shows a selected app rather than crashing on missing fields.
+
+test("legacy whisper settings migrate onto the whisper-cpp app", async ({
+ page,
+}) => {
+ await resetData(null);
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ transcribeBin: "whisper-cli",
+ transcribeModel: "/models/ggml.bin",
+ transcribeArgs: ["-ojf", "-l", "en", "-m", "{model}", "-of", "{outputBase}", "{audioFile}"],
+ });
+ await page.goto("/settings");
+ await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("whisper-cpp");
+});
+
+test("legacy settings whose binary is chough migrate onto the chough app", async ({
+ page,
+}) => {
+ await resetData(null);
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ transcribeBin: "chough",
+ transcribeModel: "",
+ transcribeArgs: ["-f", "json", "-o", "{outputBase}", "{audioFile}"],
+ });
+ await page.goto("/settings");
+ await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("chough");
+});
diff --git a/editor/package.json b/editor/package.json
@@ -5,8 +5,8 @@
"type": "module",
"scripts": {
"dev": "next dev --port 3001",
- "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011",
- "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011",
+ "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011",
+ "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011",
"build": "next build",
"start": "next start --port 3001",
"lint": "eslint",