Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c199f78397fc06f359cd903394e52519d34f4d67
parent 3c6332662dd4ae49d5af7fa2301b9afbd499fccf
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  9 Jun 2026 12:12:14 -0400

support multi-app / chough transcription

Diffstat:
Mcommon/controller/buildIndex.ts | 4++--
Mcommon/controller/duplicateShorts.ts | 4++--
Mcommon/controller/normalizeTranscript.ts | 44+++++++++++++++++++++++++++++++++++++++++---
Mcommon/controller/transcribeOne.ts | 56++++++++++++++++++++++----------------------------------
Mcommon/jobs/progressParsers.ts | 27+++++++++++++++++++++++++++
Mcommon/jobs/taskHooks.ts | 13++++++++-----
Mcommon/lib/settings.ts | 187++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------------------
Acommon/lib/transcriptionApps.ts | 229+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/whisper.ts | 96+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----------
Meditor/CHANGELOG.md | 1+
Meditor/app/settings/actions.ts | 77+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------
Meditor/app/settings/components/SettingsForm.tsx | 122++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----------------------
Meditor/app/settings/page.tsx | 9+++++----
Aeditor/e2e/chough.spec.ts | 82+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/fixtures/bin/fake-chough.mjs | 57+++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/transcription-app-migration.spec.ts | 38++++++++++++++++++++++++++++++++++++++
Meditor/package.json | 4++--
17 files changed, 878 insertions(+), 172 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -25,7 +25,7 @@ import type { Dirent, WriteStream } from "node:fs"; import { once } from "node:events"; import { open } from "lmdb"; import { parseVtt, type Cue } from "../lib/vtt"; -import { parseWhisper } from "../lib/whisper"; +import { parseTranscriptJson } from "../lib/whisper"; import { parseLiveChat } from "../lib/liveChat"; import { summarize, @@ -494,7 +494,7 @@ export async function buildIndex({ cueList = s.transcriptKind === "vtt" ? parseVtt(raw) - : parseWhisper(raw); + : parseTranscriptJson(raw); } catch { cueList = undefined; } diff --git a/common/controller/duplicateShorts.ts b/common/controller/duplicateShorts.ts @@ -27,7 +27,7 @@ import type { Paths } from "../lib/paths"; import type { VideoStat } from "../lib/stats"; import { STATS_SCHEMA_VERSION } from "../lib/stats"; import type { Cue } from "../lib/vtt"; -import { parseWhisper } from "../lib/whisper"; +import { parseTranscriptJson } from "../lib/whisper"; import { readVideoFiles, pickIndexTranscript } from "../lib/videoStatus"; import { DEFAULT_CONTAINMENT_THRESHOLD, @@ -407,7 +407,7 @@ async function readRawCues( } catch { return null; } - return picked.kind === "whisper" ? parseWhisper(raw) : lenientVttCues(raw); + return picked.kind === "whisper" ? parseTranscriptJson(raw) : lenientVttCues(raw); } // Lenient VTT text extraction: unlike parseVtt (which only keeps YouTube diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts @@ -6,7 +6,11 @@ import path from "node:path"; import { readdir, readFile, rename, stat, writeFile } from "node:fs/promises"; import { parseVtt, type Cue } from "../lib/vtt"; -import { parseWhisper } from "../lib/whisper"; +import { + detectTranscriptFormat, + parseTranscriptJson, +} from "../lib/whisper"; +import type { TranscriptOutputFormat } from "../lib/transcriptionApps"; import { summarize, type RawMetadata } from "../lib/transcripts-server"; import type { TranscriptDetail } from "../lib/transcripts"; import { @@ -25,6 +29,9 @@ export const CUES_FILE_VERSION = 1; export type NormalizedTranscript = TranscriptDetail & { version: number; source: IndexTranscript["kind"] | "live_chat"; + // The raw transcript format this was parsed from. Durable per-video record so + // a later re-normalize knows how to read the raw file without re-sniffing. + transcriptFormat?: TranscriptOutputFormat; }; export type NormalizedLiveChat = NormalizedTranscript & { source: "live_chat" }; @@ -33,6 +40,10 @@ export type NormalizeOptions = { videoDir: string; channelSlug: string; configName?: string; + // Authoritative raw-transcript format, passed by transcribeOne right after a + // run (it knows which app produced the file). When omitted, the format is + // resolved from the prior cues.json tag, then by content sniff. + formatHint?: TranscriptOutputFormat; log?: (msg: string) => void; force?: boolean; }; @@ -90,8 +101,21 @@ export async function normalizeTranscript( const rawTranscript = await readFile(transcriptPath, "utf8"); let cues: Cue[]; + let transcriptFormat: TranscriptOutputFormat; try { - cues = picked.kind === "vtt" ? parseVtt(rawTranscript) : parseWhisper(rawTranscript); + if (picked.kind === "vtt") { + cues = parseVtt(rawTranscript); + transcriptFormat = "vtt"; + } else { + // Resolve the JSON format: authoritative hint -> recorded per-video tag + // -> content sniff -> whisper fallback (the only app before chough). + transcriptFormat = + opts.formatHint ?? + (await readRecordedFormat(cuesPath)) ?? + detectTranscriptFormat(rawTranscript) ?? + "whisper-json"; + cues = parseTranscriptJson(rawTranscript, transcriptFormat); + } } catch (err) { throw new Error( `Failed to parse ${picked.filename} for ${opts.channelSlug}/${path.basename(opts.videoDir)}: ${(err as Error).message}`, @@ -101,6 +125,7 @@ export async function normalizeTranscript( const out: NormalizedTranscript = { version: CUES_FILE_VERSION, source: picked.kind, + transcriptFormat, ...summary, cues, }; @@ -109,11 +134,24 @@ export async function normalizeTranscript( await writeFile(tmp, JSON.stringify(out)); await rename(tmp, cuesPath); opts.log?.( - `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${picked.kind}, ${cues.length} cues)`, + `Normalized ${opts.channelSlug}/${path.basename(opts.videoDir)} (${transcriptFormat}, ${cues.length} cues)`, ); return { status: "wrote", cuesPath }; } +// Read the format recorded in a prior transcript.cues.json, if any. This is the +// durable per-video tag that lets a re-normalize parse the raw file the same way +// it was first parsed, without re-sniffing. +async function readRecordedFormat( + cuesPath: string, +): Promise<TranscriptOutputFormat | undefined> { + const prior = await readNormalizedTranscript(cuesPath); + const f = prior?.transcriptFormat; + return f === "whisper-json" || f === "chough-json" || f === "vtt" + ? f + : undefined; +} + export async function readNormalizedTranscript( cuesPath: string, ): Promise<NormalizedTranscript | null> { diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts @@ -2,13 +2,8 @@ import path from "node:path"; import fs from "fs-extra"; import { execa } from "execa"; import type { Paths } from "../lib/paths"; -import { - TRANSCRIBE_PLACEHOLDER_AUDIO, - TRANSCRIBE_PLACEHOLDER_MODEL, - TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, - TRANSCRIBE_KNOWN_PLACEHOLDERS, - getSettings, -} from "../lib/settings"; +import { getSettings } from "../lib/settings"; +import { getTranscriptionApp } from "../lib/transcriptionApps"; import { normalizeTranscript } from "./normalizeTranscript"; import { isRealAudioFile } from "../lib/videoStatus"; @@ -32,25 +27,6 @@ async function resolveAudioFile( return [...candidates].sort()[0]; } -function substitutePlaceholders( - args: string[], - values: { audioFile: string; outputBase: string; model: string }, -): string[] { - return args.map((arg) => { - return arg.replace(/\{[^}]+\}/g, (token) => { - if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { - throw new Error( - `Unknown placeholder ${token} in transcribeArgs; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`, - ); - } - if (token === TRANSCRIBE_PLACEHOLDER_AUDIO) return values.audioFile; - if (token === TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE) return values.outputBase; - if (token === TRANSCRIBE_PLACEHOLDER_MODEL) return values.model; - return token; - }); - }); -} - export type TranscribeOneOptions = { paths: Paths; videoDir: string; @@ -92,26 +68,37 @@ export async function transcribeOneVideo( } const settings = getSettings(); - log(`Transcribe ${opts.videoId} start (${resolvedAudio})`); + const app = getTranscriptionApp(settings.transcriptionApp); + const appConfig = settings.transcriptionApps[app.id] ?? {}; + const bin = appConfig.bin?.trim() || app.defaultBin(); + log(`Transcribe ${opts.videoId} start (${resolvedAudio}) via ${app.id}`); const start = Date.now(); - // Whisper writes "<outputBase>.json" relative to cwd. Use a tmp name and - // rename on success so a SIGTERM mid-write can't leave a half-baked file. + // Each app writes "<outputBase><ext>" relative to cwd (the ext differs: whisper + // appends ".json", chough writes the exact -o path). Use a tmp base and rename + // the app-declared output file on success so a SIGTERM mid-write can't leave a + // half-baked transcript.json. const tmpBase = `transcript.tmp-${process.pid}`; - const args = substitutePlaceholders(settings.transcribeArgs, { + const build = app.build({ audioFile: resolvedAudio, outputBase: tmpBase, - model: settings.transcribeModel, + config: appConfig, }); - const child = execa(settings.transcribeBin, args, { + const child = execa(bin, build.argv, { cwd: opts.videoDir, cancelSignal: opts.signal, all: true, buffer: false, + env: build.env ? { ...process.env, ...build.env } : undefined, }); child.all?.on("data", (c: Buffer) => log(c.toString("utf8"))); await child; - const tmpPath = path.join(opts.videoDir, `${tmpBase}.json`); - await rename(tmpPath, transcriptPath); + const producedPath = path.join(opts.videoDir, build.outputFile); + if (!(await pathExists(producedPath))) { + throw new Error( + `transcription with ${app.id} produced no ${build.outputFile} in ${opts.videoDir}`, + ); + } + await rename(producedPath, transcriptPath); log( `Transcribe ${opts.videoId} done in ${((Date.now() - start) / 1000).toFixed(2)}s`, ); @@ -122,6 +109,7 @@ export async function transcribeOneVideo( await normalizeTranscript({ videoDir: opts.videoDir, channelSlug, + formatHint: build.outputFormat, log, }); } catch (err) { diff --git a/common/jobs/progressParsers.ts b/common/jobs/progressParsers.ts @@ -173,3 +173,30 @@ export function createTranscribeProgressParser(): { }, }; } + +// chough output is bar-based, not segment-timestamp based. It prints a header +// audio: 28595.7s • chunks: 60s • format: json +// then repeatedly overwrites a progress bar of full (█) and light (░) blocks +// with a trailing ETA, e.g. +// ████░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░░ ETA 38m 57s +// Progress = filled blocks / total blocks; the ETA text is surfaced as detail. +export function createChoughProgressParser(): { + feed: (line: string) => ProgressUpdate | null; +} { + return { + feed(line: string): ProgressUpdate | null { + // Consume (and ignore) the audio header line. + if (/\baudio:\s*[\d.]+s\b/.test(line)) return null; + const filled = (line.match(/█/g) ?? []).length; + const empty = (line.match(/░/g) ?? []).length; + const total = filled + empty; + if (total === 0) return null; + const out: ProgressUpdate = { fraction: clamp01(filled / total) }; + // ETA text up to an ANSI clear ("[K") or end of line. + const etaMatch = line.match(/ETA\s+([0-9hdms\s]+?)\s*(?:\x1b?\[K|$)/); + const eta = etaMatch?.[1]?.trim(); + if (eta) out.detail = `ETA ${eta}`; + return out; + }, + }; +} diff --git a/common/jobs/taskHooks.ts b/common/jobs/taskHooks.ts @@ -1,9 +1,8 @@ import type { JobTaskKind } from "./registry"; import type { JobRunContext } from "./streamCommand"; -import { - parseDownloadProgress, - createTranscribeProgressParser, -} from "./progressParsers"; +import { parseDownloadProgress } from "./progressParsers"; +import { getSettings } from "../lib/settings"; +import { getTranscriptionApp } from "../lib/transcriptionApps"; // A handle for one in-flight sub-operation. `onLog` is a drop-in replacement // for the controller's existing `onLog`: it forwards every line to the shared @@ -34,8 +33,12 @@ export function makeTaskTracker( return { onLog: forwardLog, end: () => {} }; } ctx.addTask({ id, label, kind, startedAt: Date.now() }); + // Transcription progress output is app-specific (whisper's segment + // timestamps vs chough's ETA bars), so pick the active app's parser. const transcribeParser = - kind === "transcribe" ? createTranscribeProgressParser() : null; + kind === "transcribe" + ? getTranscriptionApp(getSettings().transcriptionApp).makeProgressParser() + : null; let ended = false; const onLog = (line: string) => { forwardLog(line); diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -1,5 +1,26 @@ import fs from "node:fs"; +import path from "node:path"; import { getPaths } from "./paths"; +import { + type AppInstanceConfig, + DEFAULT_TRANSCRIBE_ARGS, + DEFAULT_TRANSCRIPTION_APP_ID, + getTranscriptionApp, + TRANSCRIPTION_APPS, + validateTranscribeArgs, +} from "./transcriptionApps"; + +// Transcribe placeholder/arg helpers now live with the whisper-cpp app in +// transcriptionApps.ts. Re-exported here so existing import sites keep working. +export { + type AppInstanceConfig, + TRANSCRIBE_PLACEHOLDER_AUDIO, + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_MODEL, + TRANSCRIBE_KNOWN_PLACEHOLDERS, + DEFAULT_TRANSCRIBE_ARGS, + validateTranscribeArgs, +} from "./transcriptionApps"; // Global, OPERATIONAL settings shared across every site this editor powers. // Per-site presentation (branding, social links, channel groups, membership) @@ -10,9 +31,12 @@ export type SiteSettings = { // from site.json. adminTitle: string; maxTranscriptPageBytes: number; - transcribeBin: string; - transcribeArgs: string[]; - transcribeModel: string; + // Active transcription app id (key into TRANSCRIPTION_APPS, e.g. "whisper-cpp" + // or "chough"). Selected globally; see common/lib/transcriptionApps.ts. + transcriptionApp: string; + // Per-app configuration, keyed by app id. Each app reads only its own block; + // a missing block means "use the app's defaults". + transcriptionApps: Record<string, AppInstanceConfig>; // Browser spec (e.g. "firefox", "chrome:Default") passed to // `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the // managed per-video download flow. Empty string = disabled. @@ -62,36 +86,14 @@ export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; -export const TRANSCRIBE_PLACEHOLDER_AUDIO = "{audioFile}"; -export const TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE = "{outputBase}"; -export const TRANSCRIBE_PLACEHOLDER_MODEL = "{model}"; -export const TRANSCRIBE_KNOWN_PLACEHOLDERS: ReadonlyArray<string> = [ - TRANSCRIBE_PLACEHOLDER_AUDIO, - TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, - TRANSCRIBE_PLACEHOLDER_MODEL, -]; - -export const DEFAULT_TRANSCRIBE_ARGS: ReadonlyArray<string> = [ - "-ojf", - "-l", - "en", - "-m", - TRANSCRIBE_PLACEHOLDER_MODEL, - "-of", - TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, - TRANSCRIBE_PLACEHOLDER_AUDIO, -]; - export const DEFAULT_ADMIN_TITLE = "Transcript Browser Admin"; function defaults(): SiteSettings { - const paths = getPaths(); return { adminTitle: DEFAULT_ADMIN_TITLE, maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES, - transcribeBin: paths.whisperBin, - transcribeArgs: [...DEFAULT_TRANSCRIBE_ARGS], - transcribeModel: paths.whisperModel, + transcriptionApp: DEFAULT_TRANSCRIPTION_APP_ID, + transcriptionApps: {}, cookiesFromBrowser: "", sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS, parallelTranscriptions: PARALLEL_TRANSCRIPTIONS_DEFAULT, @@ -196,14 +198,17 @@ export function getSettings(): SiteSettings { merged.adminTitle = DEFAULT_ADMIN_TITLE; } merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes); - if (!Array.isArray(merged.transcribeArgs) || merged.transcribeArgs.length === 0) { - merged.transcribeArgs = [...DEFAULT_TRANSCRIBE_ARGS]; + merged.transcriptionApps = sanitizeTranscriptionApps(merged.transcriptionApps); + // Migrate a pre-multi-app settings.json (transcribeBin/transcribeArgs/ + // transcribeModel, with no transcriptionApp key) onto the app registry. + if (parsed.transcriptionApp === undefined) { + migrateLegacyTranscription(parsed as LegacyTranscribeFields, merged); } - if (typeof merged.transcribeBin !== "string" || !merged.transcribeBin.trim()) { - merged.transcribeBin = defaults().transcribeBin; - } - if (typeof merged.transcribeModel !== "string") { - merged.transcribeModel = defaults().transcribeModel; + if ( + typeof merged.transcriptionApp !== "string" || + !TRANSCRIPTION_APPS[merged.transcriptionApp] + ) { + merged.transcriptionApp = DEFAULT_TRANSCRIPTION_APP_ID; } if (typeof merged.cookiesFromBrowser !== "string") { merged.cookiesFromBrowser = ""; @@ -234,33 +239,98 @@ function clampPageBytes(value: unknown): number { return Math.floor(n); } -export function validateTranscribeArgs(args: string[]): string | null { - if (!Array.isArray(args) || args.length === 0) { - return "Transcribe args must contain at least one entry"; - } - const joined = args.join(" "); - if (!joined.includes(TRANSCRIBE_PLACEHOLDER_AUDIO)) { - return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_AUDIO} placeholder`; - } - if (!joined.includes(TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE)) { - return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE} placeholder`; - } - for (const arg of args) { - const tokens = arg.match(/\{[^}]+\}/g) ?? []; - for (const token of tokens) { - if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { - return `Unknown placeholder ${token}; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`; - } +type LegacyTranscribeFields = { + transcribeBin?: unknown; + transcribeArgs?: unknown; + transcribeModel?: unknown; +}; + +// Coerce a raw settings.transcriptionApps value into a clean keyed map of +// AppInstanceConfig, dropping unknown/ill-typed fields. +export function sanitizeTranscriptionApps( + value: unknown, +): Record<string, AppInstanceConfig> { + if (!value || typeof value !== "object" || Array.isArray(value)) return {}; + const out: Record<string, AppInstanceConfig> = {}; + for (const [id, raw] of Object.entries(value as Record<string, unknown>)) { + if (!raw || typeof raw !== "object") continue; + const r = raw as Record<string, unknown>; + const cfg: AppInstanceConfig = {}; + if (typeof r.bin === "string") cfg.bin = r.bin; + if (typeof r.model === "string") cfg.model = r.model; + if (typeof r.remoteUrl === "string") cfg.remoteUrl = r.remoteUrl; + if (typeof r.chunkSize === "number" && Number.isFinite(r.chunkSize)) { + cfg.chunkSize = Math.floor(r.chunkSize); } + if (Array.isArray(r.customArgs)) cfg.customArgs = r.customArgs.map(String); + out[id] = cfg; + } + return out; +} + +function argsAreDefault(args: string[]): boolean { + return ( + args.length === DEFAULT_TRANSCRIBE_ARGS.length && + args.every((a, i) => a === DEFAULT_TRANSCRIBE_ARGS[i]) + ); +} + +function migrateLegacyTranscription( + raw: LegacyTranscribeFields, + merged: SiteSettings, +): void { + const bin = typeof raw.transcribeBin === "string" ? raw.transcribeBin.trim() : ""; + const model = typeof raw.transcribeModel === "string" ? raw.transcribeModel : ""; + const legacyArgs = Array.isArray(raw.transcribeArgs) + ? raw.transcribeArgs.map(String) + : []; + if (!bin && !model && legacyArgs.length === 0) return; // nothing to migrate + + // If the legacy binary names a known app (e.g. "chough"), adopt that app so + // its native arg/output handling applies; otherwise fall back to whisper-cpp, + // which preserves the old placeholder-template behavior. + const binBase = bin ? path.basename(bin).toLowerCase() : ""; + const appId = + binBase && TRANSCRIPTION_APPS[binBase] + ? binBase + : DEFAULT_TRANSCRIPTION_APP_ID; + const cfg: AppInstanceConfig = { ...merged.transcriptionApps[appId] }; + if (bin) cfg.bin = bin; + if (model) cfg.model = model; + // Carry a customArgs template only for whisper-cpp and only when it differs + // from the default — a default install migrates to a clean config. The chough + // app builds its own argv, so legacy args are dropped for it. + if ( + appId === DEFAULT_TRANSCRIPTION_APP_ID && + legacyArgs.length > 0 && + !argsAreDefault(legacyArgs) + ) { + cfg.customArgs = legacyArgs; } - return null; + merged.transcriptionApp = appId; + merged.transcriptionApps = { ...merged.transcriptionApps, [appId]: cfg }; } export async function writeSettings(next: SiteSettings): Promise<void> { - const argsErr = validateTranscribeArgs(next.transcribeArgs); - if (argsErr) throw new Error(argsErr); - if (!next.transcribeBin.trim()) { - throw new Error("Transcribe binary is required"); + const appId = + typeof next.transcriptionApp === "string" && + TRANSCRIPTION_APPS[next.transcriptionApp] + ? next.transcriptionApp + : DEFAULT_TRANSCRIPTION_APP_ID; + const app = getTranscriptionApp(appId); + const transcriptionApps = sanitizeTranscriptionApps(next.transcriptionApps); + const activeCfg = transcriptionApps[appId] ?? {}; + const resolvedBin = activeCfg.bin?.trim() || app.defaultBin(); + if (!resolvedBin) { + throw new Error(`Transcribe binary is required for ${app.label}`); + } + if ( + app.fields.customArgs && + activeCfg.customArgs && + activeCfg.customArgs.length > 0 + ) { + const argsErr = validateTranscribeArgs(activeCfg.customArgs); + if (argsErr) throw new Error(argsErr); } const socialLinks: SocialLink[] = []; for (const link of parseSocialLinks(next.socialLinks)) { @@ -281,9 +351,8 @@ export async function writeSettings(next: SiteSettings): Promise<void> { ? next.adminTitle.trim() : DEFAULT_ADMIN_TITLE, maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes), - transcribeBin: next.transcribeBin, - transcribeModel: next.transcribeModel, - transcribeArgs: next.transcribeArgs.map((a) => String(a)), + transcriptionApp: appId, + transcriptionApps, cookiesFromBrowser: typeof next.cookiesFromBrowser === "string" ? next.cookiesFromBrowser.trim() diff --git a/common/lib/transcriptionApps.ts b/common/lib/transcriptionApps.ts @@ -0,0 +1,229 @@ +// First-class registry of transcription apps. Each app owns how it builds its +// argv/env from a resolved per-app config, what file it actually writes (the +// `-o`/`-of` naming differs between tools), how its raw output is shaped, and +// how its live progress output is parsed. The editor selects ONE active app +// globally (settings.transcriptionApp) and stores a small per-app config block +// (settings.transcriptionApps[id]); transcribeOne resolves the app and runs it. +// +// Pure definitions only (no I/O beyond reading process.env / getPaths inside +// builders) so both the `common` controllers and the editor UI can import this. +// Dependency direction is one-way: settings.ts -> transcriptionApps.ts -> +// { paths, jobs/progressParsers }. Keep it that way to avoid import cycles. + +import { getPaths } from "./paths"; +import { + type ProgressUpdate, + createTranscribeProgressParser, + createChoughProgressParser, +} from "../jobs/progressParsers"; + +export type TranscriptOutputFormat = "whisper-json" | "chough-json" | "vtt"; + +// Per-app configuration persisted under settings.transcriptionApps[id]. Every +// field is optional; an app falls back to its own defaults (defaultBin, env). +export type AppInstanceConfig = { + // Binary path/name override. Empty/undefined falls back to app.defaultBin(). + bin?: string; + // whisper.cpp model path (substituted for {model}); for chough this is the + // optional CHOUGH_MODEL env (chough auto-downloads a model when unset). + model?: string; + // chough remote server URL (CHOUGH_URL). Empty/undefined = local transcription. + remoteUrl?: string; + // chough chunk size in seconds (-c). Undefined = chough's own default. + chunkSize?: number; + // whisper.cpp custom argv template using the {audioFile}/{outputBase}/{model} + // placeholders. Undefined = DEFAULT_TRANSCRIBE_ARGS. + customArgs?: string[]; +}; + +export type TranscribeBuild = { + // Args passed after the binary. + argv: string[]; + // Extra environment variables merged over process.env for this run. + env?: Record<string, string>; + // The file the app actually writes, relative to the video dir (cwd). + // transcribeOne renames THIS to transcript.json — this is where the + // chough-vs-whisper `-o`/`-of` naming difference is absorbed. + outputFile: string; + // Shape of the raw output, used as the authoritative parse hint downstream. + outputFormat: TranscriptOutputFormat; +}; + +export type TranscribeBuildInput = { + audioFile: string; // basename relative to videoDir + outputBase: string; // e.g. "transcript.tmp-<pid>" (NO extension) + config: AppInstanceConfig; +}; + +export type TranscriptionApp = { + id: string; + label: string; + // Which config fields this app surfaces in the settings UI / consumes. + fields: { + model?: boolean; + remoteUrl?: boolean; + chunkSize?: boolean; + customArgs?: boolean; + }; + // Default binary when the per-app `bin` override is empty. + defaultBin: () => string; + build: (input: TranscribeBuildInput) => TranscribeBuild; + makeProgressParser: () => { feed: (line: string) => ProgressUpdate | null }; +}; + +// --------------------------------------------------------------------------- +// whisper.cpp argv template + placeholders (formerly in settings.ts; kept here +// so the whisper-cpp app owns its own arg construction). settings.ts re-exports +// these for backward compatibility with existing import sites. +// --------------------------------------------------------------------------- + +export const TRANSCRIBE_PLACEHOLDER_AUDIO = "{audioFile}"; +export const TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE = "{outputBase}"; +export const TRANSCRIBE_PLACEHOLDER_MODEL = "{model}"; +export const TRANSCRIBE_KNOWN_PLACEHOLDERS: ReadonlyArray<string> = [ + TRANSCRIBE_PLACEHOLDER_AUDIO, + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_MODEL, +]; + +export const DEFAULT_TRANSCRIBE_ARGS: ReadonlyArray<string> = [ + "-ojf", + "-l", + "en", + "-m", + TRANSCRIBE_PLACEHOLDER_MODEL, + "-of", + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_AUDIO, +]; + +// Validate a whisper.cpp customArgs template: must reference {audioFile} and +// {outputBase}, may reference {model}, and must contain no unknown placeholders. +// Returns an error string or null when valid. +export function validateTranscribeArgs(args: string[]): string | null { + if (!Array.isArray(args) || args.length === 0) { + return "Transcribe args must contain at least one entry"; + } + const joined = args.join(" "); + if (!joined.includes(TRANSCRIBE_PLACEHOLDER_AUDIO)) { + return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_AUDIO} placeholder`; + } + if (!joined.includes(TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE)) { + return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE} placeholder`; + } + for (const arg of args) { + const tokens = arg.match(/\{[^}]+\}/g) ?? []; + for (const token of tokens) { + if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { + return `Unknown placeholder ${token}; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`; + } + } + } + return null; +} + +function substitutePlaceholders( + args: ReadonlyArray<string>, + values: { audioFile: string; outputBase: string; model: string }, +): string[] { + return args.map((arg) => + arg.replace(/\{[^}]+\}/g, (token) => { + if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { + throw new Error( + `Unknown placeholder ${token} in transcribeArgs; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`, + ); + } + if (token === TRANSCRIBE_PLACEHOLDER_AUDIO) return values.audioFile; + if (token === TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE) return values.outputBase; + if (token === TRANSCRIBE_PLACEHOLDER_MODEL) return values.model; + return token; + }), + ); +} + +// --------------------------------------------------------------------------- +// App registry +// --------------------------------------------------------------------------- + +const whisperCpp: TranscriptionApp = { + id: "whisper-cpp", + label: "whisper.cpp (whisper-cli)", + fields: { model: true, customArgs: true }, + defaultBin: () => getPaths().whisperBin, + build({ audioFile, outputBase, config }) { + const template = + config.customArgs && config.customArgs.length > 0 + ? config.customArgs + : DEFAULT_TRANSCRIBE_ARGS; + const model = config.model ?? getPaths().whisperModel; + const argv = substitutePlaceholders(template, { + audioFile, + outputBase, + model, + }); + // whisper-cli's `-of <base>` writes "<base>.json". + return { argv, outputFile: `${outputBase}.json`, outputFormat: "whisper-json" }; + }, + makeProgressParser: createTranscribeProgressParser, +}; + +const chough: TranscriptionApp = { + id: "chough", + label: "chough", + fields: { model: true, remoteUrl: true, chunkSize: true }, + defaultBin: () => process.env.CHOUGH_BIN ?? "chough", + build({ audioFile, outputBase, config }) { + const argv = ["-f", "json", "-o", outputBase]; + if (typeof config.chunkSize === "number" && config.chunkSize > 0) { + argv.push("-c", String(Math.floor(config.chunkSize))); + } + const env: Record<string, string> = {}; + const remoteUrl = config.remoteUrl?.trim(); + if (remoteUrl) { + argv.push("-r"); + env.CHOUGH_URL = remoteUrl; + } + const model = config.model?.trim(); + if (model) env.CHOUGH_MODEL = model; + argv.push(audioFile); + // chough writes EXACTLY the `-o` path — no ".json" is appended. + return { + argv, + env: Object.keys(env).length > 0 ? env : undefined, + outputFile: outputBase, + outputFormat: "chough-json", + }; + }, + makeProgressParser: createChoughProgressParser, +}; + +export const TRANSCRIPTION_APPS: Record<string, TranscriptionApp> = { + [whisperCpp.id]: whisperCpp, + [chough.id]: chough, +}; + +export const DEFAULT_TRANSCRIPTION_APP_ID = "whisper-cpp"; + +export function getTranscriptionApp(id: string | undefined): TranscriptionApp { + return ( + (id ? TRANSCRIPTION_APPS[id] : undefined) ?? + TRANSCRIPTION_APPS[DEFAULT_TRANSCRIPTION_APP_ID] + ); +} + +// A client-safe view of an app (id/label/fields only — no functions). Build it +// on the server and pass it to the settings form so the registry module (which +// reaches getPaths/process.env) never ends up in the client bundle. +export type TranscriptionAppDescriptor = { + id: string; + label: string; + fields: TranscriptionApp["fields"]; +}; + +export function listTranscriptionApps(): TranscriptionAppDescriptor[] { + return Object.values(TRANSCRIPTION_APPS).map((a) => ({ + id: a.id, + label: a.label, + fields: a.fields, + })); +} diff --git a/common/lib/whisper.ts b/common/lib/whisper.ts @@ -1,4 +1,5 @@ import type { Cue } from "./vtt"; +import type { TranscriptOutputFormat } from "./transcriptionApps"; type WhisperSegment = { offsets?: { from?: number; to?: number }; @@ -9,24 +10,95 @@ type WhisperDoc = { transcription?: WhisperSegment[]; }; -export function parseWhisper(src: string): Cue[] { - const doc = JSON.parse(src) as WhisperDoc; +type ChoughChunk = { + start_time?: number; + end_time?: number; + text?: string; +}; + +type ChoughDoc = { + chunk_data?: ChoughChunk[]; +}; + +// Append a cue, merging into the previous one when the text is identical +// (whisper/chough both repeat a line across adjacent chunks). Shared by both +// parsers so the merge behavior stays consistent. +function pushCue(cues: Cue[], start: number, end: number, text: string): void { + const trimmed = text.trim(); + if (!trimmed) return; + const prev = cues[cues.length - 1]; + if (prev && prev.text === trimmed) { + prev.end = end; + return; + } + cues.push({ start, end, text: trimmed }); +} + +function whisperCuesFromDoc(doc: WhisperDoc): Cue[] { const segments = doc.transcription ?? []; const cues: Cue[] = []; for (const seg of segments) { const fromMs = seg.offsets?.from; const toMs = seg.offsets?.to; if (typeof fromMs !== "number" || typeof toMs !== "number") continue; - const text = (seg.text ?? "").trim(); - if (!text) continue; - const start = fromMs / 1000; - const end = toMs / 1000; - const prev = cues[cues.length - 1]; - if (prev && prev.text === text) { - prev.end = end; - continue; - } - cues.push({ start, end, text }); + pushCue(cues, fromMs / 1000, toMs / 1000, seg.text ?? ""); } return cues; } + +function choughCuesFromDoc(doc: ChoughDoc): Cue[] { + const chunks = doc.chunk_data ?? []; + const cues: Cue[] = []; + for (const chunk of chunks) { + const start = chunk.start_time; + const end = chunk.end_time; + if (typeof start !== "number" || typeof end !== "number") continue; + // chough times are already in seconds. + pushCue(cues, start, end, chunk.text ?? ""); + } + return cues; +} + +// whisper.cpp JSON: { transcription: [{ offsets: { from, to (ms) }, text }] }. +export function parseWhisper(src: string): Cue[] { + return whisperCuesFromDoc(JSON.parse(src) as WhisperDoc); +} + +// chough native JSON: { chunk_data: [{ start_time, end_time (sec), text }], ... }. +export function parseChough(src: string): Cue[] { + return choughCuesFromDoc(JSON.parse(src) as ChoughDoc); +} + +// Sniff a transcript.json shape. Returns null for an unrecognized shape so the +// caller can fall back to whisper (the only app before chough). +export function detectTranscriptFormat( + src: string, +): "whisper-json" | "chough-json" | null { + let doc: unknown; + try { + doc = JSON.parse(src); + } catch { + return null; + } + if (doc && typeof doc === "object") { + const d = doc as Record<string, unknown>; + if (Array.isArray(d.chunk_data)) return "chough-json"; + if (Array.isArray(d.transcription)) return "whisper-json"; + } + return null; +} + +// Parse a transcript.json into cues. `hint` is the authoritative format when +// known (e.g. the app that just produced the file, or a recorded per-video +// tag); when absent or "vtt", the shape is sniffed, and anything unrecognized +// falls back to whisper — the only transcription format in use before chough. +export function parseTranscriptJson( + src: string, + hint?: TranscriptOutputFormat | null, +): Cue[] { + const fmt = + hint === "whisper-json" || hint === "chough-json" + ? hint + : detectTranscriptFormat(src) ?? "whisper-json"; + return fmt === "chough-json" ? parseChough(src) : parseWhisper(src); +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template). - **Global default for parallel transcriptions.** A new **Settings → Parallel transcriptions** field sets how many videos a "Transcribe missing" / bucket run transcribes at once when its per-run Concurrency input is left blank (default **2**, clamped 1–16). This replaces the old `PARALLEL_TRANSCRIBE_LIMIT` env-var default of 4 as the source of the default: the value is now persisted in `settings.json`, shown as the placeholder in each channel's Concurrency input, and used as the server-side fallback. The per-run Concurrency input still overrides it for a single run. - **Managed downloads skip live and upcoming videos by default.** Before downloading each video, the managed downloader now runs a quick metadata-only pass, then evaluates an app-level filter: videos that are currently live or scheduled/upcoming are skipped (their finished VODs still download normally — a skip isn't archived, so the next **Sync** / **Download missing** retries the video once the stream ends). A skip is recorded in the video's `download-outcome.json` (`status: "skipped-filtered"` with the filter name and reason) and logged to `download.log`; it never counts as a failure or aborts the batch, and the run's summary line reports how many were skipped. Skipped-live videos are surfaced in a new **Skipped: live or upcoming** bucket in the channel's Diagnostics (with a **Retry** to force an attempt now). On by default for every channel via a new **Settings → Skip live and upcoming videos** toggle, with a per-channel override (`config.json` `skipLiveDownloads`). To support the filter (and as a reusable building block for future per-video decisions), each managed download is now split into a metadata fetch followed by the real download, which reuses that metadata via `--load-info-json` so the second pass doesn't re-extract; the audio-integrity-checked download path re-extracts as before. Turn the toggle off for a channel that streams nothing to allow live captures. - **Global default social links, overridable per site.** Footer social links can now be defined once in **Settings → Social links** and apply to every site by default. Each site's form gained a **Use global default social links** checkbox (on by default): leave it checked to inherit the global list, or uncheck it to give that site its own links (an empty list shows none). Existing sites keep their current footer — sites that already had links stay as explicit overrides until you flip the toggle. Validation and SVG sanitization are unchanged and shared by both the global and per-site editors. diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -13,6 +13,10 @@ import { type SiteSettings, type SocialLink, } from "yt-dlp-transcript-common/lib/settings"; +import { + TRANSCRIPTION_APPS, + type AppInstanceConfig, +} from "yt-dlp-transcript-common/lib/transcriptionApps"; export type SaveResult = { ok: true } | { ok: false; error: string }; @@ -22,8 +26,6 @@ export async function saveSettingsAction( ): Promise<SaveResult> { const adminTitle = String(formData.get("adminTitle") ?? "").trim(); const maxBytesRaw = String(formData.get("maxTranscriptPageBytes") ?? "").trim(); - const transcribeBin = String(formData.get("transcribeBin") ?? "").trim(); - const transcribeModel = String(formData.get("transcribeModel") ?? "").trim(); const cookiesFromBrowser = String( formData.get("cookiesFromBrowser") ?? "", ).trim(); @@ -36,19 +38,67 @@ export async function saveSettingsAction( const inlineTranscribeOnFallback = formData.get("inlineTranscribeOnFallback") === "on"; const skipLiveDownloads = formData.get("skipLiveDownloads") === "on"; - const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? ""); - const transcribeArgs = transcribeArgsRaw - .split("\n") - .map((s) => s.trim()) - .filter(Boolean); if (!adminTitle) return { ok: false, error: "Admin title is required" }; - if (!transcribeBin) { - return { ok: false, error: "Transcribe binary is required" }; + + const transcriptionApp = String(formData.get("transcriptionApp") ?? "").trim(); + const activeApp = TRANSCRIPTION_APPS[transcriptionApp]; + if (!activeApp) { + return { ok: false, error: "Unknown transcription app" }; } - const argsErr = validateTranscribeArgs(transcribeArgs); - if (argsErr) return { ok: false, error: argsErr }; + // Collect per-app config from namespaced fields (app.<id>.<field>). Only the + // selected app's customArgs are validated; writeSettings re-checks the + // resolved binary. + const transcriptionApps: Record<string, AppInstanceConfig> = {}; + for (const app of Object.values(TRANSCRIPTION_APPS)) { + const cfg: AppInstanceConfig = {}; + const bin = String(formData.get(`app.${app.id}.bin`) ?? "").trim(); + if (bin) cfg.bin = bin; + if (app.fields.model) { + const model = String(formData.get(`app.${app.id}.model`) ?? "").trim(); + if (model) cfg.model = model; + } + if (app.fields.remoteUrl) { + const remoteUrl = String( + formData.get(`app.${app.id}.remoteUrl`) ?? "", + ).trim(); + if (remoteUrl) cfg.remoteUrl = remoteUrl; + } + if (app.fields.chunkSize) { + const chunkRaw = String( + formData.get(`app.${app.id}.chunkSize`) ?? "", + ).trim(); + if (chunkRaw) { + const n = Number.parseInt(chunkRaw, 10); + if (!Number.isFinite(n) || n <= 0) { + return { + ok: false, + error: `${app.label} chunk size must be a positive number`, + }; + } + cfg.chunkSize = n; + } + } + if (app.fields.customArgs) { + const customArgs = String(formData.get(`app.${app.id}.customArgs`) ?? "") + .split("\n") + .map((s) => s.trim()) + .filter(Boolean); + if (customArgs.length > 0) cfg.customArgs = customArgs; + } + if (Object.keys(cfg).length > 0) transcriptionApps[app.id] = cfg; + } + + const activeCfg = transcriptionApps[transcriptionApp]; + if ( + activeApp.fields.customArgs && + activeCfg?.customArgs && + activeCfg.customArgs.length > 0 + ) { + const argsErr = validateTranscribeArgs(activeCfg.customArgs); + if (argsErr) return { ok: false, error: argsErr }; + } const parsed = Number.parseInt(maxBytesRaw, 10); if (!Number.isFinite(parsed)) { @@ -121,9 +171,8 @@ export async function saveSettingsAction( const next: SiteSettings = { adminTitle, maxTranscriptPageBytes: parsed, - transcribeBin, - transcribeModel, - transcribeArgs, + transcriptionApp, + transcriptionApps, cookiesFromBrowser, sleepBetweenDownloadsSeconds: sleepParsed, parallelTranscriptions: parallelParsed, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -6,6 +6,7 @@ import { type SaveResult, } from "../actions"; import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings"; +import type { TranscriptionAppDescriptor } from "yt-dlp-transcript-common/lib/transcriptionApps"; import { SocialLinksField, toSocialRow, @@ -14,9 +15,10 @@ import { type Props = { initial: SiteSettings; + apps: TranscriptionAppDescriptor[]; }; -export function SettingsForm({ initial }: Props) { +export function SettingsForm({ initial, apps }: Props) { const [state, formAction] = useActionState<SaveResult | undefined, FormData>( saveSettingsAction, undefined, @@ -24,6 +26,7 @@ export function SettingsForm({ initial }: Props) { const [social, setSocial] = useState<SocialRow[]>(() => initial.socialLinks.map(toSocialRow), ); + const [appId, setAppId] = useState<string>(initial.transcriptionApp); return ( <form action={formAction} className="flex flex-col gap-4 max-w-xl"> @@ -42,44 +45,93 @@ export function SettingsForm({ initial }: Props) { type="number" /> <fieldset className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded p-3"> - <legend className="px-1 text-sm font-medium">Transcription command</legend> + <legend className="px-1 text-sm font-medium">Transcription</legend> <p className="text-xs text-zinc-500"> - Used by the channel "Transcribe missing" / "Retry failures" actions. - The command must write <code>{`<outputBase>.json`}</code> in - whisper.cpp's <code>{`{ transcription: [{ offsets, text }] }`}</code>{" "} - format. Available placeholders:{" "} - <code>{"{audioFile}"}</code> (required),{" "} - <code>{"{outputBase}"}</code> (required),{" "} - <code>{"{model}"}</code> (optional). + The app used by the channel &ldquo;Transcribe missing&rdquo; / + &ldquo;Retry failures&rdquo; actions. Each app handles its own + arguments and output format. Per-app fields below; only the selected + app&apos;s settings are used. </p> - <Field - label="Binary" - name="transcribeBin" - defaultValue={initial.transcribeBin} - required - hint="e.g., whisper-cli, /usr/local/bin/whisper, or your own wrapper." - /> - <Field - label="Model" - name="transcribeModel" - defaultValue={initial.transcribeModel} - hint="Substituted for {model}. Leave blank if your binary doesn't take a model." - /> <label className="flex flex-col gap-1 text-sm"> - <span className="font-medium">Args (one per line)</span> - <textarea - name="transcribeArgs" - defaultValue={initial.transcribeArgs.join("\n")} - rows={Math.max(6, initial.transcribeArgs.length + 1)} - className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono" - /> - <span className="text-xs text-zinc-500"> - Default whisper.cpp call:{" "} - <code> - -ojf -l en -m {"{model}"} -of {"{outputBase}"} {"{audioFile}"} - </code> - </span> + <span className="font-medium">App</span> + <select + name="transcriptionApp" + value={appId} + onChange={(e) => setAppId(e.target.value)} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm" + > + {apps.map((app) => ( + <option key={app.id} value={app.id}> + {app.label} + </option> + ))} + </select> </label> + {apps.map((app) => { + const cfg = initial.transcriptionApps[app.id] ?? {}; + return ( + <div + key={app.id} + hidden={app.id !== appId} + className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-3" + > + <Field + label="Binary" + name={`app.${app.id}.bin`} + defaultValue={cfg.bin ?? ""} + hint="Path or name of the executable. Leave blank to use the default for this app." + /> + {app.fields.model && ( + <Field + label="Model" + name={`app.${app.id}.model`} + defaultValue={cfg.model ?? ""} + hint="whisper.cpp: substituted for {model}. chough: sets CHOUGH_MODEL. Leave blank for the app default." + /> + )} + {app.fields.remoteUrl && ( + <Field + label="Remote URL" + name={`app.${app.id}.remoteUrl`} + defaultValue={cfg.remoteUrl ?? ""} + hint="chough only: transcribe via a remote chough --server (CHOUGH_URL). Leave blank for local transcription." + /> + )} + {app.fields.chunkSize && ( + <Field + label="Chunk size (seconds)" + name={`app.${app.id}.chunkSize`} + defaultValue={ + cfg.chunkSize !== undefined ? String(cfg.chunkSize) : "" + } + type="number" + hint="chough only (-c). Leave blank for the app default (60s)." + /> + )} + {app.fields.customArgs && ( + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Custom args (one per line)</span> + <textarea + name={`app.${app.id}.customArgs`} + defaultValue={(cfg.customArgs ?? []).join("\n")} + rows={6} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono" + /> + <span className="text-xs text-zinc-500"> + Leave blank for the default whisper.cpp call:{" "} + <code> + -ojf -l en -m {"{model}"} -of {"{outputBase}"}{" "} + {"{audioFile}"} + </code> + . Placeholders: <code>{"{audioFile}"}</code> (required),{" "} + <code>{"{outputBase}"}</code> (required),{" "} + <code>{"{model}"}</code> (optional). + </span> + </label> + )} + </div> + ); + })} </fieldset> <Field label="Cookies from browser (retry only)" diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx @@ -1,6 +1,7 @@ import type { Metadata } from "next"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; +import { listTranscriptionApps } from "yt-dlp-transcript-common/lib/transcriptionApps"; import { SettingsForm } from "./components/SettingsForm"; export const dynamic = "force-dynamic"; @@ -35,7 +36,7 @@ export default function SettingsPage() { Used by the export build to title the static site. </p> </div> - <SettingsForm initial={settings} /> + <SettingsForm initial={settings} apps={listTranscriptionApps()} /> </section> <section className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6"> @@ -46,9 +47,9 @@ export default function SettingsPage() { <p className="text-sm text-zinc-500 mt-2"> Read-only view of <code>getPaths()</code>. Set the matching env vars (TRANSCRIPTS_DIR, SETTINGS_FILE, EXPORT_PUBLIC_DIR, YTDLP_BIN, - WHISPER_BIN, WHISPER_MODEL) before launching to override. The - transcription command above takes - precedence over <code>whisperBin</code> / <code>whisperModel</code>{" "} + WHISPER_BIN, WHISPER_MODEL, CHOUGH_BIN, CHOUGH_URL, CHOUGH_MODEL) + before launching to override. The per-app Binary / Model fields in + the transcription settings above take precedence over these defaults when running channel transcribe actions. </p> <table className="text-sm mt-3 w-full"> diff --git a/editor/e2e/chough.spec.ts b/editor/e2e/chough.spec.ts @@ -0,0 +1,82 @@ +import { readFile, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath, writeSettings } from "./helpers"; + +const CHANNEL = "test-transcribe"; +const DATA = `test-transcripts/channels/${CHANNEL}/data`; + +// Minimal metadata so transcribeOne's normalize pass runs (it needs +// metadata.info.json) and writes transcript.cues.json. +const META = JSON.stringify({ + id: "vidA", + title: "Synthetic chough video", + upload_date: "20240101", + duration: 10, +}); + +async function selectChough() { + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + transcriptionApp: "chough", + transcriptionApps: { chough: {} }, + }); +} + +test("chough app transcribes audio and writes a transcript.json", async ({ + page, +}) => { + await resetData("one-transcribe-channel-with-audio"); + await selectChough(); + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Transcribe missing" }).click(); + await expect(page.getByLabel("Transcribe missing output")).toContainText( + "3 succeeded", + { timeout: 30_000 }, + ); + + for (const id of ["vidA", "vidB", "vidC"]) { + expect(await pathExists(`${DATA}/${id}/transcript.json`)).toBe(true); + } + + // The raw transcript is chough-native (chunk_data with seconds). Reaching this + // shape proves the chough app's argv ran AND that transcribeOne correctly + // renamed chough's exact `-o` output (no ".json" appended) to transcript.json. + const raw = JSON.parse( + await readFile(resolvePath(`${DATA}/vidA/transcript.json`), "utf8"), + ); + expect(Array.isArray(raw.chunk_data)).toBe(true); + expect(raw.transcription).toBeUndefined(); +}); + +test("chough output normalizes to chough-tagged cues", async ({ page }) => { + await resetData("one-transcribe-channel-with-audio"); + await selectChough(); + // Give vidA metadata so the normalize pass produces transcript.cues.json. + await writeFile(resolvePath(`${DATA}/vidA/metadata.info.json`), META); + + await page.goto(`/channels/${CHANNEL}`); + await page.getByRole("button", { name: "Transcribe missing" }).click(); + await expect(page.getByLabel("Transcribe missing output")).toContainText( + "3 succeeded", + { timeout: 30_000 }, + ); + + await expect + .poll(() => pathExists(`${DATA}/vidA/transcript.cues.json`), { + timeout: 15_000, + }) + .toBe(true); + + const cues = JSON.parse( + await readFile(resolvePath(`${DATA}/vidA/transcript.cues.json`), "utf8"), + ); + // The durable per-video format tag is recorded… + expect(cues.transcriptFormat).toBe("chough-json"); + // …and the cues come from chough's seconds-based chunk_data (start 0, end 5), + // not misread as whisper milliseconds. + expect(cues.cues.length).toBeGreaterThan(0); + expect(cues.cues[0].start).toBe(0); + expect(cues.cues[0].end).toBe(5); +}); diff --git a/editor/e2e/fixtures/bin/fake-chough.mjs b/editor/e2e/fixtures/bin/fake-chough.mjs @@ -0,0 +1,57 @@ +#!/usr/bin/env node +// E2E fake chough. Mimics the args the chough app builds (transcriptionApps.ts): +// chough -f json -o <tmpBase> [-c <n>] [-r] <audioFilename> +// Writes the output to EXACTLY <tmpBase> (no ".json" appended — this is the key +// behavioural difference from whisper-cli's -of), in chough's native JSON shape +// { duration_seconds, chunks, text, chunk_data: [{ start_time, end_time, text }] } +// with times in SECONDS. Run from cwd == the video dir (set by runWhisperBatch). +import { writeFile } from "node:fs/promises"; + +const argv = process.argv.slice(2); +function arg(flag) { + const i = argv.indexOf(flag); + return i < 0 ? undefined : argv[i + 1]; +} + +const out = arg("-o"); +const audio = argv[argv.length - 1]; +if (!out) { + process.stderr.write(`[fake-chough] missing -o\n`); + process.exit(2); +} + +const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)); + +// Videos whose dir (== video id) contains the SLOWOP marker emit chough-style +// progress output (audio header + █/░ bars + ETA) with real delays, so the +// per-operation progress bars and drain behaviour can be observed. Videos +// without the marker stay instant so the rest of the suite is fast. +const totalSec = 50; +if (process.cwd().toLowerCase().includes("slowop")) { + process.stdout.write(`mode: local\n`); + process.stdout.write(`audio: ${totalSec}.0s • chunks: 60s • format: json\n`); + const width = 40; + for (let step = 1; step <= 10; step += 1) { + const filled = Math.round((step / 10) * width); + const bar = "█".repeat(filled) + "░".repeat(width - filled); + const etaSec = Math.max(0, Math.round(((10 - step) / 10) * 30)); + process.stdout.write( + `${bar} ETA ${Math.floor(etaSec / 60)}m ${etaSec % 60}s\x1b[K\n`, + ); + await sleep(700); + } +} + +const doc = { + duration_seconds: totalSec, + chunks: 2, + text: `Synthetic chough output for ${audio} #1 Synthetic chough output for ${audio} #2`, + chunk_data: [ + { start_time: 0, end_time: 5, text: `Synthetic chough output for ${audio} #1` }, + { start_time: 5, end_time: 10, text: `Synthetic chough output for ${audio} #2` }, + ], +}; + +// chough writes EXACTLY the -o path (no extension munging). +await writeFile(out, JSON.stringify(doc)); +process.stdout.write(`Output: ${out}\n`); diff --git a/editor/e2e/transcription-app-migration.spec.ts b/editor/e2e/transcription-app-migration.spec.ts @@ -0,0 +1,38 @@ +import { test, expect } from "@playwright/test"; +import { resetData, writeSettings } from "./helpers"; + +// A pre-multi-app settings.json (transcribeBin/transcribeArgs/transcribeModel, +// no transcriptionApp key) must migrate onto the app registry when read, so the +// Settings page shows a selected app rather than crashing on missing fields. + +test("legacy whisper settings migrate onto the whisper-cpp app", async ({ + page, +}) => { + await resetData(null); + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + transcribeBin: "whisper-cli", + transcribeModel: "/models/ggml.bin", + transcribeArgs: ["-ojf", "-l", "en", "-m", "{model}", "-of", "{outputBase}", "{audioFile}"], + }); + await page.goto("/settings"); + await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("whisper-cpp"); +}); + +test("legacy settings whose binary is chough migrate onto the chough app", async ({ + page, +}) => { + await resetData(null); + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + transcribeBin: "chough", + transcribeModel: "", + transcribeArgs: ["-f", "json", "-o", "{outputBase}", "{audioFile}"], + }); + await page.goto("/settings"); + await expect(page.locator('select[name="transcriptionApp"]')).toHaveValue("chough"); +}); diff --git a/editor/package.json b/editor/package.json @@ -5,8 +5,8 @@ "type": "module", "scripts": { "dev": "next dev --port 3001", - "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011", - "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011", + "dev:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next dev --port 3011", + "start:test": "TRANSCRIPTS_DIR=$(pwd)/test-transcripts EXPORT_PUBLIC_DIR=$(pwd)/test-transcripts/.export-public SETTINGS_FILE=$(pwd)/test-settings.json YTDLP_BIN=$(pwd)/e2e/fixtures/bin/fake-ytdlp.mjs WHISPER_BIN=$(pwd)/e2e/fixtures/bin/fake-whisper.mjs WHISPER_MODEL=/dev/null CHOUGH_BIN=$(pwd)/e2e/fixtures/bin/fake-chough.mjs CHOUGH_MODEL=/dev/null FFMPEG_BIN=$(pwd)/e2e/fixtures/bin/fake-ffmpeg.mjs AUDIO_CHECK_INTERVAL_MS_OVERRIDE=300 AUDIO_CHECK_SIZE_GATE_OVERRIDE=4096 next start --port 3011", "build": "next build", "start": "next start --port 3001", "lint": "eslint",