// First-class registry of DIGEST apps — the engines that turn a transcript into // derived content (chapters, topic tags). Deliberately mirrors // transcriptionApps.ts: each app owns how it is invoked, how its config maps onto // that invocation, and how its raw output becomes a parsed JSON body. The editor // selects one app per section, storing a small per-app config block // (settings.digest.apps[id]); digestVideo resolves the app and runs it. // // LOCAL-FIRST, BY DECISION. Hosted AI is not a dependency this project can rely // on given the nature of the archived content, so `ollama-direct` carries the // corpus and making it good enough IS the goal. `claude-code` is built but OFF by // default (settings.digest.remoteEnabled === false) — an opt-in overflow for the // >4h tail or a channel where local quality is poor, never the default. // // Pure definitions only (no I/O beyond reading process.env / getPaths inside // builders) so both the `common` controllers and the editor UI can import this. // Dependency direction is one-way: settings.ts -> digestApps.ts -> { paths, // parseStdoutJson }. Keep it that way to avoid import cycles. import { execa } from "execa"; import { getPaths } from "./paths"; import { extractJsonObject, parseStdoutJson } from "./parseStdoutJson"; import { CLAUDE_DIGEST_APP_ID, DEFAULT_DIGEST_APP_ID, DEFAULT_DIGEST_NUM_CTX, OLLAMA_DIGEST_APP_ID, type DigestAppConfig, type DigestLane, } from "./digest"; // The lane type, the app ids and DigestAppConfig are defined in the client-safe // digest.ts and re-exported here, so settings.ts can consume them without // reaching this module (which imports execa). `lane` is also WHY the two engines // get SEPARATE queue keys: registry.ts hardcodes concurrency 1 per key, so one // shared key would serialize a GPU-bound lane behind a network-bound one and // waste half the throughput of a multi-week sweep. export { CLAUDE_DIGEST_APP_ID, DEFAULT_DIGEST_APP_ID, OLLAMA_DIGEST_APP_ID }; export type { DigestAppConfig, DigestLane }; export type DigestRunInput = { system: string; prompt: string; // JSON schema for the expected body. Engines with constrained decoding // (ollama's `format`) enforce it; CLI lanes get it embedded in the prompt. schema: Record; config: DigestAppConfig; signal?: AbortSignal; onLog?: (msg: string) => void; }; export type DigestRunResult = { // The parsed JSON body. Shape validation is the parser's job (digestParse.ts), // not the engine's — an engine only guarantees "this is JSON". data: unknown; // The model that ACTUALLY ran, as reported by the engine when it says so. // Recorded in provenance, so a config of "qwen2.5" resolving to "qwen2.5:7b" // doesn't later look like a model change and trigger a needless regeneration. model: string; // WALL time, bracketing the fetch — not engine time. The distinction is the // whole reason the fields below exist: "27 s/audio-hour vs 90" was a comparison // of an engine number against a wall number, and nothing recorded could tell // them apart after the fact. durationMs: number; // Metered lanes only. costUsd?: number; inputTokens?: number; outputTokens?: number; // Engine-reported timing breakdown, when the engine reports one (ollama does, // on every /api/chat response; the CLI lane does not). All optional and all // ADDITIVE — nothing here reaches DigestProvenance, so recording them changes // no freshness identity and invalidates no existing digest. // // Why it is worth the four lines: this splits a call into model-load, prefill // and decode. Warmup and a cold model load then stop masquerading as // contention, and a contended run shows WHERE it was hurt — prefill (shader // competition) reads differently from decode (memory bandwidth). Comparing // totalMs against durationMs also quantifies non-engine wall inflation, which // on an idle box measured 1.004× and under contention is unbounded. loadMs?: number; promptEvalMs?: number; evalMs?: number; // The engine's own total. NOT the sum of the three above: ollama's // total_duration also covers queueing and tokenization inside the server. totalMs?: number; }; export type DigestApp = { id: string; label: string; lane: DigestLane; // True when a run costs money. Metered apps are opt-in, never auto-queued, and // their per-call count + cumulative cost is logged by the batch. metered: boolean; // Which config fields this app surfaces in the settings UI / consumes. fields: { bin?: boolean; baseUrl?: boolean; model?: boolean; numCtx?: boolean; temperature?: boolean; }; // Model used when the per-app `model` override is empty. defaultModel: () => string; // Cheap reachability probe, so the editor can say "ollama is down" instead of // failing 74k items one at a time. Mirrors pingRemoteHealth's contract: true // only on a positive response, never throws. probe: (config: DigestAppConfig) => Promise; run: (input: DigestRunInput) => Promise; }; // Fallbacks shared by both apps. The context default lives in digest.ts, so the // pure prompt module can size a chunk against the same number this resolves to — // two copies would let the chunker and the engine disagree, and ollama truncates // the excess SILENTLY. const DEFAULT_NUM_CTX = DEFAULT_DIGEST_NUM_CTX; const DEFAULT_TEMPERATURE = 0; const DEFAULT_TIMEOUT_MS = 10 * 60_000; function resolveNumCtx(config: DigestAppConfig): number { return typeof config.numCtx === "number" && config.numCtx > 0 ? Math.floor(config.numCtx) : DEFAULT_NUM_CTX; } function resolveTemperature(config: DigestAppConfig): number { return typeof config.temperature === "number" && config.temperature >= 0 ? config.temperature : DEFAULT_TEMPERATURE; } function resolveTimeoutMs(config: DigestAppConfig): number { return typeof config.timeoutMs === "number" && config.timeoutMs > 0 ? Math.floor(config.timeoutMs) : DEFAULT_TIMEOUT_MS; } // --------------------------------------------------------------------------- // ollama-direct — the local GPU lane that carries the corpus // --------------------------------------------------------------------------- function ollamaBase(config: DigestAppConfig): string { const override = config.baseUrl?.trim(); // Strip trailing slashes the way remoteTranscribe.ts normalizes a worker base, // so callers can always concatenate a leading-slash path. return (override ? override.replace(/\/+$/, "") : getPaths().ollamaUrl) || ""; } const ollamaDirect: DigestApp = { id: OLLAMA_DIGEST_APP_ID, label: "ollama (local, JSON schema)", lane: "local-gpu", metered: false, fields: { baseUrl: true, model: true, numCtx: true, temperature: true }, defaultModel: () => process.env.OLLAMA_DIGEST_MODEL ?? "qwen2.5:7b", async probe(config) { const base = ollamaBase(config); if (!base) return false; try { const res = await fetch(`${base}/api/tags`, { signal: AbortSignal.timeout(3000), }); return res.ok; } catch { return false; } }, async run({ system, prompt, schema, config, signal, onLog }) { const base = ollamaBase(config); if (!base) throw new Error("ollama URL is not configured (set OLLAMA_URL)"); const model = config.model?.trim() || ollamaDirect.defaultModel(); const numCtx = resolveNumCtx(config); const startedAt = Date.now(); // The timeout is combined with the caller's cancel signal so a drain/cancel // stops a long generation promptly AND a wedged engine can't stall forever. const timeout = AbortSignal.timeout(resolveTimeoutMs(config)); const composed = signal ? AbortSignal.any([signal, timeout]) : timeout; let res: Response; try { res = await fetch(`${base}/api/chat`, { method: "POST", headers: { "content-type": "application/json" }, signal: composed, body: JSON.stringify({ model, stream: false, // Omitted entirely unless configured: ollama rejects `think` for // models that have no reasoning mode, so sending a default would // break every non-reasoning engine. ...(typeof config.think === "boolean" ? { think: config.think } : {}), // Schema-constrained decoding. This — not prompt wording — is what // made a 7B model emit well-formed timestamps. format: schema, options: { // Explicit, always. See DigestAppConfig.numCtx. num_ctx: numCtx, temperature: resolveTemperature(config), }, messages: [ { role: "system", content: system }, { role: "user", content: prompt }, ], }), }); } catch (err) { const message = (err as Error)?.message ?? String(err); if (signal?.aborted) throw err; throw new Error( `ollama request failed (${base}): ${message}. Is the ollama service running?`, ); } if (!res.ok) { const body = await res.text().catch(() => ""); throw new Error( `ollama returned ${res.status}: ${body.slice(0, 300) || res.statusText}`, ); } const body = (await res.json()) as { model?: string; message?: { content?: string }; prompt_eval_count?: number; eval_count?: number; // Nanoseconds, all four. ollama has always returned these; they were simply // being thrown away. total_duration?: number; load_duration?: number; prompt_eval_duration?: number; eval_duration?: number; }; const content = body.message?.content ?? ""; const data = extractJsonObject(content); if (data === null) { throw new Error( `ollama returned unparseable content: ${content.slice(0, 300)}`, ); } const durationMs = Date.now() - startedAt; const loadMs = nsToMs(body.load_duration); const promptEvalMs = nsToMs(body.prompt_eval_duration); const evalMs = nsToMs(body.eval_duration); const totalMs = nsToMs(body.total_duration); // The breakdown goes in the LOG, not just the return value: a multi-week // sweep's job logs are the only surviving record of how it actually behaved, // and the last measurement's logs had already rotated away when the numbers // needed reconciling. `wall` beside `engine` is what makes the yield gate's // own idling visible. const parts = [ loadMs !== undefined ? `load ${secs(loadMs)}` : null, promptEvalMs !== undefined ? `prefill ${secs(promptEvalMs)}` : null, evalMs !== undefined ? `decode ${secs(evalMs)}` : null, totalMs !== undefined ? `engine ${secs(totalMs)}` : null, ].filter(Boolean); onLog?.( `ollama ${body.model ?? model}: ${body.prompt_eval_count ?? "?"} in / ${ body.eval_count ?? "?" } out tokens in ${secs(durationMs)} wall` + (parts.length ? ` (${parts.join(", ")})` : "") + ` (num_ctx ${numCtx})`, ); return { data, model: body.model ?? model, durationMs, inputTokens: body.prompt_eval_count, outputTokens: body.eval_count, ...(loadMs !== undefined ? { loadMs } : {}), ...(promptEvalMs !== undefined ? { promptEvalMs } : {}), ...(evalMs !== undefined ? { evalMs } : {}), ...(totalMs !== undefined ? { totalMs } : {}), }; }, }; // ollama reports durations in NANOSECONDS. Absent or non-finite means the engine // did not report that phase (an older build, or a cached-prompt call that skipped // prefill entirely), and the field is then omitted rather than recorded as 0 — a // zero here would look like a measurement, not a gap. function nsToMs(ns: number | undefined): number | undefined { return typeof ns === "number" && Number.isFinite(ns) && ns >= 0 ? Math.round(ns / 1e6) : undefined; } function secs(ms: number): string { return `${Math.round(ms / 100) / 10}s`; } // --------------------------------------------------------------------------- // claude-code — the metered overflow lane, OFF by default // --------------------------------------------------------------------------- const claudeCode: DigestApp = { id: CLAUDE_DIGEST_APP_ID, label: "Claude Code CLI (metered)", lane: "remote-api", metered: true, fields: { bin: true, model: true }, defaultModel: () => process.env.CLAUDE_DIGEST_MODEL ?? "", async probe(config) { const bin = config.bin?.trim() || getPaths().claudeBin; try { const res = await execa(bin, ["--version"], { buffer: true, reject: false, timeout: 10_000, }); return res.exitCode === 0; } catch { return false; } }, async run({ system, prompt, schema, config, signal, onLog }) { const bin = config.bin?.trim() || getPaths().claudeBin; const model = config.model?.trim() || claudeCode.defaultModel(); const startedAt = Date.now(); const argv = ["-p", "--output-format", "json"]; if (model) argv.push("--model", model); // No constrained decoding over the CLI, so the contract goes in the prompt // and the parser's guards do the enforcing. The prompt is passed on stdin // rather than argv: a 12k-token chunk is ~50 KB and does not belong in an // argument list. const stdin = [ system, "", "Reply with a single JSON object and nothing else — no prose, no code fence.", "It must validate against this JSON schema:", JSON.stringify(schema), "", prompt, ].join("\n"); let result: { exitCode: number | null; stdout: string; stderr: string }; try { result = (await execa(bin, argv, { input: stdin, buffer: true, reject: false, cancelSignal: signal, timeout: resolveTimeoutMs(config), })) as typeof result; } catch (err) { const message = (err as Error)?.message ?? String(err); // Same ENOENT-to-friendly-message treatment as the gallery-dl fetcher: a // missing optional binary is a configuration problem, not a crash. throw new Error( /ENOENT/.test(message) ? `claude CLI not found (set CLAUDE_BIN)` : message, ); } if (result.exitCode !== 0) { throw new Error( `claude exited ${result.exitCode}: ${(result.stderr || result.stdout).slice(0, 300)}`, ); } // Two layers: the CLI's own JSON wrapper, then the model's JSON body inside // the wrapper's `result` string. const wrapper = parseStdoutJson(result.stdout) as { result?: unknown; total_cost_usd?: unknown; usage?: { input_tokens?: number; output_tokens?: number }; is_error?: unknown; } | null; if (!wrapper || typeof wrapper !== "object") { throw new Error( `claude produced no parseable JSON wrapper: ${result.stdout.slice(0, 300)}`, ); } if (wrapper.is_error === true) { throw new Error( `claude reported an error: ${String(wrapper.result).slice(0, 300)}`, ); } const inner = typeof wrapper.result === "string" ? wrapper.result : ""; const data = extractJsonObject(inner); if (data === null) { throw new Error( `claude returned unparseable content: ${inner.slice(0, 300)}`, ); } const costUsd = typeof wrapper.total_cost_usd === "number" ? wrapper.total_cost_usd : undefined; onLog?.( `claude ${model || "(default model)"}: ${ costUsd !== undefined ? `$${costUsd.toFixed(4)}` : "cost unknown" } in ${Math.round((Date.now() - startedAt) / 100) / 10}s`, ); return { data, model: model || "claude-default", durationMs: Date.now() - startedAt, ...(costUsd !== undefined ? { costUsd } : {}), inputTokens: wrapper.usage?.input_tokens, outputTokens: wrapper.usage?.output_tokens, }; }, }; // --------------------------------------------------------------------------- // Registry // --------------------------------------------------------------------------- export const DIGEST_APPS: Record = { [ollamaDirect.id]: ollamaDirect, [claudeCode.id]: claudeCode, }; // Total by construction — an unknown id falls back to the local default rather // than throwing, so a hand-edited settings.json can never crash a sweep. export function getDigestApp(id: string | undefined): DigestApp { return ( (id ? DIGEST_APPS[id] : undefined) ?? DIGEST_APPS[DEFAULT_DIGEST_APP_ID] ); } export function isDigestAppId(id: unknown): id is string { return typeof id === "string" && Boolean(DIGEST_APPS[id]); } // A client-safe view of an app (no functions). Build it on the server and pass it // to the settings form so this module — which reaches getPaths()/process.env — // never ends up in the client bundle. export type DigestAppDescriptor = { id: string; label: string; lane: DigestLane; metered: boolean; fields: DigestApp["fields"]; defaultModel: string; }; export function listDigestApps(): DigestAppDescriptor[] { return Object.values(DIGEST_APPS).map((a) => ({ id: a.id, label: a.label, lane: a.lane, metered: a.metered, fields: a.fields, defaultModel: a.defaultModel(), })); }