#!/usr/bin/env node // A stand-in for the ollama HTTP API, for e2e. // // WHY THIS IS A SERVER AND NOT A FAKE BINARY. Every other engine in this repo is // a subprocess, so the e2e idiom is a fake executable in e2e/fixtures/bin/ wired // in through an env var. `ollama-direct` is different: it POSTs to // `${ollamaUrl}/api/chat` (common/lib/digestApps.ts), so there is no binary to // replace. It therefore runs as a third playwright `webServer` and the editor is // pointed at it with OLLAMA_URL. // // It answers from the PROMPT, not from a canned script, so it exercises the real // contract: the chunk range the model is told to stay inside is parsed back out // of the prompt text and every emitted start is placed within it. That means the // stub works unchanged for both timestamp modes — in chunk-local mode the prompt // states 00:00:00–00:44:26 and the stub answers in that numbering, which is // exactly what a real model does and what the parser must offset back. // // DETERMINISTIC BAD-OUTPUT MODE. When the request carries the BADOUT sentinel // (the SLOWOP convention from fake-whisper.mjs et al — matched case-insensitively // anywhere in the request body, so it can be carried by a video id or a title), // the stub emits one valid chapter plus one of each measured failure: a malformed // stamp, a non-Latin title, and an out-of-range start. One VALID chapter is // included on purpose: digestVideo deliberately leaves the sidecar untouched when // a section yields nothing (so the video retries), so an all-bad response would // write no file and there would be no warnings[] on disk to assert against. import { createServer } from "node:http"; import { portFor } from "yt-dlp-transcript-common/lib/ports.mjs"; const PORT = portFor("OLLAMA_STUB_PORT"); const MODEL = process.env.E2E_OLLAMA_STUB_MODEL ?? "qwen2.5:7b"; const SENTINEL = /badout/i; function hms(total) { const n = Math.max(0, Math.floor(total)); return [Math.floor(n / 3600), Math.floor((n % 3600) / 60), n % 60] .map((v) => String(v).padStart(2, "0")) .join(":"); } function toSeconds(clock) { const m = /^(\d\d):(\d\d):(\d\d)$/.exec(clock); if (!m) return null; return Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]); } // The prompt states its own range ("This section covers HH:MM:SS to HH:MM:SS"), // which is the one piece of state the stub needs to answer plausibly. function rangeFromPrompt(prompt) { const m = /This section covers (\d\d:\d\d:\d\d) to (\d\d:\d\d:\d\d)/.exec( prompt ?? "", ); if (!m) return { start: 0, end: 600 }; return { start: toSeconds(m[1]) ?? 0, end: toSeconds(m[2]) ?? 600 }; } // Titles are concrete noun phrases, not "Discussion" — the stub should not be // the thing that fails a generic-title assertion. const TITLES = [ "Court filing deadlines", "Sponsor read and housekeeping", "Audience questions on the ruling", "Closing arguments recap", ]; function goodChapters({ start, end }) { const span = Math.max(1, end - start); const out = []; // Three evenly spaced starts, the last comfortably inside the range so a // rounding difference can never push it past the clamp. for (let i = 0; i < 3; i++) { const at = start + Math.floor((span * i) / 4); out.push({ start: hms(at), title: TITLES[i % TITLES.length] }); } return out; } function badChapters(range) { const [first] = goodChapters(range); return [ // Survives every guard, so the section is written and warnings[] lands on // disk where a spec can read it. first, // GUARD 1 — the exact malformed shape the naive prompt produced. { start: ":00:27", title: "Malformed stamp" }, // GUARD 2 — language drift. { start: hms(range.start + 5), title: "法廷の締め切り" }, // GUARD 3 — an hour past the end of this chunk's range. { start: hms(range.end + 3600), title: "Out of range topic" }, ]; } // --------------------------------------------------------------------------- // Attribution — the two lanes, answered from the schema and the prompt // --------------------------------------------------------------------------- // The names the stub hands out, in cluster-weight order. Real names rather than // "Speaker 1", which the parser rejects on purpose (the number is already known). const SPEAKERS = ["Marla Vance", "Guest", "Caller", "Producer"]; // Diarized lane: name the clusters the schema ENUMERATES. Reading them back out // of the schema is the point — it proves the enum the runner built from the real // diarization.json is what reached the engine, so a spec cannot pass on a // hardcoded cluster the video never had. function namedClusters(speakersSchema, bad) { const enumerated = speakersSchema?.items?.properties?.cluster?.enum ?? [0]; const out = enumerated.map((cluster, i) => ({ cluster, label: SPEAKERS[i % SPEAKERS.length], confidence: i === 0 ? 0.9 : 0.5, })); if (!bad) return out; // BADOUT: one cluster that does not exist, and one useless label. Both must be // dropped by the parser while the good entries survive — the same "one valid // item so the file is still written" shape the chapter bad-output mode uses. return [ ...out, { cluster: 9999, label: "Nobody At All", confidence: 0.9 }, { cluster: enumerated[0] ?? 0, label: "Speaker 2", confidence: 0.9 }, ]; } // Text-only lane: alternate two speakers across the chunk, staying inside the // range the prompt states. // // It REUSES a label it was already given. The prompt carries the roster // established by earlier chunks ("These labels were already used earlier in THIS // video"), and a model that ignores it produces a fresh cast per chunk — which // is precisely the cross-chunk identity failure the lane is judged on. Echoing // the roster is what a good model does, so the stub does it too. function speakerTurns(range, prompt, bad) { const known = [...(prompt ?? "").matchAll(/^ {2}- (.+)$/gm)].map((m) => m[1].trim(), ); const cast = known.length > 0 ? known : SPEAKERS.slice(0, 2); const span = Math.max(1, range.end - range.start); const out = []; for (let i = 0; i < 2; i++) { out.push({ start: hms(range.start + Math.floor((span * i) / 3)), speaker: cast[i % cast.length], }); } if (!bad) return out; return [ ...out, // GUARD 1 — the malformed shape the naive prompt produced. { start: ":00:27", speaker: "Malformed" }, // GUARD 2 — an hour past the end of this chunk's range. { start: hms(range.end + 3600), speaker: "Out Of Range" }, // GUARD 3 — the label that carries no information. { start: hms(range.start + 1), speaker: "Speaker 1" }, ]; } function readBody(req) { return new Promise((resolve, reject) => { let raw = ""; req.on("data", (c) => { raw += c; }); req.on("end", () => resolve(raw)); req.on("error", reject); }); } const server = createServer(async (req, res) => { const url = req.url ?? "/"; // The reachability probe digestBatch runs before touching 74k videos. if (req.method === "GET" && url.startsWith("/api/tags")) { res.writeHead(200, { "content-type": "application/json" }); res.end(JSON.stringify({ models: [{ name: MODEL, model: MODEL }] })); return; } if (req.method === "POST" && url.startsWith("/api/chat")) { const raw = await readBody(req); let body = {}; try { body = JSON.parse(raw); } catch { /* fall through to the default range */ } const messages = Array.isArray(body.messages) ? body.messages : []; const prompt = messages.map((m) => m?.content ?? "").join("\n"); const range = rangeFromPrompt(prompt); const bad = SENTINEL.test(raw); // Answer whichever shape the caller's SCHEMA names, so one stub covers every // workload driven through this engine: digest tags, digest chapters, and the // two attribution lanes. const props = body.format?.properties ?? {}; let data; if (props.tags) { data = { tags: ["court filings", "podcast", "legal news"] }; } else if (props.speakers) { data = { speakers: namedClusters(props.speakers, bad) }; } else if (props.turns) { data = { turns: speakerTurns(range, prompt, bad) }; } else { data = { chapters: bad ? badChapters(range) : goodChapters(range) }; } res.writeHead(200, { "content-type": "application/json" }); res.end( JSON.stringify({ model: body.model || MODEL, message: { role: "assistant", content: JSON.stringify(data) }, done: true, prompt_eval_count: 100, eval_count: 50, // NANOSECONDS, as the real API reports them. Included so the timing // breakdown digestApps.ts records is exercised by e2e rather than only by // a live GPU — a run that silently stopped parsing these would otherwise // look identical to one that worked. total_duration: 1_500_000_000, load_duration: 300_000_000, prompt_eval_duration: 400_000_000, eval_duration: 800_000_000, }), ); return; } res.writeHead(404, { "content-type": "application/json" }); res.end(JSON.stringify({ error: `no stub route for ${req.method} ${url}` })); }); server.listen(PORT, "127.0.0.1", () => { console.log(`ollama stub listening on http://127.0.0.1:${PORT}`); });