import { test } from "node:test"; import assert from "node:assert/strict"; import http from "node:http"; import type { AddressInfo } from "node:net"; import type { Worker } from "../lib/workers"; import { WorkerPool } from "../jobs/workerPool"; import { acquireLlmSlot, clearLlmVerificationCache, freeLlmSlots, } from "./llmWorkers"; // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/llmWorkers.test.ts // // A fake ollama (/api/tags) on a loopback port drives the verification path: // an endpoint serving the model tag is leased, one that lacks it (or is dead) // is DEGRADED and the next candidate tried — one probe per bad endpoint, not a // failure per item. Pools are private instances, never the global singleton. async function withOllamaStub( models: string[], fn: (baseUrl: string) => Promise, ): Promise { const server = http.createServer((req, res) => { res.setHeader("content-type", "application/json"); if (req.url?.startsWith("/api/tags")) { res.end(JSON.stringify({ models: models.map((name) => ({ name })) })); } else { res.statusCode = 404; res.end("{}"); } }); await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); const { port } = server.address() as AddressInfo; try { await fn(`http://127.0.0.1:${port}`); } finally { await new Promise((resolve) => server.close(() => resolve())); } } function llmWorker(id: string, baseUrl: string, priority = 0): Worker { return { id, name: id, kind: "llm", enabled: true, priority, llm: { baseUrl }, }; } const quiet = () => {}; test("acquireLlmSlot leases a verified endpoint and hands back its baseUrl", async () => { clearLlmVerificationCache(); await withOllamaStub(["qwen2.5:7b"], async (baseUrl) => { const pool = new WorkerPool(); pool.reconfigure([llmWorker("mac", baseUrl)], { applyEnabled: true }); const slot = await acquireLlmSlot("digest", "qwen2.5:7b", quiet, pool); assert.ok(slot, "expected a lease on the serving endpoint"); assert.equal(slot!.baseUrl, baseUrl); assert.equal(freeLlmSlots(["digest"], pool), 0); slot!.lease.release(); assert.equal(freeLlmSlots(["digest"], pool), 1); }); }); test("a bare model tag matches the endpoint's :latest", async () => { clearLlmVerificationCache(); await withOllamaStub(["qwen2.5:latest"], async (baseUrl) => { const pool = new WorkerPool(); pool.reconfigure([llmWorker("mac", baseUrl)], { applyEnabled: true }); const slot = await acquireLlmSlot("digest", "qwen2.5", quiet, pool); assert.ok(slot); slot!.lease.release(); }); }); test("an endpoint missing the tag is degraded, and the next candidate serves", async () => { clearLlmVerificationCache(); await withOllamaStub(["llama3:8b"], async (wrongUrl) => { await withOllamaStub(["qwen2.5:7b"], async (rightUrl) => { const pool = new WorkerPool(); pool.reconfigure( [llmWorker("wrong", wrongUrl, 0), llmWorker("right", rightUrl, 1)], { applyEnabled: true }, ); const logs: string[] = []; const slot = await acquireLlmSlot( "digest", "qwen2.5:7b", (m) => logs.push(m), pool, ); assert.ok(slot, "the second endpoint should have been leased"); assert.equal(slot!.workerId, "right"); slot!.lease.release(); // The wrong endpoint is OUT — degraded, one probe, no per-item retries. assert.equal( pool.summary().find((w) => w.id === "wrong")?.degraded, true, ); assert.ok(logs.some((m) => /does not serve/.test(m))); // And a later claim goes straight to the survivor. const again = await acquireLlmSlot("digest", "qwen2.5:7b", quiet, pool); assert.equal(again?.workerId, "right"); again?.lease.release(); }); }); }); test("a dead endpoint is degraded rather than looped on", async () => { clearLlmVerificationCache(); const pool = new WorkerPool(); pool.reconfigure([llmWorker("dead", "http://127.0.0.1:59599")], { applyEnabled: true, }); const slot = await acquireLlmSlot("digest", "qwen2.5:7b", quiet, pool, 500); assert.equal(slot, null); assert.equal(pool.summary().find((w) => w.id === "dead")?.degraded, true); });