// A read-only benchmark for the archilyzer MCP server. // // It drives the REAL server the way a client does — spawning `src/index.ts` // over StdioClientTransport and calling advertised tools — rather than importing // the internals. Anything it can measure is therefore something a real caller // actually pays for. // // Deliberately outside `src/`, so `tsconfig`'s `include` ignores it and it never // joins the test suite (`pnpm test` globs `src/*.test.ts`). // // pnpm --filter yt-dlp-transcript-mcp bench // pnpm --filter yt-dlp-transcript-mcp bench -- --json out.json // pnpm --filter yt-dlp-transcript-mcp bench -- --source local:/some/dir // pnpm --filter yt-dlp-transcript-mcp bench -- --max-load 6 --repeat 3 // // ─── Why there are two kinds of number here ─── // // This box is shared: other agents' jobs run on it, and a `compose:site && // next build` was running during the session that motivated this work (load // average 27). Wall-clock timings taken under that are not measurements, they // are noise with units. // // So the bench reports two families and treats them differently: // // pages / bytes STRUCTURAL. Properties of the query plan — how many shard // pages the server had to open and how many bytes it parsed. // Identical on an idle box and a hammered one, so a // before/after comparison of these is always valid. // wall ms CONTINGENT. Only meaningful below the load threshold, and // the run REFUSES to print it as a headline above that // (see the precondition gate below) — it is marked UNRELIABLE // instead, so a number taken under load cannot later be // quoted as if it weren't. // // Structural counters come from the server itself: MCP_IO_STATS=1 makes it emit // one `[io] {...}` JSON line per tool call on stderr, which this reads. import { Client } from "@modelcontextprotocol/client"; import { StdioClientTransport } from "@modelcontextprotocol/client/stdio"; import { readFile, writeFile } from "node:fs/promises"; import { cpus, loadavg } from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; const HERE = path.dirname(fileURLToPath(import.meta.url)); const MCP_ROOT = path.resolve(HERE, ".."); const REPO_ROOT = path.resolve(MCP_ROOT, ".."); // ─── args ─── type Args = { source?: string; local?: string; json?: string; maxLoad: number; repeat: number; force: boolean; only?: string; timeoutMs: number; }; function parseArgs(argv: string[]): Args { const flag = (name: string): string | undefined => { const i = argv.indexOf(name); return i >= 0 && i + 1 < argv.length ? argv[i + 1] : undefined; }; const num = (name: string, fallback: number): number => { const raw = flag(name); const n = raw === undefined ? NaN : Number(raw); return Number.isFinite(n) ? n : fallback; }; return { source: flag("--source"), local: flag("--local"), json: flag("--json"), // Per-core 1-minute load. 1.0 means "as many runnable tasks as cores", // which is already a contended box for a latency measurement. maxLoad: num("--max-load", 0.7), repeat: Math.max(1, Math.floor(num("--repeat", 3))), force: argv.includes("--force"), only: flag("--only"), // A full-corpus scan of 1.3 GB exceeds the client's 60 s default, which is // the thing being measured — so the bench must not time it out. timeoutMs: num("--timeout", 900_000), }; } // ─── the precondition gate ─── // // This box runs other agents' jobs concurrently. A timing taken while a build // or a GPU sweep is running is not a measurement of this server, and the way // that bites is not at collection time — it is three weeks later, when the // number is quoted from a changelog with no memory of what else was running. // So the gate is not advisory: above the threshold, wall times are stamped // UNRELIABLE in the output itself and in the JSON, permanently. type Preconditions = { cores: number; load1: number; perCore: number; ok: boolean; note: string; }; function checkPreconditions(maxLoad: number): Preconditions { const cores = cpus().length || 1; const load1 = loadavg()[0]; const perCore = load1 / cores; const ok = perCore <= maxLoad; return { cores, load1, perCore, ok, note: ok ? `load ${load1.toFixed(2)} over ${cores} core(s) = ${perCore.toFixed(2)}/core — under the ${maxLoad}/core threshold` : `load ${load1.toFixed(2)} over ${cores} core(s) = ${perCore.toFixed(2)}/core — ABOVE the ${maxLoad}/core threshold`, }; } // ─── the query suite ─── // // Fixed, and each case exists to pin one specific thing. Terms are chosen to be // corpus-agnostic enough to run against any archilyzer site; the fingerprint // printed above the table is what makes two runs comparable, not the terms. type Case = { id: string; what: string; tool: string; args: Record; // Filled at runtime from an earlier case (the id batch needs real ids). needsIds?: "same-page" | "spread"; }; const CASES: Case[] = [ { id: "cold-channels", what: "list_channels (cold: corpus.json + groups)", tool: "list_channels", args: {}, }, { id: "rare", what: "rare term, whole corpus", tool: "search_transcripts", args: { query: "defenestration", limit: 5, content_types: ["video"] }, }, { id: "common", what: "common term, whole corpus", tool: "search_transcripts", args: { query: "lawsuit", limit: 5, content_types: ["video"] }, }, { id: "channel-scoped", what: "common term, one channel", tool: "search_transcripts", args: { query: "lawsuit", limit: 5, content_types: ["video"], channels: ["__FIRST_CHANNEL__"] }, }, { id: "date-scoped", what: "common term + upload-date range (filter-first)", tool: "search_transcripts", args: { query: "lawsuit", limit: 5, content_types: ["video"], date_from: "20240101", date_to: "20241231", }, }, { id: "state-scoped", what: "common term + missing states only (filter-first, the signature question)", tool: "search_transcripts", args: { query: "lawsuit", limit: 5, content_types: ["video"], states: ["deleted", "private", "members_only", "unlisted", "maybe_missing"], }, }, { id: "enumerate", what: "enumerate_matches, whole corpus", tool: "enumerate_matches", args: { query: "lawsuit", content_types: ["video"] }, }, { id: "batch-20", what: "get_transcripts × 20 ids from one channel", tool: "get_transcripts", args: { query: "lawsuit", before: 15, after: 15 }, needsIds: "same-page", }, ]; // ─── stderr [io] line collection ─── type IoLine = { tool: string; ms: number; reads: number; bytes: number }; class IoCollector { private lines: IoLine[] = []; private buffered = ""; feed(chunk: string): void { this.buffered += chunk; const parts = this.buffered.split("\n"); this.buffered = parts.pop() ?? ""; for (const line of parts) { const at = line.indexOf("[io] "); if (at === -1) continue; try { this.lines.push(JSON.parse(line.slice(at + 5)) as IoLine); } catch { // a partial or malformed line is simply not a measurement } } } // Everything since the mark, summed. Tool calls are serial here, so this is // exactly the calls made by the case being timed. drain(): { reads: number; bytes: number; serverMs: number } { const taken = this.lines; this.lines = []; return { reads: taken.reduce((a, l) => a + l.reads, 0), bytes: taken.reduce((a, l) => a + l.bytes, 0), serverMs: taken.reduce((a, l) => a + l.ms, 0), }; } } // ─── corpus fingerprint ─── // // Printed above every table. Two runs are only comparable if these match — the // composed dir gets rebuilt (one rebuild landed mid-session and swapped the // 30,923-video composition for a 3,330-video one), and a before/after that // silently spans two corpora is worse than no measurement at all. type Fingerprint = { handle: string; channels: number; transcriptPages: number; transcriptBytes: number; summariesVideos: number | null; hasStats: boolean; hasDuplicates: boolean; hasDigests: boolean; }; async function fingerprintLocal(dir: string): Promise> { const readJson = async (p: string): Promise => { try { return JSON.parse(await readFile(path.join(dir, p), "utf8")) as T; } catch { return null; } }; const { statSync, readdirSync } = await import("node:fs"); const corpus = await readJson<{ channels?: { slug: string }[] }>("corpus.json"); const summaries = await readJson<{ totalCount?: number }>("summaries/manifest.json"); let pages = 0; let bytes = 0; for (const c of corpus?.channels ?? []) { const chDir = path.join(dir, "transcripts", c.slug); try { for (const f of readdirSync(chDir)) { if (!f.startsWith("page-")) continue; pages++; bytes += statSync(path.join(chDir, f)).size; } } catch { // channel not present in this composition } } const exists = (p: string): boolean => { try { statSync(path.join(dir, p)); return true; } catch { return false; } }; return { channels: corpus?.channels?.length ?? 0, transcriptPages: pages, transcriptBytes: bytes, summariesVideos: summaries?.totalCount ?? null, hasStats: exists("stats/manifest.json"), hasDuplicates: exists("duplicates.json"), hasDigests: exists("digests/manifest.json"), }; } // ─── running ─── type Row = { id: string; what: string; wallMs: number[]; medianMs: number; reads: number; bytes: number; note: string; }; function median(xs: number[]): number { const s = [...xs].sort((a, b) => a - b); const mid = Math.floor(s.length / 2); return s.length % 2 === 1 ? s[mid] : Math.round((s[mid - 1] + s[mid]) / 2); } function mb(bytes: number): string { return bytes === 0 ? "0" : (bytes / 1024 / 1024).toFixed(1); } function firstText(result: unknown): string { const content = (result as { content?: { type: string; text?: string }[] }) .content; return (content ?? []) .filter((c) => c.type === "text") .map((c) => c.text ?? "") .join("\n"); } async function main(): Promise { const args = parseArgs(process.argv.slice(2)); const pre = checkPreconditions(args.maxLoad); console.log("archilyzer MCP benchmark"); console.log(""); console.log(` preconditions: ${pre.note}`); if (!pre.ok && !args.force) { console.log(""); console.log( " REFUSING to report wall-clock timings: this box is shared, and a\n" + " timing taken under this load measures the other jobs, not the server.\n" + " Structural counters (pages read, bytes parsed) are load-independent —\n" + " re-run with --force to collect those anyway, and the wall column will\n" + " be stamped UNRELIABLE.", ); process.exitCode = 2; return; } const wallTrusted = pre.ok; const io = new IoCollector(); // Spawn the server through the SAME command line the MCP client is // registered with (`pnpm --filter … exec tsx src/index.ts --local …`). Not a // detail: `tsx src/index.ts` run directly cannot resolve // `yt-dlp-transcript-common` — the workspace link comes from pnpm — so a // bench that invented its own invocation would be measuring a process the // real client never starts, if it started at all. const corpusDir = args.local ?? path.join(REPO_ROOT, "export", "public"); const transport = new StdioClientTransport({ command: "pnpm", args: [ "-C", REPO_ROOT, "--filter", "yt-dlp-transcript-mcp", "exec", "tsx", "src/index.ts", "--local", corpusDir, ], cwd: REPO_ROOT, stderr: "pipe", env: { ...process.env, MCP_IO_STATS: "1" } as Record, }); transport.stderr?.on("data", (c: Buffer) => io.feed(c.toString("utf8"))); const client = new Client({ name: "mcp-bench", version: "1.0.0" }); await client.connect(transport); // Resolve the corpus + fingerprint it. const sourceArg = args.source ? { source: args.source } : {}; const resolved = firstText( await client.callTool( { name: "resolve_source", arguments: { source: args.source ?? "default" } }, { timeout: args.timeoutMs }, ), ); const handle = /Handle:\s*(\S+)/.exec(resolved)?.[1] ?? "(unknown)"; const fp: Fingerprint = { handle, channels: 0, transcriptPages: 0, transcriptBytes: 0, summariesVideos: null, hasStats: false, hasDuplicates: false, hasDigests: false, ...(handle.startsWith("local:") ? await fingerprintLocal(path.resolve(REPO_ROOT, handle.slice("local:".length))) : {}), }; io.drain(); console.log(""); console.log(" corpus fingerprint (two runs are comparable only if these match):"); console.log(` handle: ${fp.handle}`); console.log(` channels: ${fp.channels}`); console.log( ` transcripts: ${fp.transcriptPages} page(s), ${mb(fp.transcriptBytes)} MB`, ); console.log(` summaries: ${fp.summariesVideos ?? "?"} video(s)`); console.log( ` layers: stats=${fp.hasStats} duplicates=${fp.hasDuplicates} digests=${fp.hasDigests}`, ); // The channel-scoped case needs a real channel name. const channelsText = firstText( await client.callTool( { name: "list_channels", arguments: sourceArg }, { timeout: args.timeoutMs }, ), ); const firstChannel = /slug: ([^,)]+)/.exec(channelsText)?.[1]?.trim(); io.drain(); // The id batch needs real ids: take them from one channel's worklist so they // cluster onto as few shard pages as possible — that IS the case under test. const worklist = firstText( await client.callTool( { name: "enumerate_matches", arguments: { ...sourceArg, query: "lawsuit", content_types: ["video"], ...(firstChannel ? { channels: [firstChannel] } : {}), }, }, { timeout: args.timeoutMs }, ), ); const ids = [...worklist.matchAll(/^- (\S+) \| video \|/gm)] .map((m) => m[1]) .slice(0, 20); io.drain(); const rows: Row[] = []; for (const c of CASES) { if (args.only && !c.id.includes(args.only)) continue; const callArgs: Record = { ...sourceArg, ...c.args }; if (Array.isArray(callArgs.channels)) { callArgs.channels = (callArgs.channels as string[]).map((x) => x === "__FIRST_CHANNEL__" ? (firstChannel ?? "") : x, ); if ((callArgs.channels as string[]).some((x) => x === "")) continue; } if (c.needsIds) { if (ids.length === 0) continue; callArgs.video_ids = ids; if (firstChannel) callArgs.channels = [firstChannel]; } const wall: number[] = []; let reads = 0; let bytes = 0; let note = ""; for (let r = 0; r < args.repeat; r++) { const t0 = performance.now(); const out = await client.callTool( { name: c.tool, arguments: callArgs }, { timeout: args.timeoutMs }, ); wall.push(Math.round(performance.now() - t0)); const stats = io.drain(); // Report the FIRST (cold) repetition's structural cost. Later runs hit // the process-lifetime caches, which is a different question — one the // 'warm' note answers rather than hides. if (r === 0) { reads = stats.reads; bytes = stats.bytes; const text = firstText(out); const m = /scanned (\d+) page\(s\)/.exec(text); if (m) note = `${m[1]} page(s) scanned`; if (/coverage PARTIAL|COVERAGE PARTIAL/.test(text)) note += " · CAPPED"; if (/filter-pruned/.test(text)) note += " · pruned"; } else if (r === 1) { note += note ? `; warm ${stats.reads} read(s)` : `warm ${stats.reads} read(s)`; } } rows.push({ id: c.id, what: c.what, wallMs: wall, medianMs: median(wall), reads, bytes, note, }); console.log( ` ran ${c.id} (${wall.map((w) => `${w}ms`).join(", ")})`, ); } await client.close(); // ─── the table ─── const wallHeader = wallTrusted ? "wall ms" : "wall ms (UNRELIABLE)"; const w = [ Math.max(28, ...rows.map((r) => r.what.length)), Math.max(wallHeader.length, 12), 12, 12, ]; console.log(""); console.log( `| ${"query".padEnd(w[0])} | ${wallHeader.padEnd(w[1])} | ${"reads".padEnd(w[2])} | ${"MB parsed".padEnd(w[3])} |`, ); console.log( `| ${"-".repeat(w[0])} | ${"-".repeat(w[1])} | ${"-".repeat(w[2])} | ${"-".repeat(w[3])} |`, ); for (const r of rows) { console.log( `| ${r.what.padEnd(w[0])} | ${String(r.medianMs).padEnd(w[1])} | ${String(r.reads).padEnd(w[2])} | ${mb(r.bytes).padEnd(w[3])} |`, ); } console.log(""); for (const r of rows) { if (r.note) console.log(` ${r.id}: ${r.note}`); } if (!wallTrusted) { console.log(""); console.log( ` ⚠ wall times above were taken at ${pre.perCore.toFixed(2)} load/core and are NOT\n` + ` a measurement of this server. The reads/MB columns are structural and\n` + ` remain valid. Re-run under ${args.maxLoad}/core for usable timings.`, ); } if (args.json) { await writeFile( args.json, JSON.stringify( { preconditions: pre, wallTrusted, fingerprint: fp, rows }, null, 2, ), "utf8", ); console.log(`\n wrote ${args.json}`); } } main().catch((e: unknown) => { console.error(e); process.exit(1); });