import { test } from "node:test"; import assert from "node:assert/strict"; import { Client, InMemoryTransport } from "@modelcontextprotocol/client"; import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { ChannelGroups, ChannelRef, ShardSource, VideoAvailability, } from "./source"; import { SourceRegistry } from "./sourceRegistry"; import { createServer } from "./server"; import type { FetchClipDeps, HttpInit } from "./fetchClip"; // ─── fetch_clip through the real tools/call path ─── // // The unit tests pin the HTTP exchange; these pin what only the server does: // `source` resolves first, the cited id is mapped to the editor's directory id // through the corpus record, and the corpus trailer lands on every answer. const RUMBLE = "the-quartering-rumble"; const YT = "chan-yt"; // A published Rumble record is keyed by yt-dlp's native id — the EMBED id — // while its webpageUrl is the slug URL the editor names the directory for. const RECORDS: Record> = { [RUMBLE]: { id: "vxe1ae", slug: `${RUMBLE}/vxe1ae`, title: "a rumble stream", webpageUrl: "https://rumble.com/v1007ay-x.html?e9s=1", }, [YT]: { id: "dQw4w9WgXcQ", slug: `${YT}/dQw4w9WgXcQ`, title: "a youtube video", webpageUrl: "https://www.youtube.com/watch?v=dQw4w9WgXcQ", }, }; // How many times any source was asked for its channels — the first step of // every corpus lookup. let channelListings = 0; class RecordSource implements ShardSource { readonly label = "local:/srv/fixture"; async loadAliases(): Promise { return []; } async loadGroups(): Promise { return { groups: [], defaultGroupId: "default" }; } async listChannels(): Promise { channelListings++; return Object.keys(RECORDS).map((slug) => ({ key: slug, slug, name: slug })); } async transcriptsManifest(ch: ChannelRef): Promise { const rec = RECORDS[ch.slug]; return { slugToPage: { [String(rec.id)]: 0 } } as unknown as ChannelTranscriptsManifest; } async transcriptPage(ch: ChannelRef): Promise { return [RECORDS[ch.slug] as unknown as TranscriptDetail]; } publicOrigin(): string | null { return null; } async subsManifest(): Promise { return null; } async subsPage(): Promise<[]> { return []; } async postsManifest(): Promise { return null; } async postsPage(): Promise<[]> { return []; } async availabilityMap(): Promise> { return new Map(); } } type Posted = { url: string; method: string; body?: Record }; // A fake editor that answers every POST "cached" (or with `answer`), and a // deps bundle that records what was sent. function fakeEditor( env: Record = { WORKER_TOKEN: "tok" }, answer: (call: Posted) => { status: number; body: unknown } = () => ({ status: 200, body: { cached: true, file: "/corpus/channels/x/data/y/clips/7.00-23.00.mp4", from: 7, to: 23, bytes: 10, provenance: null, }, }), ): { deps: Partial; calls: Posted[] } { const calls: Posted[] = []; let clock = 0; return { calls, deps: { env, fetch: async (url: string, init?: HttpInit) => { const call: Posted = { url, method: init?.method ?? "GET", body: init?.body ? JSON.parse(init.body) : undefined, }; calls.push(call); const a = answer(call); return { status: a.status, json: async () => a.body }; }, sleep: async (ms: number) => { clock += ms; }, now: () => clock, }, }; } async function connect(deps: Partial): Promise { const registry = SourceRegistry.forSource(new RecordSource()); const server = createServer(registry, { fetchClipDeps: deps }); const [ct, st] = InMemoryTransport.createLinkedPair(); const client = new Client({ name: "test", version: "0" }, { capabilities: {} }); await Promise.all([server.connect(st), client.connect(ct)]); return client; } function textOf(res: unknown): string { const content = (res as { content: { type: string; text: string }[] }).content; return content.map((c) => c.text).join("\n"); } const isError = (res: unknown) => (res as { isError?: boolean }).isError === true; test("a Rumble embed id is fetched under the editor's slug id, with the record's URL", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); const res = await client.callTool({ name: "fetch_clip", arguments: { channel: RUMBLE, video: "vxe1ae", start: 10, end: 20, reason: "proof" }, }); assert.equal(isError(res), false, textOf(res)); assert.equal(calls.length, 1); assert.equal(calls[0].body?.videoId, "v1007ay"); assert.equal(calls[0].body?.webpageUrl, "https://rumble.com/v1007ay-x.html?e9s=1"); assert.equal(calls[0].body?.channelSlug, RUMBLE); assert.match(textOf(res), /\n\n\(corpus: local:\/srv\/fixture\)$/); }); test("the channel is the record's own slug, however the citation spelled it", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); await client.callTool({ name: "fetch_clip", arguments: { channel: "The-Quartering-Rumble", video: "vxe1ae", start: 10, end: 20, reason: "proof" }, }); assert.equal(calls[0].body?.channelSlug, RUMBLE); }); test("a YouTube id maps to itself", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); await client.callTool({ name: "fetch_clip", arguments: { channel: YT, video: "dQw4w9WgXcQ", start: "1:00", end: "1:10", reason: "proof" }, }); assert.equal(calls[0].body?.videoId, "dQw4w9WgXcQ"); assert.equal(calls[0].body?.from, 57); assert.equal(calls[0].body?.to, 73); }); test("full: true still maps a Rumble id to the slug, and sends no URL or span", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); await client.callTool({ name: "fetch_clip", arguments: { channel: RUMBLE, video: "vxe1ae", full: true, start: 1, end: 2, reason: "summary" }, }); assert.deepEqual(calls[0].body, { channelSlug: RUMBLE, videoId: "v1007ay", full: true, requestedBy: "mcp", reason: "summary", }); }); test("no editor configured: an error naming both variables, and nothing sent", async () => { const { deps, calls } = fakeEditor({}); const client = await connect(deps); const res = await client.callTool({ name: "fetch_clip", arguments: { channel: YT, video: "dQw4w9WgXcQ", start: 1, end: 2, reason: "proof" }, }); assert.equal(isError(res), true); const t = textOf(res); assert.match(t, /ARCHILYZER_EDITOR_URL/); assert.match(t, /WORKER_TOKEN/); assert.match(t, /\(corpus: local:\/srv\/fixture\)$/); assert.equal(calls.length, 0); }); test("a video the corpus does not hold goes through as cited, with a note", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); const res = await client.callTool({ name: "fetch_clip", arguments: { channel: YT, video: "zzzz", start: 1, end: 2, reason: "proof" }, }); assert.equal(isError(res), false); assert.equal(calls[0].body?.videoId, "zzzz"); assert.equal(calls[0].body?.channelSlug, YT); assert.ok(!("webpageUrl" in (calls[0].body ?? {}))); assert.ok( textOf(res).startsWith( 'note: "zzzz" was not found in corpus local:/srv/fixture; the id was passed to ' + "the editor as-is (for a Rumble citation this may be the embed id — pass the " + "corpus the citation came from as source).\n\nAlready on disk", ), textOf(res), ); }); test("a job alone is a valid call: one poll, no POST, no corpus lookup needed", async () => { const { deps, calls } = fakeEditor({ WORKER_TOKEN: "tok" }, () => ({ status: 200, body: { status: "running", jobId: "j5" }, })); const client = await connect(deps); const before = channelListings; const res = await client.callTool({ name: "fetch_clip", arguments: { job: "j5", wait_seconds: 0 }, }); assert.equal(isError(res), false, textOf(res)); assert.deepEqual(calls.map((c) => c.method), ["GET"]); assert.equal(channelListings, before, "a resume reads nothing from the corpus"); assert.match(textOf(res), /^Still running on the editor \(job j5, waited 0s\)/); }); test("a bad argument is refused before any HTTP", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); const res = await client.callTool({ name: "fetch_clip", arguments: { channel: YT, video: "dQw4w9WgXcQ", start: 10, end: 5, reason: "proof" }, }); assert.equal(isError(res), true); assert.match(textOf(res), /^fetch_clip: start \(10\) must be less than end \(5\)/); assert.equal(calls.length, 0); }); test("the tool is advertised with job-only calls allowed", async () => { const client = await connect(fakeEditor().deps); const { tools } = await client.listTools(); const tool = tools.find((t) => t.name === "fetch_clip"); assert.ok(tool, "fetch_clip is listed"); assert.deepEqual(tool.inputSchema.required ?? [], []); const props = Object.keys(tool.inputSchema.properties ?? {}); for (const p of ["source", "channel", "video", "start", "end", "pad", "full", "maxHeight", "reason", "report", "wait_seconds", "job"]) { assert.ok(props.includes(p), `has ${p}`); } assert.match(tool.description ?? "", /NEVER run yt-dlp/); }); test("a client that asks for progress gets one notification per poll", async () => { let polls = 0; const { deps } = fakeEditor({ WORKER_TOKEN: "tok" }, (call) => { if (call.method === "POST") { return { status: 202, body: { cached: false, jobId: "j6", file: "/f.mp4", from: 7, to: 23 } }; } polls++; return polls < 3 ? { status: 200, body: { status: "running", jobId: "j6" } } : { status: 200, body: { status: "done", jobId: "j6", file: "/f.mp4", from: 7, to: 23, bytes: 3 } }; }); const client = await connect(deps); const got: { progress: number; message?: string }[] = []; const res = await client.callTool( { name: "fetch_clip", arguments: { channel: YT, video: "dQw4w9WgXcQ", start: 10, end: 20, reason: "proof" }, }, { onprogress: (p) => got.push({ progress: p.progress, message: p.message }), resetTimeoutOnProgress: true }, ); assert.equal(isError(res), false, textOf(res)); assert.deepEqual(got, [ { progress: 1, message: "editor job j6: running, 1s waited" }, { progress: 2, message: "editor job j6: running, 2s waited" }, ]); }); test("maxHeight goes to the editor as given, and a bad one is refused before any HTTP", async () => { const { deps, calls } = fakeEditor(); const client = await connect(deps); await client.callTool({ name: "fetch_clip", arguments: { channel: RUMBLE, video: "vxe1ae", full: true, maxHeight: 720, reason: "summary" }, }); assert.equal((calls[0].body as Record).maxHeight, 720); const bad = fakeEditor(); const res = await (await connect(bad.deps)).callTool({ name: "fetch_clip", arguments: { channel: RUMBLE, video: "vxe1ae", start: 1, end: 5, maxHeight: 4320, reason: "why" }, }); assert.equal(isError(res), true); assert.match(textOf(res), /^fetch_clip: maxHeight "4320" must be a whole number of pixels from 144 to 2160/); assert.equal(bad.calls.length, 0); });