commit d7a2445848574eb70106bbbe3465bfa94b035a71
parent b1f0c1bc1a33e89d0622277033ecc58ab034c2cd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 26 Sep 2026 14:27:09 -0400
mcp: fetch_clip tool — asks the local editor for a cited moment's media; a Rumble embed id maps to the slug id
The schema follows get_video_metadata (required: [] so a job alone is
valid). source resolves first and withCorpus adds the trailer, as for every
tool. No editor configured is said before any corpus read. The cited id is
looked up in source: the canonical id comes from the record's webpageUrl
(extractVideoId, the editor's own directory naming), the webpageUrl rides
along in window mode, and the channel is the record's own slug. A video not
in source goes through as cited, with a note. createServer takes
{fetchClipDeps} for tests. EXPECTED_TOOLS +1; 9 tool-level tests.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 436 insertions(+), 0 deletions(-)
diff --git a/mcp/src/fetchClip.tool.test.ts b/mcp/src/fetchClip.tool.test.ts
@@ -0,0 +1,272 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { Client, InMemoryTransport } from "@modelcontextprotocol/client";
+import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import type {
+ ChannelGroups,
+ ChannelRef,
+ ShardSource,
+ VideoAvailability,
+} from "./source";
+import { SourceRegistry } from "./sourceRegistry";
+import { createServer } from "./server";
+import type { FetchClipDeps, HttpInit } from "./fetchClip";
+
+// ─── fetch_clip through the real tools/call path ───
+//
+// The unit tests pin the HTTP exchange; these pin what only the server does:
+// `source` resolves first, the cited id is mapped to the editor's directory id
+// through the corpus record, and the corpus trailer lands on every answer.
+
+const RUMBLE = "the-quartering-rumble";
+const YT = "chan-yt";
+
+// A published Rumble record is keyed by yt-dlp's native id — the EMBED id —
+// while its webpageUrl is the slug URL the editor names the directory for.
+const RECORDS: Record<string, Record<string, unknown>> = {
+ [RUMBLE]: {
+ id: "vxe1ae",
+ slug: `${RUMBLE}/vxe1ae`,
+ title: "a rumble stream",
+ webpageUrl: "https://rumble.com/v1007ay-x.html?e9s=1",
+ },
+ [YT]: {
+ id: "dQw4w9WgXcQ",
+ slug: `${YT}/dQw4w9WgXcQ`,
+ title: "a youtube video",
+ webpageUrl: "https://www.youtube.com/watch?v=dQw4w9WgXcQ",
+ },
+};
+
+// How many times any source was asked for its channels — the first step of
+// every corpus lookup.
+let channelListings = 0;
+
+class RecordSource implements ShardSource {
+ readonly label = "local:/srv/fixture";
+ async loadAliases(): Promise<SearchAlias[]> {
+ return [];
+ }
+ async loadGroups(): Promise<ChannelGroups> {
+ return { groups: [], defaultGroupId: "default" };
+ }
+ async listChannels(): Promise<ChannelRef[]> {
+ channelListings++;
+ return Object.keys(RECORDS).map((slug) => ({ key: slug, slug, name: slug }));
+ }
+ async transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> {
+ const rec = RECORDS[ch.slug];
+ return { slugToPage: { [String(rec.id)]: 0 } } as unknown as ChannelTranscriptsManifest;
+ }
+ async transcriptPage(ch: ChannelRef): Promise<TranscriptDetail[]> {
+ return [RECORDS[ch.slug] as unknown as TranscriptDetail];
+ }
+ publicOrigin(): string | null {
+ return null;
+ }
+ async subsManifest(): Promise<null> {
+ return null;
+ }
+ async subsPage(): Promise<[]> {
+ return [];
+ }
+ async postsManifest(): Promise<null> {
+ return null;
+ }
+ async postsPage(): Promise<[]> {
+ return [];
+ }
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map();
+ }
+}
+
+type Posted = { url: string; method: string; body?: Record<string, unknown> };
+
+// A fake editor that answers every POST "cached" (or with `answer`), and a
+// deps bundle that records what was sent.
+function fakeEditor(
+ env: Record<string, string | undefined> = { WORKER_TOKEN: "tok" },
+ answer: (call: Posted) => { status: number; body: unknown } = () => ({
+ status: 200,
+ body: {
+ cached: true,
+ file: "/corpus/channels/x/data/y/clips/7.00-23.00.mp4",
+ from: 7,
+ to: 23,
+ bytes: 10,
+ provenance: null,
+ },
+ }),
+): { deps: Partial<FetchClipDeps>; calls: Posted[] } {
+ const calls: Posted[] = [];
+ let clock = 0;
+ return {
+ calls,
+ deps: {
+ env,
+ fetch: async (url: string, init?: HttpInit) => {
+ const call: Posted = {
+ url,
+ method: init?.method ?? "GET",
+ body: init?.body ? JSON.parse(init.body) : undefined,
+ };
+ calls.push(call);
+ const a = answer(call);
+ return { status: a.status, json: async () => a.body };
+ },
+ sleep: async (ms: number) => {
+ clock += ms;
+ },
+ now: () => clock,
+ },
+ };
+}
+
+async function connect(deps: Partial<FetchClipDeps>): Promise<Client> {
+ const registry = SourceRegistry.forSource(new RecordSource());
+ const server = createServer(registry, { fetchClipDeps: deps });
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+
+function textOf(res: unknown): string {
+ const content = (res as { content: { type: string; text: string }[] }).content;
+ return content.map((c) => c.text).join("\n");
+}
+const isError = (res: unknown) => (res as { isError?: boolean }).isError === true;
+
+test("a Rumble embed id is fetched under the editor's slug id, with the record's URL", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ const res = await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: RUMBLE, video: "vxe1ae", start: 10, end: 20, reason: "proof" },
+ });
+ assert.equal(isError(res), false, textOf(res));
+ assert.equal(calls.length, 1);
+ assert.equal(calls[0].body?.videoId, "v1007ay");
+ assert.equal(calls[0].body?.webpageUrl, "https://rumble.com/v1007ay-x.html?e9s=1");
+ assert.equal(calls[0].body?.channelSlug, RUMBLE);
+ assert.match(textOf(res), /\n\n\(corpus: local:\/srv\/fixture\)$/);
+});
+
+test("the channel is the record's own slug, however the citation spelled it", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: "The-Quartering-Rumble", video: "vxe1ae", start: 10, end: 20, reason: "proof" },
+ });
+ assert.equal(calls[0].body?.channelSlug, RUMBLE);
+});
+
+test("a YouTube id maps to itself", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: YT, video: "dQw4w9WgXcQ", start: "1:00", end: "1:10", reason: "proof" },
+ });
+ assert.equal(calls[0].body?.videoId, "dQw4w9WgXcQ");
+ assert.equal(calls[0].body?.from, 57);
+ assert.equal(calls[0].body?.to, 73);
+});
+
+test("full: true still maps a Rumble id to the slug, and sends no URL or span", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: RUMBLE, video: "vxe1ae", full: true, start: 1, end: 2, reason: "summary" },
+ });
+ assert.deepEqual(calls[0].body, {
+ channelSlug: RUMBLE,
+ videoId: "v1007ay",
+ full: true,
+ requestedBy: "mcp",
+ reason: "summary",
+ });
+});
+
+test("no editor configured: an error naming both variables, and nothing sent", async () => {
+ const { deps, calls } = fakeEditor({});
+ const client = await connect(deps);
+ const res = await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: YT, video: "dQw4w9WgXcQ", start: 1, end: 2, reason: "proof" },
+ });
+ assert.equal(isError(res), true);
+ const t = textOf(res);
+ assert.match(t, /ARCHILYZER_EDITOR_URL/);
+ assert.match(t, /WORKER_TOKEN/);
+ assert.match(t, /\(corpus: local:\/srv\/fixture\)$/);
+ assert.equal(calls.length, 0);
+});
+
+test("a video the corpus does not hold goes through as cited, with a note", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ const res = await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: YT, video: "zzzz", start: 1, end: 2, reason: "proof" },
+ });
+ assert.equal(isError(res), false);
+ assert.equal(calls[0].body?.videoId, "zzzz");
+ assert.equal(calls[0].body?.channelSlug, YT);
+ assert.ok(!("webpageUrl" in (calls[0].body ?? {})));
+ assert.ok(
+ textOf(res).startsWith(
+ 'note: "zzzz" was not found in corpus local:/srv/fixture; the id was passed to ' +
+ "the editor as-is (for a Rumble citation this may be the embed id — pass the " +
+ "corpus the citation came from as source).\n\nAlready on disk",
+ ),
+ textOf(res),
+ );
+});
+
+test("a job alone is a valid call: one poll, no POST, no corpus lookup needed", async () => {
+ const { deps, calls } = fakeEditor({ WORKER_TOKEN: "tok" }, () => ({
+ status: 200,
+ body: { status: "running", jobId: "j5" },
+ }));
+ const client = await connect(deps);
+ const before = channelListings;
+ const res = await client.callTool({
+ name: "fetch_clip",
+ arguments: { job: "j5", wait_seconds: 0 },
+ });
+ assert.equal(isError(res), false, textOf(res));
+ assert.deepEqual(calls.map((c) => c.method), ["GET"]);
+ assert.equal(channelListings, before, "a resume reads nothing from the corpus");
+ assert.match(textOf(res), /^Still running on the editor \(job j5, waited 0s\)/);
+});
+
+test("a bad argument is refused before any HTTP", async () => {
+ const { deps, calls } = fakeEditor();
+ const client = await connect(deps);
+ const res = await client.callTool({
+ name: "fetch_clip",
+ arguments: { channel: YT, video: "dQw4w9WgXcQ", start: 10, end: 5, reason: "proof" },
+ });
+ assert.equal(isError(res), true);
+ assert.match(textOf(res), /^fetch_clip: start \(10\) must be less than end \(5\)/);
+ assert.equal(calls.length, 0);
+});
+
+test("the tool is advertised with job-only calls allowed", async () => {
+ const client = await connect(fakeEditor().deps);
+ const { tools } = await client.listTools();
+ const tool = tools.find((t) => t.name === "fetch_clip");
+ assert.ok(tool, "fetch_clip is listed");
+ assert.deepEqual(tool.inputSchema.required ?? [], []);
+ const props = Object.keys(tool.inputSchema.properties ?? {});
+ for (const p of ["source", "channel", "video", "start", "end", "pad", "full", "reason", "report", "wait_seconds", "job"]) {
+ assert.ok(props.includes(p), `has ${p}`);
+ }
+ assert.match(tool.description ?? "", /NEVER run yt-dlp/);
+});
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -40,6 +40,7 @@ const EXPECTED_TOOLS = [
"get_thread",
"get_transcripts",
"get_video_metadata",
+ "fetch_clip",
"list_sources",
"resolve_source",
"open_link",
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -62,6 +62,17 @@ import {
type PlanContext,
} from "./instructions";
import {
+ fetchClip,
+ renderFetchClip,
+ validateFetchClipArgs,
+ editorFromEnv,
+ notFoundNote,
+ isVideoId,
+ NO_EDITOR_TEXT,
+ type FetchClipDeps,
+} from "./fetchClip";
+import { extractVideoId } from "yt-dlp-transcript-common/lib/videoId";
+import {
decodeShareLink,
applyLinkOverrides,
renderQueryTree,
@@ -702,6 +713,89 @@ export const TOOLS: Tool[] = [
},
},
{
+ name: "fetch_clip",
+ description:
+ "Get the media behind a cited moment — by asking the local Archilyzer " +
+ "editor, which fetches it through its own paced, cookie-aware, " +
+ "provenanced job (the per-platform sleeps, the rate-limit cooldown, a " +
+ "note beside the file saying who asked and why). NEVER run yt-dlp " +
+ "yourself instead. Give the citation's channel slug, video id, start " +
+ "and end, and a one-line reason; the window is padded (pad, default 3 " +
+ "s) and may be at most 15 min. full: true fetches the whole recording " +
+ "instead, into the editor's saved-video store. The answer names the " +
+ "file on disk: a read-only corpus artifact to play or copy, never to " +
+ "move, edit or delete. The call waits up to wait_seconds; if the fetch " +
+ "is still running it returns the job id — call again with job to keep " +
+ "waiting. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN in this server's " +
+ "environment; without them it says so and fetches nothing.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ ...SOURCE_ARG,
+ channel: {
+ type: "string",
+ description: "The channel slug, as the citation names it.",
+ },
+ video: {
+ type: "string",
+ description:
+ "The archive's video id, as cited. Pass the corpus the citation " +
+ "came from as `source`, so a Rumble embed id resolves to the " +
+ "editor's directory.",
+ },
+ start: {
+ oneOf: [{ type: "number" }, { type: "string" }],
+ description: "Start of the cited span: seconds, or mm:ss / h:mm:ss.",
+ },
+ end: {
+ oneOf: [{ type: "number" }, { type: "string" }],
+ description: "End of the cited span: seconds, or mm:ss / h:mm:ss.",
+ },
+ pad: {
+ type: "number",
+ description:
+ "Seconds added before start and after end (default 3). The padded " +
+ "window is what is fetched, and it may be at most 900 s.",
+ },
+ full: {
+ type: "boolean",
+ description:
+ "The whole recording instead of a window (default false) — for a " +
+ "video that must be re-cut freely or watched end to end. It lands " +
+ "in the editor's saved-video store, is much larger than a window, " +
+ "and needs a video the editor already knows (its metadata or " +
+ "playlist entry). start, end and pad are ignored.",
+ },
+ reason: {
+ type: "string",
+ description:
+ "Required: one line saying why these seconds are needed. Stored " +
+ "beside the file (at most 400 characters are kept).",
+ },
+ report: {
+ type: "string",
+ description:
+ "Optional: the report or manifest this clip is for, recorded with " +
+ "the file.",
+ },
+ wait_seconds: {
+ type: "number",
+ description:
+ "How long to wait for the fetch before returning its job id " +
+ "(default 90, max 300). Nothing is lost when it runs out.",
+ },
+ job: {
+ type: "string",
+ description:
+ "Resume: the job id an earlier call returned. When set, every " +
+ "other argument is ignored and the call just waits on that job.",
+ },
+ },
+ required: [],
+ additionalProperties: false,
+ },
+ },
+ {
name: "list_sources",
description:
"Show the corpora this server can read: its DEFAULT corpus (the one used " +
@@ -916,6 +1010,65 @@ function withCorpus(result: ToolResult, resolved: ResolvedSource): ToolResult {
return { ...result, content };
}
+// The real world for fetch_clip. `env` is process.env itself, so a token set
+// after startup is still seen.
+const DEFAULT_FETCH_CLIP_DEPS: FetchClipDeps = {
+ env: process.env,
+ fetch: (url, init) => globalThis.fetch(url, init),
+ sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
+ now: () => Date.now(),
+};
+
+// fetch_clip: validate, map the cited id to the editor's directory id, ask the
+// editor, and render its answer. The one tool that causes a write — and the
+// editor makes it, not this process.
+//
+// THE RUMBLE TRAP. A published record's `id` is yt-dlp's native id, which on
+// Rumble is the EMBED id (`vxe1ae`); the editor's directory is the canonical id
+// from the URL slug (`v1007ay`). Passing the embed id would make the editor
+// create a NEW data/vxe1ae/ and fetch rumble.com/vxe1ae. The record's
+// `webpageUrl` is the slug URL, so the canonical id comes from it — the same
+// `extractVideoId` the editor names its directories with. On YouTube the two
+// are equal. A video the corpus does not hold goes through as cited, with a
+// note saying what that risks.
+async function handleFetchClip(
+ source: ShardSource,
+ resolved: ResolvedSource,
+ args: Record<string, unknown>,
+ deps: FetchClipDeps,
+): Promise<ToolResult> {
+ // No editor, no fetch — said first, before a corpus read that could only be
+ // wasted.
+ if (!editorFromEnv(deps.env)) return errorText(NO_EDITOR_TEXT);
+ const v = validateFetchClipArgs(args);
+ if (!v.ok) return errorText(v.error);
+ const request = v.request;
+
+ let note = "";
+ let ctx: { channel?: string; video?: string } = {};
+ if ("target" in request) {
+ const target = request.target;
+ const found = await findVideo(source, target.video, target.channel);
+ if (found) {
+ const webpageUrl = found.record.webpageUrl;
+ const mapped = webpageUrl ? extractVideoId(webpageUrl) : null;
+ target.video = mapped && isVideoId(mapped) ? mapped : target.video;
+ // The channel the record lives in, which is the editor's directory even
+ // when the citation spelled it by name or in another case.
+ target.channel = found.ch.slug;
+ if (target.kind === "window" && webpageUrl) target.webpageUrl = webpageUrl;
+ } else {
+ note = `${notFoundNote(target.video, resolved.handle)}\n\n`;
+ }
+ ctx = { channel: target.channel, video: target.video };
+ }
+
+ const outcome = await fetchClip(request, deps);
+ const rendered = renderFetchClip(outcome, ctx);
+ const body = `${note}${rendered.text}`;
+ return rendered.isError ? errorText(body) : text(body);
+}
+
// Build a configured MCP server over a data source or a SourceRegistry. The
// core read tools work for local / remote / hub sources — only the ShardSource
// differs — and which one a call reads is decided per call by its `source`
@@ -923,7 +1076,15 @@ function withCorpus(result: ToolResult, resolved: ResolvedSource): ToolResult {
// one-source registry, so `createServer(someSource)` still works.
export function createServer(
sourceOrRegistry: ShardSource | SourceRegistry,
+ opts: { fetchClipDeps?: Partial<FetchClipDeps> } = {},
): Server {
+ // fetch_clip's world — env, HTTP, the clock — injected so a test drives it
+ // with no network and no timers. Read per call, never cached: the env is the
+ // live process.env unless a test says otherwise.
+ const fetchClipDeps: FetchClipDeps = {
+ ...DEFAULT_FETCH_CLIP_DEPS,
+ ...opts.fetchClipDeps,
+ };
const registry =
sourceOrRegistry instanceof SourceRegistry
? sourceOrRegistry
@@ -987,6 +1148,8 @@ export function createServer(
return handleGetThread(source, args);
case "get_video_metadata":
return handleGetMetadata(source, args);
+ case "fetch_clip":
+ return handleFetchClip(source, resolved, args, fetchClipDeps);
case "list_sources":
return handleListSources(registry, resolved);
case "resolve_source":