Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b1f0c1bc1a33e89d0622277033ecc58ab034c2cd
parent 1e7f5c9847a7c190f28b7f2897d8fa0a01a9fbbc
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 26 Sep 2026 14:27:09 -0400

mcp: fetchClip — the editor's fetch-window client, pure with injected env/fetch/sleep/now

parseSeconds (seconds, mm:ss, h:mm:ss), planWindow (umtool's padding and
two-decimal rounding, the 900 s cap after padding), argument validation (a
job alone is a whole request; full: true ignores start/end/pad), then POST
/api/media/fetch-window and a 1 s poll until terminal or the wait runs out.
renderFetchClip gives the plan's texts. A 503 is a token problem only when
it says the endpoint is disabled: the same status carries an unreachable-
media refusal, passed through verbatim. HTTP only; the process writes
nothing. 30 unit tests.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Amcp/src/fetchClip.test.ts | 570+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Amcp/src/fetchClip.ts | 631+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 1201 insertions(+), 0 deletions(-)

diff --git a/mcp/src/fetchClip.test.ts b/mcp/src/fetchClip.test.ts @@ -0,0 +1,570 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + fetchClip, + parseSeconds, + planWindow, + renderFetchClip, + validateFetchClipArgs, + type FetchClipDeps, + type FetchClipRequest, + type HttpInit, +} from "./fetchClip"; + +// ─── A scripted editor: no network, no timers ─── +// +// `respond` answers each request in turn; `sleep` records and advances the +// fake clock by exactly what it was asked to wait, so a 90 s wait costs no real +// time and the poll count is exact. + +type Answer = { status: number; body: unknown } | Error; +type Call = { url: string; method: string; headers: Record<string, string>; body?: unknown }; + +function editor( + respond: (call: Call, n: number) => Answer, + env: Record<string, string | undefined> = { + WORKER_TOKEN: "tok", + ARCHILYZER_EDITOR_URL: "http://editor.test/", + }, +) { + const calls: Call[] = []; + const sleeps: number[] = []; + let clock = 1_000_000; + const deps: FetchClipDeps = { + env, + fetch: async (url: string, init?: HttpInit) => { + const call: Call = { + url, + method: init?.method ?? "GET", + headers: init?.headers ?? {}, + body: init?.body !== undefined ? JSON.parse(init.body) : undefined, + }; + calls.push(call); + const a = respond(call, calls.length - 1); + if (a instanceof Error) throw a; + return { status: a.status, json: async () => a.body }; + }, + sleep: async (ms: number) => { + sleeps.push(ms); + clock += ms; + }, + now: () => clock, + }; + return { deps, calls, sleeps }; +} + +// Answers in order; running off the end is a test bug, said loudly. +function seq(...answers: Answer[]) { + return (call: Call, n: number): Answer => { + const a = answers[n]; + if (!a) throw new Error(`unscripted request #${n}: ${call.method} ${call.url}`); + return a; + }; +} + +const CLIP_FILE = + "/corpus/channels/chan/data/vid/clips/7.00-23.00.mp4"; + +function windowRequest(extra: Partial<Record<string, unknown>> = {}): FetchClipRequest { + const v = validateFetchClipArgs({ + channel: "chan", + video: "vid", + start: 10, + end: 20, + reason: "the quote in the report", + report: "rep-1", + ...extra, + }); + assert.ok(v.ok, v.ok ? "" : v.error); + return v.request; +} + +function fullRequest(extra: Partial<Record<string, unknown>> = {}): FetchClipRequest { + const v = validateFetchClipArgs({ + channel: "chan", + video: "vid", + full: true, + reason: "summarise the whole stream", + report: "rep-1", + ...extra, + }); + assert.ok(v.ok, v.ok ? "" : v.error); + return v.request; +} + +const CTX = { channel: "chan", video: "vid" }; + +// ─── Times and windows ─── + +test("parseSeconds reads seconds, mm:ss and h:mm:ss", () => { + assert.equal(parseSeconds("1:02:03"), 3723); + assert.equal(parseSeconds("1:02"), 62); + assert.equal(parseSeconds("75:30"), 4530); + assert.equal(parseSeconds("0:07.5"), 7.5); + assert.equal(parseSeconds("12"), 12); + assert.equal(parseSeconds(" 12.25 "), 12.25); + assert.equal(parseSeconds(12.5), 12.5); + assert.equal(parseSeconds(0), 0); + for (const bad of ["abc", "1:60", "1:2:3:4", "1:60:00", "-3", "", -3, Number.NaN, null, undefined, {}]) { + assert.equal(parseSeconds(bad), null, `${String(bad)} is not a time`); + } +}); + +test("planWindow pads, rounds to the editor's two decimals, and clamps at 0", () => { + assert.deepEqual(planWindow({ video: "vid", start: 10, end: 20, pad: 3 }), { from: 7, to: 23 }); + // 7.004 → "7.00", 23.006 → "23.01": the name IS the window. + assert.deepEqual( + planWindow({ video: "vid", start: 10.004, end: 20.006, pad: 3 }), + { from: 7, to: 23.01 }, + ); + assert.deepEqual(planWindow({ video: "vid", start: 1, end: 5, pad: 3 }), { from: 0, to: 8 }); + assert.deepEqual(planWindow({ video: "vid", start: 1, end: 5, pad: 0 }), { from: 1, to: 5 }); +}); + +test("planWindow caps the window at 900 s AFTER padding", () => { + // 3 → 897 padded is exactly 0 → 900: allowed. + assert.deepEqual(planWindow({ video: "v", start: 3, end: 897, pad: 3 }), { from: 0, to: 900 }); + // 895 s of citation is 901 s once padded: refused, with the padded numbers. + assert.deepEqual(planWindow({ video: "v", start: 10, end: 905, pad: 3 }), { + error: + "fetch_clip: the window 7.00–908.00 is 901s; the editor fetches at most " + + "900s per window — cite a narrower span", + }); +}); + +test("planWindow refuses a bad id and an inverted span, in the plan's words", () => { + for (const video of ["..", ".", "a/b", "a b", ""]) { + assert.deepEqual(planWindow({ video, start: 1, end: 2, pad: 3 }), { + error: `fetch_clip: video "${video}" must match /^[\\w.-]+$/`, + }); + } + assert.deepEqual(planWindow({ video: "v", start: 20, end: 10, pad: 3 }), { + error: "fetch_clip: start (20) must be less than end (10)", + }); + assert.deepEqual(planWindow({ video: "v", start: 10, end: 10, pad: 3 }), { + error: "fetch_clip: start (10) must be less than end (10)", + }); +}); + +// ─── Arguments ─── + +test("argument errors say what to fix, before any HTTP", () => { + const err = (args: Record<string, unknown>) => { + const v = validateFetchClipArgs(args); + assert.equal(v.ok, false); + return v.ok ? "" : v.error; + }; + const base = { channel: "chan", video: "vid", start: 10, end: 20, reason: "why" }; + assert.equal( + err({ ...base, channel: undefined }), + "fetch_clip: channel is required (the channel slug)", + ); + assert.equal( + err({ ...base, video: " " }), + "fetch_clip: video is required (the archive's video id)", + ); + assert.equal(err({ ...base, video: "../x" }), 'fetch_clip: video "../x" must match /^[\\w.-]+$/'); + assert.equal( + err({ ...base, start: "soon" }), + 'fetch_clip: start "soon" is not a time (use seconds, mm:ss or h:mm:ss)', + ); + assert.equal( + err({ ...base, end: "1:99" }), + 'fetch_clip: end "1:99" is not a time (use seconds, mm:ss or h:mm:ss)', + ); + assert.equal(err({ ...base, start: "0:20", end: "0:10" }), "fetch_clip: start (20) must be less than end (10)"); + assert.equal( + err({ ...base, reason: " " }), + "fetch_clip: reason is required — one line saying why these seconds are " + + "needed (it is stored beside the file)", + ); + assert.match(err({ ...base, pad: -1 }), /^fetch_clip: pad "-1" must be a finite number/); + assert.match(err({ ...base, pad: "3" }), /^fetch_clip: pad "3" must be a finite number/); + assert.match(err({ ...base, start: undefined }), /^fetch_clip: start is required/); +}); + +test("wait_seconds is clamped to [0, 300] and defaults to 90", () => { + const wait = (w: unknown) => { + const v = validateFetchClipArgs({ job: "j1", wait_seconds: w }); + assert.ok(v.ok); + return v.request.waitSeconds; + }; + assert.equal(wait(undefined), 90); + assert.equal(wait(500), 300); + assert.equal(wait(-5), 0); + assert.equal(wait(12), 12); + assert.equal(wait("30"), 90); +}); + +test("a job alone is a whole request; everything else is then ignored", () => { + const v = validateFetchClipArgs({ job: " j7 ", video: "../bad", start: "nope" }); + assert.deepEqual(v, { ok: true, request: { job: "j7", waitSeconds: 90 } }); +}); + +test("full mode ignores start/end/pad entirely — not required, not validated", () => { + const v = validateFetchClipArgs({ + channel: "chan", + video: "vid", + full: true, + start: "not a time", + end: -4, + pad: -1, + reason: "why", + }); + assert.ok(v.ok, v.ok ? "" : v.error); + assert.deepEqual(v.request, { + target: { kind: "full", channel: "chan", video: "vid" }, + reason: "why", + report: undefined, + waitSeconds: 90, + }); + // …but a reason is still required. + const noReason = validateFetchClipArgs({ channel: "chan", video: "vid", full: true }); + assert.equal(noReason.ok, false); +}); + +// ─── The POST ─── + +test("the POST carries requestedBy mcp, the report as manifest, and a reason cut to 400", async () => { + const { deps, calls } = editor( + seq({ status: 200, body: { cached: true, file: CLIP_FILE, from: 7, to: 23, bytes: 10, provenance: null } }), + ); + const long = "x".repeat(500); + const req = windowRequest({ reason: long }); + if ("target" in req && req.target.kind === "window") { + req.target.webpageUrl = "https://www.youtube.com/watch?v=vid"; + } + await fetchClip(req, deps); + assert.equal(calls.length, 1); + assert.equal(calls[0].method, "POST"); + // The trailing slash on ARCHILYZER_EDITOR_URL does not double up. + assert.equal(calls[0].url, "http://editor.test/api/media/fetch-window"); + assert.equal(calls[0].headers.authorization, "Bearer tok"); + assert.deepEqual(calls[0].body, { + channelSlug: "chan", + videoId: "vid", + webpageUrl: "https://www.youtube.com/watch?v=vid", + from: 7, + to: 23, + pad: 3, + requestedBy: "mcp", + manifest: "rep-1", + reason: "x".repeat(400), + }); +}); + +test("no editor configured is an outcome, and nothing is sent", async () => { + for (const env of [{}, { ARCHILYZER_EDITOR_URL: "http://localhost:3001" }, { WORKER_TOKEN: " " }]) { + const { deps, calls } = editor(seq(), env); + const outcome = await fetchClip(windowRequest(), deps); + assert.deepEqual(outcome, { kind: "no_editor" }); + assert.equal(calls.length, 0); + const r = renderFetchClip(outcome); + assert.equal(r.isError, true); + assert.match(r.text, /^fetch_clip: no editor configured\./); + assert.match(r.text, /ARCHILYZER_EDITOR_URL/); + assert.match(r.text, /WORKER_TOKEN/); + assert.match(r.text, /no-editor fallback/); + } +}); + +test("the editor URL defaults to localhost:3001", async () => { + const { deps, calls } = editor( + seq({ status: 200, body: { cached: true, file: CLIP_FILE, from: 7, to: 23, bytes: 1 } }), + { WORKER_TOKEN: "tok" }, + ); + await fetchClip(windowRequest(), deps); + assert.equal(calls[0].url, "http://localhost:3001/api/media/fetch-window"); +}); + +// ─── Answers ─── + +test("200 cached: a WIDER window is named as such, with its own span", async () => { + const wide = "/corpus/channels/chan/data/vid/clips/0.00-60.00.mp4"; + const { deps } = editor( + seq({ + status: 200, + body: { cached: true, file: wide, from: 0, to: 60, bytes: 123456, provenance: { requestedBy: "umtool" } }, + }), + ); + const outcome = await fetchClip(windowRequest(), deps); + const r = renderFetchClip(outcome, CTX); + assert.equal(r.isError, false); + assert.equal( + r.text, + [ + "Already on disk — a WIDER cached window that contains 7.00–23.00: 0.00–60.00.", + "", + `file: ${wide}`, + "window: 0.00–60.00 (60s)", + "bytes: 123456", + "requested by umtool", + "This path is a read-only corpus artifact: play or copy it, never move, " + + "edit or delete it. The editor prunes clips by age (evict-clips); " + + "provenance sits beside it as 0.00-60.00.json.", + ].join("\n"), + ); +}); + +test("200 cached: the exact window (within 0.02 s) says so", async () => { + const { deps } = editor( + seq({ status: 200, body: { cached: true, file: CLIP_FILE, from: 7.01, to: 23, bytes: 5, provenance: null } }), + ); + const r = renderFetchClip(await fetchClip(windowRequest(), deps), CTX); + assert.match(r.text, /^Already on disk — the exact window: 7\.01–23\.00\./); + assert.ok(!/requested by/.test(r.text)); +}); + +test("202 → running → done: the poll's file, span and bytes", async () => { + const { deps, calls, sleeps } = editor( + seq( + { status: 202, body: { cached: false, jobId: "j1", file: CLIP_FILE, from: 7, to: 23 } }, + { status: 200, body: { status: "running", jobId: "j1" } }, + { status: 200, body: { status: "done", jobId: "j1", file: CLIP_FILE, from: 7, to: 23, bytes: 4096 } }, + ), + ); + const outcome = await fetchClip(windowRequest(), deps); + assert.equal(outcome.kind, "fetched"); + assert.deepEqual(calls.map((c) => c.method), ["POST", "GET", "GET"]); + assert.equal(calls[1].url, "http://editor.test/api/media/fetch-window/j1"); + assert.equal(calls[1].headers.authorization, "Bearer tok"); + assert.deepEqual(sleeps, [1000, 1000]); + const r = renderFetchClip(outcome, CTX); + assert.equal(r.isError, false); + assert.match(r.text, /^Fetched 7\.00–23\.00 of chan\/vid \(job j1, 2s waited\)\.\n\n/); + assert.match(r.text, new RegExp(`\\nfile: ${CLIP_FILE.replace(/\./g, "\\.")}\\n`)); + assert.match(r.text, /\nwindow: 7\.00–23\.00 \(16s\)\n/); + assert.match(r.text, /\nbytes: 4096\n/); + assert.match(r.text, /provenance sits beside it as 7\.00-23\.00\.json\.$/); +}); + +test("202 → failed: the job's log tail comes back", async () => { + const { deps } = editor( + seq( + { status: 202, body: { cached: false, jobId: "j1", file: CLIP_FILE, from: 7, to: 23 } }, + { status: 200, body: { status: "failed", jobId: "j1", error: "ERROR: [youtube] vid: Video unavailable" } }, + ), + ); + const r = renderFetchClip(await fetchClip(windowRequest(), deps), CTX); + assert.equal(r.isError, true); + assert.equal(r.text, "Editor job j1 failed.\n\nlog tail:\nERROR: [youtube] vid: Video unavailable"); +}); + +test("the wait runs out → queued with the job id, without real timers", async () => { + const { deps, calls, sleeps } = editor((call, n) => + n === 0 + ? { status: 202, body: { cached: false, jobId: "j1", file: CLIP_FILE, from: 7, to: 23 } } + : { status: 200, body: { status: "running", jobId: "j1" } }, + ); + const outcome = await fetchClip(windowRequest({ wait_seconds: 3 }), deps); + assert.deepEqual(outcome, { kind: "queued", jobId: "j1", status: "running", waited: 3 }); + assert.deepEqual(sleeps, [1000, 1000, 1000]); + assert.equal(calls.length, 4); + const r = renderFetchClip(outcome, CTX); + assert.equal(r.isError, false, "a wait that runs out is not a failure"); + assert.equal( + r.text, + 'Still running on the editor (job j1, waited 3s). Call fetch_clip again with job: "j1" ' + + "to keep waiting — nothing is lost, the fetch continues on the editor and the next " + + "ask finds it cached.", + ); +}); + +test("wait_seconds 0 returns the job at once, still queued", async () => { + const { deps, calls, sleeps } = editor( + seq({ status: 202, body: { cached: false, jobId: "j1", file: CLIP_FILE, from: 7, to: 23 } }), + ); + const outcome = await fetchClip(windowRequest({ wait_seconds: 0 }), deps); + assert.deepEqual(outcome, { kind: "queued", jobId: "j1", status: "queued", waited: 0 }); + assert.equal(calls.length, 1); + assert.deepEqual(sleeps, []); +}); + +test("resume by job issues no POST, and polls before it sleeps", async () => { + const { deps, calls, sleeps } = editor( + seq({ status: 200, body: { status: "done", jobId: "j9", file: CLIP_FILE, from: 7, to: 23, bytes: 77 } }), + ); + const outcome = await fetchClip({ job: "j9", waitSeconds: 90 }, deps); + assert.deepEqual(calls.map((c) => `${c.method} ${c.url}`), [ + "GET http://editor.test/api/media/fetch-window/j9", + ]); + assert.deepEqual(sleeps, []); + // A resume knows no channel/video; the window comes from the poll. + const r = renderFetchClip(outcome); + assert.match(r.text, /^Fetched 7\.00–23\.00 \(job j9, 0s waited\)\.\n\n/); + assert.match(r.text, /\nwindow: 7\.00–23\.00 \(16s\)\n/); +}); + +test("409: the platform and the seconds left", async () => { + const { deps } = editor( + seq({ + status: 409, + body: { error: "youtube is in a rate-limit cooldown (13s remaining).", cooldownMs: 12_500, platform: "youtube" }, + }), + ); + const r = renderFetchClip(await fetchClip(windowRequest(), deps), CTX); + assert.equal(r.isError, true); + assert.equal( + r.text, + "The editor is in a youtube rate-limit cooldown — 13s remaining. Wait, then call " + + "fetch_clip again. (youtube is in a rate-limit cooldown (13s remaining).)", + ); +}); + +test("401 and a disabled 503 are token problems; 503 disabled says the endpoint is off", async () => { + const bad = editor(seq({ status: 401, body: { error: "invalid worker token" } })); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), bad.deps), CTX), { + text: + "The editor refused the token (HTTP 401): invalid worker token. WORKER_TOKEN must " + + "equal the value the editor runs with (editor/.env).", + isError: true, + }); + const off = editor( + seq({ status: 503, body: { error: "worker endpoint disabled (set WORKER_TOKEN to enable)" } }), + ); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), off.deps), CTX), { + text: + "The editor refused the token (HTTP 503): worker endpoint disabled (set WORKER_TOKEN " + + "to enable). WORKER_TOKEN must equal the value the editor runs with (editor/.env). " + + "The editor has no WORKER_TOKEN set, so its fetch endpoint is off.", + isError: true, + }); +}); + +test("a 503 that is not the token (an unmounted drive) passes through verbatim", async () => { + const drive = "Channel media for chan is unreachable: /mnt/platter is not mounted"; + const { deps } = editor(seq({ status: 503, body: { error: drive } })); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), deps), CTX), { + text: `The editor refused (HTTP 503): ${drive}`, + isError: true, + }); +}); + +test("any other refusal is passed through verbatim", async () => { + const { deps } = editor(seq({ status: 507, body: { error: "Low disk space: 2.1 GB free" } })); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), deps), CTX), { + text: "The editor refused (HTTP 507): Low disk space: 2.1 GB free", + isError: true, + }); +}); + +test("an unreachable editor is named, with its URL", async () => { + const { deps } = editor(seq(new Error("connect ECONNREFUSED 127.0.0.1:3001"))); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), deps), CTX), { + text: "Could not reach the editor at http://editor.test: connect ECONNREFUSED 127.0.0.1:3001. Is it running?", + isError: true, + }); +}); + +test("a poll the editor does not know says it may have restarted", async () => { + const { deps } = editor(seq({ status: 404, body: { error: "no such job" } })); + assert.deepEqual(renderFetchClip(await fetchClip({ job: "j1", waitSeconds: 5 }, deps)), { + text: + "Polling editor job j1 failed (HTTP 404): no such job; the job is unknown to this " + + "editor (restarted? wrong ARCHILYZER_EDITOR_URL?)", + isError: true, + }); +}); + +// ─── full: true — the whole recording ─── + +test("full: the POST is exactly {channelSlug, videoId, full, requestedBy, manifest, reason}", async () => { + // start/end/pad given anyway: they are ignored, not sent. + for (const req of [fullRequest(), fullRequest({ start: 10, end: 20, pad: 5 })]) { + const { deps, calls } = editor( + seq({ status: 202, body: { cached: false, jobId: "j2", file: null, from: 0, to: 0 } }), + ); + await fetchClip({ ...req, waitSeconds: 0 } as FetchClipRequest, deps); + assert.deepEqual(calls[0].body, { + channelSlug: "chan", + videoId: "vid", + full: true, + requestedBy: "mcp", + manifest: "rep-1", + reason: "summarise the whole stream", + }); + } +}); + +test("full: 200 cached names the saved-video store, with no window line", async () => { + const saved = "/corpus/saved-videos/chan/vid/source.mkv"; + const { deps } = editor( + seq({ + status: 200, + body: { cached: true, file: saved, from: 0, to: 0, bytes: 9_000_000, provenance: { requestedBy: "umtool" } }, + }), + ); + const r = renderFetchClip(await fetchClip(fullRequest(), deps), CTX); + assert.equal(r.isError, false); + assert.equal( + r.text, + [ + "Already on disk — the whole recording of chan/vid.", + "", + `file: ${saved}`, + "bytes: 9000000", + "requested by umtool", + "This path is a read-only corpus artifact: play or copy it, never move, edit " + + "or delete it. It lives in the editor's saved-video store, not in clips/; " + + "the editor's keep-videos rule decides how long it stays.", + ].join("\n"), + ); +}); + +test("full: 202 → done carries the file and bytes, and no window line", async () => { + const saved = "/corpus/saved-videos/chan/vid/source.mkv"; + const { deps } = editor( + seq( + { status: 202, body: { cached: false, jobId: "j2", file: null, from: 0, to: 0 } }, + { status: 200, body: { status: "done", jobId: "j2", file: saved, bytes: 42 } }, + ), + ); + const r = renderFetchClip(await fetchClip(fullRequest(), deps), CTX); + assert.equal(r.isError, false); + assert.match(r.text, /^Fetched the whole recording of chan\/vid \(job j2, 1s waited\)\.\n\n/); + assert.match(r.text, /\nbytes: 42\n/); + assert.ok(!/window:/.test(r.text), r.text); + assert.match(r.text, /saved-video store, not in clips\//); +}); + +test("full: a done job that names no file is an error", async () => { + const { deps } = editor( + seq( + { status: 202, body: { cached: false, jobId: "j2", file: null, from: 0, to: 0 } }, + { status: 200, body: { status: "done", jobId: "j2" } }, + ), + ); + assert.deepEqual(renderFetchClip(await fetchClip(fullRequest(), deps), CTX), { + text: "Editor job j2 finished but named no file; check the video's page in the editor.", + isError: true, + }); +}); + +test("full: a 404 says a whole recording needs a video the editor knows; a window's does not", async () => { + const unknown = + "Could not determine the video URL: no metadata.info.json and the playlist does " + + "not contain a matching entry."; + const full = editor(seq({ status: 404, body: { error: unknown } })); + assert.deepEqual(renderFetchClip(await fetchClip(fullRequest(), full.deps), CTX), { + text: + `The editor refused (HTTP 404): ${unknown} A whole-recording fetch needs a video ` + + "the editor already knows (a metadata.info.json or a playlist entry); fetch a " + + "window instead, or add the video to the channel first.", + isError: true, + }); + const win = editor(seq({ status: 404, body: { error: "Channel not found" } })); + assert.deepEqual(renderFetchClip(await fetchClip(windowRequest(), win.deps), CTX), { + text: "The editor refused (HTTP 404): Channel not found", + isError: true, + }); +}); + +test("a resumed whole-recording job renders as one (no from/to in its answer)", async () => { + const saved = "/corpus/saved-videos/chan/vid/source.mkv"; + const { deps } = editor( + seq({ status: 200, body: { status: "done", jobId: "j2", file: saved, bytes: 42 } }), + ); + const r = renderFetchClip(await fetchClip({ job: "j2", waitSeconds: 90 }, deps)); + assert.match(r.text, /^Fetched the whole recording \(job j2, 0s waited\)\./); + assert.ok(!/window:/.test(r.text)); +}); diff --git a/mcp/src/fetchClip.ts b/mcp/src/fetchClip.ts @@ -0,0 +1,631 @@ +import { MAX_CLIP_WINDOW_SECONDS } from "yt-dlp-transcript-common/lib/clipWindow"; + +// ─── fetch_clip: ask the local Archilyzer editor for a clip's media ─── +// +// The operator's rule is that no fetch happens by hand: every byte of media +// goes through the editor, which has the cookie policy, the per-platform +// sleeps, the 429 cooldown and the provenance note. umtool already obeys it +// (umtool/report-to-video/fetch-via-editor.mjs); this is the same client for +// the MCP, so an agent that needs the seconds behind a citation asks the +// editor instead of shelling out to yt-dlp. +// +// THE MCP PROCESS STILL WRITES NOTHING. Everything here is HTTP through +// `deps.fetch`; the editor decides where the bytes go and says so. The file it +// names is a corpus artifact the agent may read, never one it may move. +// +// Pure apart from the injected deps (env, fetch, sleep, now), so the tests need +// no network, no timers and no editor. No MCP imports: server.ts adapts. + +export type HttpInit = { + method?: string; + headers?: Record<string, string>; + body?: string; +}; +export type HttpResponse = { status: number; json(): Promise<unknown> }; +export type FetchLike = (url: string, init?: HttpInit) => Promise<HttpResponse>; + +export type FetchClipDeps = { + env: Record<string, string | undefined>; + fetch: FetchLike; + sleep: (ms: number) => Promise<void>; + now: () => number; +}; + +// The client family's names and defaults (fetch-via-editor.mjs, +// scripts/archilyzer-ops.mjs): one editor URL, one shared token. +export const DEFAULT_EDITOR_URL = "http://localhost:3001"; +export const POLL_MS = 1000; +export const DEFAULT_PAD_SECONDS = 3; +export const DEFAULT_WAIT_SECONDS = 90; +export const MAX_WAIT_SECONDS = 300; +export const MAX_REASON_CHARS = 400; +// Who asked, as the editor records it beside the file. +export const REQUESTED_BY = "mcp"; +// A read-back window can sit a hair off the request (2 dp names); the same +// tolerance as common/lib/clipWindow.ts WIN_EPS. +const SAME_WINDOW_EPS = 0.02; + +export const NO_EDITOR_TEXT = + "fetch_clip: no editor configured. Set ARCHILYZER_EDITOR_URL (e.g. " + + "http://localhost:3001) and WORKER_TOKEN (the editor's own WORKER_TOKEN) " + + "when registering the MCP server. A public-only setup — no local Archilyzer " + + "editor — cannot fetch media through Archilyzer; see README \"Clips and " + + "video\" for the no-editor fallback."; + +// The editor's own id grammar (editor/app/api/media/fetch-window/route.ts): +// anchored, and `.` / `..` refused separately because the class allows a dot. +const ID_RE = /^[\w.-]+$/; +export function isVideoId(v: string): boolean { + return ID_RE.test(v) && v !== "." && v !== ".."; +} + +export type EditorConfig = { url: string; token: string }; + +// The editor to ask, or null when this MCP was registered without a token — +// which is every public-only setup, and is not an error until a fetch is asked. +export function editorFromEnv( + env: Record<string, string | undefined>, +): EditorConfig | null { + const token = (env.WORKER_TOKEN ?? "").trim(); + if (!token) return null; + const raw = (env.ARCHILYZER_EDITOR_URL ?? "").trim() || DEFAULT_EDITOR_URL; + return { url: raw.replace(/\/+$/, ""), token }; +} + +// A citation time: a number of seconds, or "ss", "mm:ss", "h:mm:ss" (a +// fraction allowed on the seconds). Minutes may run past 59 in the two-part +// form, because "75:30" is how a long video's moment is often written. +export function parseSeconds(v: unknown): number | null { + if (typeof v === "number") return Number.isFinite(v) && v >= 0 ? v : null; + if (typeof v !== "string") return null; + const s = v.trim(); + let m = /^(\d+(?:\.\d+)?)$/.exec(s); + if (m) return Number(m[1]); + m = /^(\d+):([0-5]?\d(?:\.\d+)?)$/.exec(s); + if (m) return Number(m[1]) * 60 + Number(m[2]); + m = /^(\d+):([0-5]?\d):([0-5]?\d(?:\.\d+)?)$/.exec(s); + if (m) return Number(m[1]) * 3600 + Number(m[2]) * 60 + Number(m[3]); + return null; +} + +// Seconds as the editor names a window: two decimals, always. +export function fmtSeconds(n: number): string { + return n.toFixed(2); +} +function fmtSpan(n: number): string { + return `${Number(n.toFixed(2))}s`; +} + +// The padded window, exactly as umtool's client computes it: the name IS the +// window, so a request that rounds differently addresses a different file and +// the cache misses forever. The 900 s cap applies AFTER padding — it is the +// span the editor is asked for. +export function planWindow(a: { + video: string; + start: number; + end: number; + pad: number; +}): { from: number; to: number } | { error: string } { + if (!isVideoId(a.video)) { + return { error: `fetch_clip: video "${a.video}" must match /^[\\w.-]+$/` }; + } + if (!(a.start < a.end)) { + return { + error: `fetch_clip: start (${a.start}) must be less than end (${a.end})`, + }; + } + const from = Number(Math.max(0, a.start - a.pad).toFixed(2)); + const to = Number((a.end + a.pad).toFixed(2)); + if (!(from < to)) { + return { error: `fetch_clip: start (${a.start}) must be less than end (${a.end})` }; + } + const span = Number((to - from).toFixed(2)); + if (span > MAX_CLIP_WINDOW_SECONDS) { + return { + error: + `fetch_clip: the window ${fmtSeconds(from)}–${fmtSeconds(to)} is ` + + `${span}s; the editor fetches at most ${MAX_CLIP_WINDOW_SECONDS}s per ` + + `window — cite a narrower span`, + }; + } + return { from, to }; +} + +// ─── Arguments ─── + +export type ClipTarget = + | { + kind: "window"; + channel: string; + video: string; + webpageUrl?: string; + from: number; + to: number; + pad: number; + } + | { kind: "full"; channel: string; video: string }; + +export type FetchClipRequest = + | { job: string; waitSeconds: number } + | { + target: ClipTarget; + reason: string; + report?: string; + waitSeconds: number; + }; + +const trimmed = (v: unknown): string => + typeof v === "string" ? v.trim() : ""; + +function waitSecondsOf(v: unknown): number { + const n = typeof v === "number" ? v : Number.NaN; + if (!Number.isFinite(n)) return DEFAULT_WAIT_SECONDS; + return Math.min(MAX_WAIT_SECONDS, Math.max(0, n)); +} + +// Validate the tool's raw arguments into a request, before any HTTP. A `job` +// alone is a whole request (resume): every other argument is then ignored. In +// full mode start/end/pad are ignored — not required, not validated. The +// returned video id is the one AS CITED; server.ts maps it to the editor's +// directory id (Rumble) before fetching. +export function validateFetchClipArgs( + args: Record<string, unknown>, +): { ok: true; request: FetchClipRequest } | { ok: false; error: string } { + const waitSeconds = waitSecondsOf(args.wait_seconds); + const job = trimmed(args.job); + if (job) return { ok: true, request: { job, waitSeconds } }; + + const channel = trimmed(args.channel); + if (!channel) { + return { ok: false, error: "fetch_clip: channel is required (the channel slug)" }; + } + const video = trimmed(args.video); + if (!video) { + return { + ok: false, + error: "fetch_clip: video is required (the archive's video id)", + }; + } + if (!isVideoId(video)) { + return { ok: false, error: `fetch_clip: video "${video}" must match /^[\\w.-]+$/` }; + } + + let target: ClipTarget; + if (args.full === true) { + target = { kind: "full", channel, video }; + } else { + const times: Record<"start" | "end", number> = { start: 0, end: 0 }; + for (const key of ["start", "end"] as const) { + const raw = args[key]; + if (raw === undefined || raw === null || raw === "") { + return { + ok: false, + error: + `fetch_clip: ${key} is required (seconds, mm:ss or h:mm:ss) — or ` + + `pass full: true for the whole recording`, + }; + } + const secs = parseSeconds(raw); + if (secs === null) { + return { + ok: false, + error: + `fetch_clip: ${key} "${String(raw)}" is not a time (use seconds, ` + + `mm:ss or h:mm:ss)`, + }; + } + times[key] = secs; + } + const pad = args.pad === undefined ? DEFAULT_PAD_SECONDS : args.pad; + if (typeof pad !== "number" || !Number.isFinite(pad) || pad < 0) { + return { + ok: false, + error: `fetch_clip: pad "${String(pad)}" must be a finite number of seconds, 0 or more`, + }; + } + const w = planWindow({ video, start: times.start, end: times.end, pad }); + if ("error" in w) return { ok: false, error: w.error }; + target = { kind: "window", channel, video, from: w.from, to: w.to, pad }; + } + + const reason = trimmed(args.reason); + if (!reason) { + return { + ok: false, + error: + "fetch_clip: reason is required — one line saying why these seconds " + + "are needed (it is stored beside the file)", + }; + } + const report = trimmed(args.report) || undefined; + return { ok: true, request: { target, reason, report, waitSeconds } }; +} + +// ─── The HTTP exchange ─── + +export type FetchClipOutcome = + | { kind: "no_editor" } + | { + kind: "cached"; + mode: "window" | "full"; + // The span that was asked for (window mode). + reqFrom: number; + reqTo: number; + file: string; + from: number; + to: number; + bytes: number; + requestedBy?: string; + } + | { + kind: "fetched"; + // For a resume, read off the poll's answer: a window job's carries + // from/to, a whole-recording job's does not. + mode: "window" | "full"; + jobId: string; + file?: string; + from?: number; + to?: number; + bytes?: number; + waited: number; + } + | { kind: "queued"; jobId: string; status: string; waited: number } + | { kind: "cooldown"; platform: string; cooldownMs: number; error: string } + | { + kind: "refused"; + phase: "post" | "poll"; + status: number; + error: string; + jobId?: string; + full?: boolean; + } + | { kind: "failed"; jobId: string; status: string; log: string } + | { kind: "unreachable"; url: string; message: string }; + +type Json = Record<string, unknown>; + +async function readJson(res: HttpResponse): Promise<Json> { + try { + const body = await res.json(); + return body && typeof body === "object" ? (body as Json) : {}; + } catch { + return {}; + } +} + +const num = (v: unknown): number | undefined => + typeof v === "number" && Number.isFinite(v) ? v : undefined; +const str = (v: unknown): string | undefined => + typeof v === "string" && v !== "" ? v : undefined; + +// POST the request (unless it is a resume), then poll every POLL_MS until the +// job is terminal or the wait runs out. A wait that runs out is not a failure: +// the job keeps running on the editor, and the answer says how to resume. +export async function fetchClip( + request: FetchClipRequest, + deps: FetchClipDeps, +): Promise<FetchClipOutcome> { + const editor = editorFromEnv(deps.env); + if (!editor) return { kind: "no_editor" }; + const headers = { + authorization: `Bearer ${editor.token}`, + "content-type": "application/json", + }; + const started = deps.now(); + const deadline = started + request.waitSeconds * 1000; + const waited = () => Math.round((deps.now() - started) / 1000); + + async function call(url: string, init: HttpInit): Promise<HttpResponse | Error> { + try { + return await deps.fetch(url, init); + } catch (e) { + return e instanceof Error ? e : new Error(String(e)); + } + } + + // Poll until terminal or out of time. `pollFirst` is a resume: the job may + // well have finished since it was queued, so ask before sleeping. + async function waitFor( + jobId: string, + mode: "window" | "full" | "unknown", + pollFirst: boolean, + ): Promise<FetchClipOutcome> { + let status = "queued"; + let skipSleep = pollFirst; + for (;;) { + if (!skipSleep) { + if (deps.now() >= deadline) { + return { kind: "queued", jobId, status, waited: waited() }; + } + await deps.sleep(POLL_MS); + } + skipSleep = false; + const res = await call( + `${editor!.url}/api/media/fetch-window/${encodeURIComponent(jobId)}`, + { method: "GET", headers }, + ); + if (res instanceof Error) { + return { kind: "unreachable", url: editor!.url, message: res.message }; + } + const body = await readJson(res); + if (res.status < 200 || res.status >= 300) { + return { + kind: "refused", + phase: "poll", + status: res.status, + error: str(body.error) ?? "no reason given", + jobId, + }; + } + status = str(body.status) ?? status; + if (status === "done") { + const from = num(body.from); + const to = num(body.to); + return { + kind: "fetched", + mode: + mode === "unknown" + ? from !== undefined && to !== undefined + ? "window" + : "full" + : mode, + jobId, + file: str(body.file), + from, + to, + bytes: num(body.bytes), + waited: waited(), + }; + } + if (status === "failed" || status === "cancelled") { + return { kind: "failed", jobId, status, log: str(body.error) ?? "" }; + } + } + } + + if ("job" in request) return waitFor(request.job, "unknown", true); + + const { target } = request; + const provenance = { + requestedBy: REQUESTED_BY, + manifest: request.report, + reason: request.reason.slice(0, MAX_REASON_CHARS), + }; + // `full` REPLACES the span (as in fetch-via-editor.mjs): the route reads + // `full === true` before it validates from/to, and the saved-video path + // resolves the URL itself, so a webpageUrl would describe a request the + // editor does not have. + const body = + target.kind === "full" + ? { + channelSlug: target.channel, + videoId: target.video, + full: true, + ...provenance, + } + : { + channelSlug: target.channel, + videoId: target.video, + webpageUrl: target.webpageUrl, + from: target.from, + to: target.to, + pad: target.pad, + ...provenance, + }; + const res = await call(`${editor.url}/api/media/fetch-window`, { + method: "POST", + headers, + body: JSON.stringify(body), + }); + if (res instanceof Error) { + return { kind: "unreachable", url: editor.url, message: res.message }; + } + const answer = await readJson(res); + const reqFrom = target.kind === "window" ? target.from : 0; + const reqTo = target.kind === "window" ? target.to : 0; + + if (res.status === 200 && str(answer.file)) { + const prov = answer.provenance as Json | null | undefined; + return { + kind: "cached", + mode: target.kind, + reqFrom, + reqTo, + file: str(answer.file)!, + from: num(answer.from) ?? reqFrom, + to: num(answer.to) ?? reqTo, + bytes: num(answer.bytes) ?? 0, + requestedBy: prov && typeof prov === "object" ? str(prov.requestedBy) : undefined, + }; + } + if (res.status === 202 && str(answer.jobId)) { + return waitFor(str(answer.jobId)!, target.kind, false); + } + if (res.status === 409) { + return { + kind: "cooldown", + platform: str(answer.platform) ?? "platform", + cooldownMs: num(answer.cooldownMs) ?? 0, + error: str(answer.error) ?? "", + }; + } + return { + kind: "refused", + phase: "post", + status: res.status, + error: str(answer.error) ?? "no reason given", + full: target.kind === "full", + }; +} + +// ─── The answer an agent reads ─── + +const READ_ONLY_NOTE = + "This path is a read-only corpus artifact: play or copy it, never move, " + + "edit or delete it."; + +function footer( + mode: "window" | "full", + f: { file: string; from?: number; to?: number; bytes?: number; requestedBy?: string }, +): string { + const lines = [`file: ${f.file}`]; + if (mode === "window" && f.from !== undefined && f.to !== undefined) { + lines.push( + `window: ${fmtSeconds(f.from)}–${fmtSeconds(f.to)} (${fmtSpan(f.to - f.from)})`, + ); + } + if (f.bytes !== undefined) lines.push(`bytes: ${f.bytes}`); + if (f.requestedBy) lines.push(`requested by ${f.requestedBy}`); + if (mode === "window") { + const name = + f.from !== undefined && f.to !== undefined + ? `${fmtSeconds(f.from)}-${fmtSeconds(f.to)}.json` + : "<from>-<to>.json"; + lines.push( + `${READ_ONLY_NOTE} The editor prunes clips by age (evict-clips); ` + + `provenance sits beside it as ${name}.`, + ); + } else { + lines.push( + `${READ_ONLY_NOTE} It lives in the editor's saved-video store, not in ` + + `clips/; the editor's keep-videos rule decides how long it stays.`, + ); + } + return lines.join("\n"); +} + +const TOKEN_HINT = + "WORKER_TOKEN must equal the value the editor runs with (editor/.env)."; + +// Only a 503 that SAYS the endpoint is disabled is a token problem: the same +// status also carries an unreachable-media refusal (a relocated channel whose +// drive is not mounted), which is the editor operator's to fix, verbatim. +function isTokenRefusal(status: number, error: string): boolean { + return status === 401 || (status === 503 && /disabled/i.test(error)); +} + +export function renderFetchClip( + outcome: FetchClipOutcome, + ctx: { channel?: string; video?: string } = {}, +): { text: string; isError: boolean } { + const of = ctx.channel && ctx.video ? ` of ${ctx.channel}/${ctx.video}` : ""; + switch (outcome.kind) { + case "no_editor": + return { text: NO_EDITOR_TEXT, isError: true }; + case "cached": { + let head: string; + if (outcome.mode === "full") { + head = `Already on disk — the whole recording${of}.`; + } else { + const same = + Math.abs(outcome.from - outcome.reqFrom) <= SAME_WINDOW_EPS && + Math.abs(outcome.to - outcome.reqTo) <= SAME_WINDOW_EPS; + head = same + ? `Already on disk — the exact window: ${fmtSeconds(outcome.from)}–${fmtSeconds(outcome.to)}.` + : `Already on disk — a WIDER cached window that contains ` + + `${fmtSeconds(outcome.reqFrom)}–${fmtSeconds(outcome.reqTo)}: ` + + `${fmtSeconds(outcome.from)}–${fmtSeconds(outcome.to)}.`; + } + return { text: `${head}\n\n${footer(outcome.mode, outcome)}`, isError: false }; + } + case "fetched": { + if (!outcome.file) { + return { + text: + `Editor job ${outcome.jobId} finished but named no file; check ` + + `the video's page in the editor.`, + isError: true, + }; + } + const head = + outcome.mode === "full" + ? `Fetched the whole recording${of} (job ${outcome.jobId}, ${outcome.waited}s waited).` + : `Fetched ${ + outcome.from !== undefined && outcome.to !== undefined + ? `${fmtSeconds(outcome.from)}–${fmtSeconds(outcome.to)}` + : "the window" + }${of} (job ${outcome.jobId}, ${outcome.waited}s waited).`; + return { + text: `${head}\n\n${footer(outcome.mode, { ...outcome, file: outcome.file })}`, + isError: false, + }; + } + case "queued": + return { + text: + `Still ${outcome.status} on the editor (job ${outcome.jobId}, waited ` + + `${outcome.waited}s). Call fetch_clip again with job: ` + + `"${outcome.jobId}" to keep waiting — nothing is lost, the fetch ` + + `continues on the editor and the next ask finds it cached.`, + isError: false, + }; + case "cooldown": + return { + text: + `The editor is in a ${outcome.platform} rate-limit cooldown — ` + + `${Math.ceil(outcome.cooldownMs / 1000)}s remaining. Wait, then call ` + + `fetch_clip again. (${outcome.error})`, + isError: true, + }; + case "refused": { + if (outcome.phase === "poll") { + const extra = + outcome.status === 404 + ? "; the job is unknown to this editor (restarted? wrong ARCHILYZER_EDITOR_URL?)" + : isTokenRefusal(outcome.status, outcome.error) + ? `. ${TOKEN_HINT}` + : ""; + return { + text: + `Polling editor job ${outcome.jobId} failed (HTTP ${outcome.status}): ` + + `${outcome.error}${extra}`, + isError: true, + }; + } + if (isTokenRefusal(outcome.status, outcome.error)) { + const off = + outcome.status === 503 + ? " The editor has no WORKER_TOKEN set, so its fetch endpoint is off." + : ""; + return { + text: + `The editor refused the token (HTTP ${outcome.status}): ` + + `${outcome.error}. ${TOKEN_HINT}${off}`, + isError: true, + }; + } + const known = + outcome.full && outcome.status === 404 + ? " A whole-recording fetch needs a video the editor already knows " + + "(a metadata.info.json or a playlist entry); fetch a window " + + "instead, or add the video to the channel first." + : ""; + return { + text: `The editor refused (HTTP ${outcome.status}): ${outcome.error}${known}`, + isError: true, + }; + } + case "failed": + return { + text: + `Editor job ${outcome.jobId} ${outcome.status}.\n\nlog tail:\n` + + (outcome.log || "(the job left no log)"), + isError: true, + }; + case "unreachable": + return { + text: `Could not reach the editor at ${outcome.url}: ${outcome.message}. Is it running?`, + isError: true, + }; + } +} + +// The prefix for a cited id the corpus does not hold: the id goes to the editor +// as-is, which for a Rumble EMBED id is the wrong directory. +export function notFoundNote(video: string, handle: string): string { + return ( + `note: "${video}" was not found in corpus ${handle}; the id was passed to ` + + `the editor as-is (for a Rumble citation this may be the embed id — pass ` + + `the corpus the citation came from as source).` + ); +}