import { describeEditorFailure, editorGet, editorPost, type EditorDeps } from "./editorOps"; import { describeFetchError, isVideoId, REQUEST_TIMEOUT_MS } from "./fetchClip"; // ─── Archival writes through the editor (release 19 A9) ─── // // The fetch_clip pattern: this server asks the LOCAL editor (its URL and its // WORKER_TOKEN) and the editor runs the job, with its queues, pacing, holds // and provenance. The MCP writes nothing itself. Each tool maps to ONE // existing ops route and nothing else: // // get_job GET /api/ops/job/?tail=N // enqueue POST /api/ops/{sync | download-missing | retry-bucket | // transcribe-bucket | fetch-posts | import-video} // channel_coverage GET /api/ops/coverage?slug=… // notes umtool (UMTOOL_URL): GET /api/browse/decisions (list), // GET /api/notes/context (read) // // Settings, storage and deletes are deliberately absent (operator ruling, // release 19): `pnpm ops` / `archilyzer` only. A notes REPLY is not written // from here: umtool stamps every write through its HTTP route as the // OPERATOR's ("there is no way to claim to be one here", umtool's // app/api/notes/route.ts), so an agent reply must go through `umtool notes // reply`, which stamps "agent" — the tool answers with that command. // // Pure apart from the injected deps; server.ts adapts. export type ToolAnswer = { text: string; isError?: boolean }; const SLUG_RE = /^[a-z0-9][a-z0-9._-]*$/i; const trimmed = (v: unknown) => (typeof v === "string" ? v.trim() : ""); // ─── get_job ─── export const DEFAULT_JOB_TAIL = 40; export const MAX_JOB_TAIL = 500; type JobView = { id: string; kind?: string; channelSlug?: string; videoId?: string; status: string; queuedAt?: number; startedAt?: number; endedAt?: number; exitCode?: number; detail?: string; cancelReason?: string; queue?: { key: string; position: number; queued: number; head?: { id: string; kind: string; channelSlug?: string } }; progress?: unknown; }; const iso = (ms?: number) => (typeof ms === "number" && ms > 0 ? new Date(ms).toISOString() : null); export function renderJob(job: JobView, tail?: string[]): string { const lines: string[] = []; lines.push( `job ${job.id}: ${job.kind ?? "job"}${job.channelSlug ? ` on ${job.channelSlug}` : ""}${job.videoId ? `/${job.videoId}` : ""} — **${job.status}**`, ); if (job.detail) lines.push(`for: ${job.detail}`); const times = [ iso(job.queuedAt) ? `queued ${iso(job.queuedAt)}` : "", iso(job.startedAt) ? `started ${iso(job.startedAt)}` : "", iso(job.endedAt) ? `ended ${iso(job.endedAt)}` : "", ].filter(Boolean); if (times.length) lines.push(times.join(" · ")); if (typeof job.exitCode === "number") lines.push(`exit code ${job.exitCode}`); if (job.cancelReason) lines.push(`cancelled because: ${job.cancelReason}`); if (job.queue) { const q = job.queue; lines.push( q.position === 0 ? `running at the head of queue "${q.key}" (${q.queued} waiting behind)` : `waiting on queue "${q.key}": position ${q.position} of ${q.queued}` + (q.head ? `, behind ${q.head.kind}${q.head.channelSlug ? ` on ${q.head.channelSlug}` : ""} (${q.head.id})` : ""), ); } if (job.progress !== undefined) lines.push(`progress: ${JSON.stringify(job.progress)}`); if (job.status === "queued" || job.status === "running") { lines.push("Still going: call get_job again later (a platform queue may hold a job for hours)."); } if (tail && tail.length) { lines.push("", `last ${tail.length} log line(s):`, "```", ...tail, "```"); } return lines.join("\n"); } export async function getJob(args: Record, deps: EditorDeps): Promise { const id = trimmed(args.job); if (!id || !/^[\w.-]+$/.test(id)) return { text: "get_job: job is required (the job id an enqueue or fetch_clip returned)", isError: true }; const rawTail = args.tail === undefined ? DEFAULT_JOB_TAIL : Number(args.tail); if (!Number.isInteger(rawTail) || rawTail < 0) return { text: "get_job: tail is a whole number of lines, 0 or more", isError: true }; const tail = Math.min(MAX_JOB_TAIL, rawTail); const a = await editorGet(deps, `/api/ops/job/${encodeURIComponent(id)}${tail > 0 ? `?tail=${tail}` : ""}`); if (a.kind === "refused" && a.status === 404) return { text: `get_job: the editor knows no job ${id}`, isError: true }; if (a.kind !== "ok") return { text: describeEditorFailure("get_job", a), isError: true }; const job = a.body.job as JobView | undefined; if (!job || typeof job.status !== "string") return { text: "get_job: the editor's answer carried no job", isError: true }; const lines = Array.isArray(a.body.tail) ? (a.body.tail as unknown[]).map(String) : undefined; return { text: renderJob(job, lines) }; } // ─── enqueue ─── export const ENQUEUE_KINDS = [ "sync", "download-missing", "retry-bucket", "transcribe-bucket", "fetch-posts", "import-video", ] as const; export type EnqueueKind = (typeof ENQUEUE_KINDS)[number]; // Which tool arguments each kind takes, beyond `channel`. Anything else is // refused by name — the ops routes refuse unknown keys too, but saying it here // costs no request. const KIND_ARGS: Record = { sync: ["full"], "download-missing": [], "retry-bucket": ["bucket", "ids"], "transcribe-bucket": ["ids"], "fetch-posts": ["full", "older"], "import-video": ["url"], }; const ALL_ARGS = ["full", "older", "bucket", "ids", "url"]; export function enqueueBody( args: Record, ): { ok: true; kind: EnqueueKind; route: string; body: Record } | { ok: false; error: string } { const kind = trimmed(args.kind) as EnqueueKind; if (!ENQUEUE_KINDS.includes(kind)) { return { ok: false, error: `enqueue: kind must be one of ${ENQUEUE_KINDS.join(", ")}` }; } const channel = trimmed(args.channel); if (!channel || !SLUG_RE.test(channel)) return { ok: false, error: "enqueue: channel is required (the channel slug)" }; const allowed = KIND_ARGS[kind]; const stray = ALL_ARGS.filter((k) => args[k] !== undefined && !allowed.includes(k)); if (stray.length) { return { ok: false, error: `enqueue: ${kind} does not take ${stray.join(", ")}${allowed.length ? ` (it takes ${allowed.join(", ")})` : ""}`, }; } const body: Record = { slug: channel }; for (const k of ["full", "older"] as const) { if (args[k] === undefined) continue; if (typeof args[k] !== "boolean") return { ok: false, error: `enqueue: ${k} must be true or false` }; if (args[k]) body[k] = true; } if (args.ids !== undefined) { const ids = args.ids; if (!Array.isArray(ids) || ids.length === 0 || ids.some((i) => typeof i !== "string" || !isVideoId(i))) { return { ok: false, error: "enqueue: ids must be a non-empty list of video ids" }; } body.ids = ids; } if (kind === "retry-bucket") { const bucket = trimmed(args.bucket); if (!bucket) return { ok: false, error: "enqueue: retry-bucket needs bucket (a bucket of the channel's report, e.g. noTranscript)" }; body.bucket = bucket; } if (kind === "import-video") { const url = trimmed(args.url); if (!/^https?:\/\//i.test(url)) return { ok: false, error: "enqueue: import-video needs url (the video's page)" }; body.url = url; } return { ok: true, kind, route: `/api/ops/${kind}`, body }; } export async function enqueue(args: Record, deps: EditorDeps): Promise { const plan = enqueueBody(args); if (!plan.ok) return { text: plan.error, isError: true }; const a = await editorPost(deps, plan.route, plan.body); if (a.kind !== "ok") return { text: describeEditorFailure(`enqueue ${plan.kind}`, a), isError: true }; const ids = Array.isArray(a.body.jobIds) ? (a.body.jobIds as unknown[]).map(String) : typeof a.body.jobId === "string" ? [a.body.jobId] : []; if (ids.length === 0) { return { text: `enqueue ${plan.kind}: the editor accepted it and started no job (${JSON.stringify(a.body)})` }; } return { text: `enqueue ${plan.kind} on ${plan.body.slug}: queued as ${ids.map((i) => `job ${i}`).join(", ")}. ` + `It runs on the editor's queue at the platform's pace and may wait behind other work; ` + `follow it with get_job (job: "${ids[0]}").`, }; } // ─── channel_coverage ─── type Coverage = { slug: string; titlePattern: string | null; held: number; dated: number; byRecordedDate: number; byUploadDate: number; undated: string[]; first: string | null; last: string | null; byYear: Record; byMonth: Record; gapDays: number; gaps: { after: string; before: string; days: number }[]; videos?: { id: string; date: string; title?: string; recordedDate?: string }[]; }; const day = (d: string | null) => (d && /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : "—"); const DATE_ARG = /^(\d{4})-?(\d{2})-?(\d{2})$/; export function renderCoverage(c: Coverage): string { const lines: string[] = []; lines.push(`# Coverage of ${c.slug} (held on the editor's disk)`); lines.push( `${c.held} held · ${c.dated} dated (${c.byRecordedDate} by recorded date, ${c.byUploadDate} by upload date)` + `${c.undated.length ? ` · ${c.undated.length} undated` : ""} · ${day(c.first)} → ${day(c.last)}`, ); lines.push( c.titlePattern ? `Dates: the recorded date read off each title (rule \`${c.titlePattern}\`), else the upload date.` : "Dates: upload dates (the channel has no recorded-date title rule).", ); const years = Object.entries(c.byYear).sort(([a], [b]) => a.localeCompare(b)); if (years.length) lines.push("", "## By year", ...years.map(([y, n]) => `- ${y}: ${n}`)); if (c.gaps.length) { lines.push("", `## Gaps longer than ${c.gapDays} days (${c.gaps.length})`); for (const g of c.gaps) lines.push(`- ${day(g.after)} → ${day(g.before)}: ${g.days} days with nothing held`); } else { lines.push("", `No gap longer than ${c.gapDays} days.`); } if (c.undated.length) { const shown = c.undated.slice(0, 20); lines.push("", `Undated (no metadata with an upload date): ${shown.join(", ")}${c.undated.length > 20 ? `, … (+${c.undated.length - 20})` : ""}`); } if (c.videos) { lines.push("", `## Videos (${c.videos.length})`); for (const v of c.videos) lines.push(`- ${day(v.date)} ${v.id}${v.title ? ` — ${v.title}` : ""}${v.recordedDate ? " (recorded)" : ""}`); } return lines.join("\n"); } export async function channelCoverageTool(args: Record, deps: EditorDeps): Promise { const channel = trimmed(args.channel); if (!channel || !SLUG_RE.test(channel)) return { text: "channel_coverage: channel is required (the channel slug)", isError: true }; const q = new URLSearchParams({ slug: channel }); if (args.gap_days !== undefined) { const n = Number(args.gap_days); if (!Number.isInteger(n) || n < 1) return { text: "channel_coverage: gap_days is a whole number of days above zero", isError: true }; q.set("gapDays", String(n)); } for (const [arg, key] of [["date_from", "from"], ["date_to", "to"]] as const) { if (args[arg] === undefined) continue; const m = DATE_ARG.exec(trimmed(args[arg])); if (!m) return { text: `channel_coverage: ${arg} is a date, YYYY-MM-DD`, isError: true }; q.set(key, `${m[1]}${m[2]}${m[3]}`); } if (args.list === true) q.set("list", "1"); const a = await editorGet(deps, `/api/ops/coverage?${q.toString()}`); if (a.kind !== "ok") return { text: describeEditorFailure("channel_coverage", a), isError: true }; return { text: renderCoverage(a.body as unknown as Coverage) }; } // ─── notes (umtool) ─── export const NO_UMTOOL_TEXT = "notes: no umtool configured. Set UMTOOL_URL (e.g. http://localhost:3050) when " + "registering the MCP server; the notes live in umtool."; export type NotesDeps = EditorDeps; function umtoolFrom(env: Record): string | null { const raw = (env.UMTOOL_URL ?? "").trim(); return raw ? raw.replace(/\/+$/, "") : null; } async function umtoolGet( deps: NotesDeps, route: string, ): Promise<{ ok: true; status: number; json?: unknown; text?: string } | { ok: false; error: string }> { const base = umtoolFrom(deps.env); if (!base) return { ok: false, error: NO_UMTOOL_TEXT }; const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS; try { const res = await deps.fetch(`${base}${route}`, { method: "GET", signal: AbortSignal.timeout(timeoutMs) }); const body = await res.json().catch(() => null); return { ok: true, status: res.status, json: body }; } catch (e) { return { ok: false, error: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}` }; } } // A target as notes list names it: `sites//` is an article, // anything else a report-video project id (which may hold a "/" too — the // prefix is what tells them apart). export function notesTarget(raw: string): { param: "article" | "project"; value: string } | null { const t = raw.trim(); if (!t || /\s|\.\.|^\//.test(t)) return null; const article = /^sites\/([a-z0-9][a-z0-9-]*\/[A-Za-z0-9][A-Za-z0-9._-]*)$/.exec(t); if (article) return { param: "article", value: article[1] }; if (t.startsWith("sites/")) return null; return { param: "project", value: t }; } type Decision = { kind: string; project: string; projectKind?: string; target?: string; why?: string; href?: string }; export async function notesTool(args: Record, deps: NotesDeps): Promise { const action = trimmed(args.action) || "list"; if (action === "reply") { const id = trimmed(args.note_id) || ""; const text = trimmed(args.text) || ""; return { isError: true, text: "notes: a reply is not written from here. umtool records every write through its HTTP " + "route as the operator's, and an agent's reply must say it is an agent's — so it goes " + "through umtool's CLI, from the repo checkout:\n\n" + ` umtool notes reply ${id} ${JSON.stringify(text)}${args.resolve === true ? " --resolve" : ""}\n\n` + "(`node umtool/bin/umtool.mjs notes reply …` when umtool is not on the PATH.) Edit the " + "note's SOURCE file first (notes read shows it), regenerate, then reply.", }; } if (action === "list") { const r = await umtoolGet(deps, "/api/browse/decisions"); if (!r.ok) return { text: r.error, isError: true }; if (r.status !== 200) return { text: `notes: umtool answered HTTP ${r.status}`, isError: true }; const items = ((r.json as { items?: Decision[] } | null)?.items ?? []).filter( (d) => d.kind === "open-note" || d.kind === "unreadable-notes", ); if (items.length === 0) return { text: "No open notes." }; const byProject = new Map(); for (const d of items) byProject.set(d.project, [...(byProject.get(d.project) ?? []), d]); const lines = [`${items.length} open note(s) on ${byProject.size} target(s):`]; for (const [project, ds] of byProject) { lines.push("", `## ${project} (${ds[0].projectKind ?? "project"}) — read with notes action "read", target "${project}"`); for (const d of ds) { lines.push(d.kind === "unreadable-notes" ? `- UNREADABLE notes.json: ${d.why ?? ""}` : `- [${d.target ?? "note"}] ${d.why ?? ""}`); } } return { text: lines.join("\n") }; } if (action === "read") { const target = notesTarget(trimmed(args.target)); if (!target) { return { text: 'notes: read needs target as notes list names it — "sites//" for an article, else a video project id', isError: true }; } const status = trimmed(args.status) || "open"; if (!["open", "resolved", "all"].includes(status)) return { text: "notes: status is open, resolved or all", isError: true }; const base = umtoolFrom(deps.env); if (!base) return { text: NO_UMTOOL_TEXT, isError: true }; const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS; const q = new URLSearchParams({ [target.param]: target.value, status }); try { const res = await deps.fetch(`${base}/api/notes/context?${q.toString()}`, { method: "GET", signal: AbortSignal.timeout(timeoutMs), }); // The digest is text/plain; an error is JSON. const raw = (res as unknown as { text?: () => Promise }).text ? await (res as unknown as { text: () => Promise }).text() : JSON.stringify(await res.json()); if (res.status !== 200) { let error = raw; try { error = (JSON.parse(raw) as { error?: string }).error ?? raw; } catch { /* plain text */ } return { text: `notes: umtool refused (HTTP ${res.status}): ${error}`, isError: true }; } return { text: raw }; } catch (e) { return { text: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}`, isError: true }; } } return { text: 'notes: action is "list", "read" or "reply"', isError: true }; } // ─── The tool definitions (server.ts's TOOLS takes them in order) ─── export const ARCHIVAL_TOOLS = [ { name: "get_job", description: "One editor job's state — kind, channel, status, times, exit code, where it waits " + "in its queue — and the last lines of its log. For the job id enqueue or fetch_clip " + "returned. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.", inputSchema: { type: "object", properties: { job: { type: "string", description: "The job id." }, tail: { type: "number", description: `Log lines to include (default ${DEFAULT_JOB_TAIL}, at most ${MAX_JOB_TAIL}; 0 for none).`, }, }, required: ["job"], additionalProperties: false, }, }, { name: "enqueue", description: "Ask the local editor to queue archival work on a channel it archives, as the " + "channel page's buttons do: sync (list and download what is new; full: true sweeps " + "the whole listing), download-missing (every listed video not held), retry-bucket " + "(one bucket of the channel's report, or ids within it — the way to download chosen " + "videos), transcribe-bucket (the downloaded, untranscribed videos, or ids among them), " + "fetch-posts (a social channel's new posts; full / older walks), import-video (one " + "video by URL into the channel). The editor runs it on its own queues at each " + "platform's pace and answers a job id — follow it with get_job. Only when the " + "operator asks for it. Settings, storage and deletes are not reachable from here. " + "Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.", inputSchema: { type: "object", properties: { kind: { type: "string", enum: [...ENQUEUE_KINDS], description: "What to queue." }, channel: { type: "string", description: "The channel slug (the editor's)." }, full: { type: "boolean", description: "sync: sweep the whole listing now; fetch-posts: re-walk the timeline." }, older: { type: "boolean", description: "fetch-posts: walk back below the oldest archived post." }, bucket: { type: "string", description: "retry-bucket: the bucket name (e.g. noTranscript, partialDownloads)." }, ids: { type: "array", items: { type: "string" }, description: "retry-bucket / transcribe-bucket: only these video ids (each must be in the bucket).", }, url: { type: "string", description: "import-video: the video's page URL." }, }, required: ["kind", "channel"], additionalProperties: false, }, }, { name: "channel_coverage", description: "What a channel HOLDS on the editor's disk, by date: how many videos, the first and " + "last day, counts per year, and every gap longer than gap_days with nothing held. " + "Each video is dated by its recorded date when the channel has a recorded-date " + "title rule (a VOD mirror), else by its upload date. Counts what is held, not what " + "is published. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.", inputSchema: { type: "object", properties: { channel: { type: "string", description: "The channel slug (the editor's)." }, gap_days: { type: "number", description: "The shortest span reported as a gap (default 30)." }, date_from: { type: "string", description: "Only videos on or after this date (YYYY-MM-DD)." }, date_to: { type: "string", description: "Only videos on or before this date (YYYY-MM-DD)." }, list: { type: "boolean", description: "Also list every dated video (default false)." }, }, required: ["channel"], additionalProperties: false, }, }, { name: "notes", description: "The operator's notes in umtool on articles and report-video projects. action " + '"list": every open note, grouped by target. action "read" with target (as list names ' + 'it: "sites//" for an article, else a project id): each note with its anchor ' + "resolved and the SOURCE file to edit (status: open, resolved or all). action " + '"reply" answers with the `umtool notes reply` command to run — umtool records a ' + "reply sent over HTTP as the operator's, so an agent's reply goes through its CLI. " + "Needs UMTOOL_URL.", inputSchema: { type: "object", properties: { action: { type: "string", enum: ["list", "read", "reply"], description: 'Default "list".' }, target: { type: "string", description: 'read: as list names it — "sites//" (an article) or a project id.' }, status: { type: "string", enum: ["open", "resolved", "all"], description: "read: which notes (default open)." }, note_id: { type: "string", description: "reply: the note id (n_…)." }, text: { type: "string", description: "reply: what changed." }, resolve: { type: "boolean", description: "reply: resolve the note too." }, }, additionalProperties: false, }, }, ] as const;