import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import { REQUEST_TIMEOUT_MS, describeFetchError, editorFromEnv, isVideoId, type FetchClipDeps, type HttpInit, } from "./fetchClip"; // ─── The editor's ops routes, for the MCP (release 19 A7, A9) ─── // // The fetch_clip pattern, generalised: the MCP asks the LOCAL editor // (ARCHILYZER_EDITOR_URL + WORKER_TOKEN, the editor's own) and the editor // reads or writes. This process still writes nothing to an archive. Every // request is bounded (REQUEST_TIMEOUT_MS); an editor that does not answer is // said in words, never as a stack. // // Settings, storage and deletes are NOT reachable from here (operator ruling, // release 19): those stay `pnpm ops` / `archilyzer` only. export type EditorDeps = Pick; export type EditorAnswer = | { kind: "ok"; status: number; body: Record } | { kind: "refused"; status: number; error: string; body: Record } | { kind: "no_editor" } | { kind: "unreachable"; url: string; message: string }; export const NO_EDITOR_OPS_TEXT = "no editor configured. Set ARCHILYZER_EDITOR_URL (e.g. http://localhost:3001) " + "and WORKER_TOKEN (the editor's own WORKER_TOKEN) when registering the MCP " + "server. A public-only setup has no editor to ask."; async function call( deps: EditorDeps, method: "GET" | "POST", route: string, body?: unknown, ): Promise { const editor = editorFromEnv(deps.env); if (!editor) return { kind: "no_editor" }; const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS; const init: HttpInit = { method, headers: { authorization: `Bearer ${editor.token}`, ...(body !== undefined ? { "content-type": "application/json" } : {}), }, ...(body !== undefined ? { body: JSON.stringify(body) } : {}), signal: AbortSignal.timeout(timeoutMs), }; let res; try { res = await deps.fetch(`${editor.url}${route}`, init); } catch (e) { return { kind: "unreachable", url: editor.url, message: describeFetchError(e, timeoutMs) }; } let parsed: Record = {}; try { const j = await res.json(); if (j && typeof j === "object" && !Array.isArray(j)) parsed = j as Record; } catch { /* a non-JSON answer is refused below with its status */ } if (res.status >= 200 && res.status < 300 && parsed.ok !== false) { return { kind: "ok", status: res.status, body: parsed }; } const error = typeof parsed.error === "string" && parsed.error ? parsed.error : `HTTP ${res.status}`; return { kind: "refused", status: res.status, error, body: parsed }; } export const editorGet = (deps: EditorDeps, route: string) => call(deps, "GET", route); export const editorPost = (deps: EditorDeps, route: string, body: unknown) => call(deps, "POST", route, body); // An answer that is not `ok`, as one sentence for the agent. export function describeEditorFailure(tool: string, a: Exclude): string { switch (a.kind) { case "no_editor": return `${tool}: ${NO_EDITOR_OPS_TEXT}`; case "unreachable": return `${tool}: the editor at ${a.url} did not answer: ${a.message}`; case "refused": return a.status === 401 || a.status === 503 ? `${tool}: the editor refused the token (HTTP ${a.status}: ${a.error}) — WORKER_TOKEN must be the editor's own` : `${tool}: the editor refused (HTTP ${a.status}): ${a.error}`; } } // ─── get_transcript's fallback: cues off the editor's disk (A7) ─── export type EditorTranscript = { slug: string; id: string; source: "cues.json" | "built" | "vtt"; cuesJson: "fresh" | "stale" | "missing"; title?: string; channel?: string; uploadDate?: string; duration?: number; webpageUrl?: string; description?: string; file?: string; cues: Cue[]; }; const SOURCE_WORDS: Record = { "cues.json": "its transcript.cues.json", built: "its raw transcript, normalized in memory (no cues.json written)", vtt: "its English VTT alone (the record has no metadata)", }; // Ask the editor for one video's cues off disk. Null when there is no editor // to ask, or it holds no such video — the caller's "video not found" stands. export async function editorTranscript( deps: EditorDeps, videoId: string, channel?: string, ): Promise<{ ok: true; t: EditorTranscript } | { ok: false; error: string | null }> { if (!isVideoId(videoId)) return { ok: false, error: null }; const q = new URLSearchParams({ id: videoId }); if (channel && /^[a-z0-9][a-z0-9._-]*$/i.test(channel)) q.set("slug", channel); const a = await editorGet(deps, `/api/ops/transcript?${q.toString()}`); if (a.kind === "no_editor") return { ok: false, error: null }; if (a.kind === "refused" && a.status === 404) return { ok: false, error: null }; if (a.kind !== "ok") return { ok: false, error: describeEditorFailure("get_transcript", a) }; const b = a.body as unknown as EditorTranscript; if (!Array.isArray(b.cues)) return { ok: false, error: "get_transcript: the editor's answer carried no cues" }; return { ok: true, t: b }; } export function renderEditorTranscript(t: EditorTranscript, opts: { timestamps: boolean }): string { const header = [ `NOT IN THE ARCHIVE — read from the local editor's disk (channel ${t.slug}): ${SOURCE_WORDS[t.source]}${t.file ? `, ${t.file}` : ""}.`, "No moment links: the video is not published yet. Cite it by its source URL and time,", "and expect the published text to differ once an index build normalizes it.", ].join(" "); const md = transcriptToMarkdown( { id: t.id, title: t.title ?? t.id, ...(t.channel ? { channel: t.channel } : {}), channelSlug: t.slug, ...(t.uploadDate ? { uploadDate: t.uploadDate } : {}), ...(typeof t.duration === "number" ? { duration: t.duration } : {}), ...(t.webpageUrl ? { webpageUrl: t.webpageUrl } : {}), ...(t.description ? { description: t.description } : {}), cues: t.cues, }, { timestamps: opts.timestamps }, ); return `> ${header}\n\n${md}`; }