commit dcf45f12e40614b846e2cc335d15823058fc342c
parent 56a52ba8c959611cae4f09c8b7fa5dcad44bf37f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 10 Oct 2026 02:50:47 -0400
mcp: archival tools through the editor — get_job, enqueue, channel_coverage, notes (release 19 A9)
On the fetch_clip pattern (editor URL + WORKER_TOKEN; the editor writes),
each tool one existing route: get_job reads GET /api/ops/job/<id> with its
log tail; enqueue posts sync, download-missing, retry-bucket (chosen ids),
transcribe-bucket, fetch-posts or import-video and answers the job id;
channel_coverage reads the new GET /api/ops/coverage (`pnpm ops get
coverage <slug>`), which dates every held video off disk with coverageDate
— the recorded date when the channel has a title rule (release 20 D1),
else the upload date — and reports per year and month and every gap past
gapDays. notes reads umtool (UMTOOL_URL): list from /api/browse/decisions,
read from /api/notes/context; a reply answers with the `umtool notes reply`
command, since umtool's HTTP route records every write as the operator's.
Settings, storage and deletes stay CLI/ops-only. The plans' instructions
gain the archive-work step; README tool table, OPERATING, ENVIRONMENT.md
(UMTOOL_URL now read by the MCP) and COMMANDS.md follow.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
16 files changed, 1185 insertions(+), 15 deletions(-)
diff --git a/COMMANDS.md b/COMMANDS.md
@@ -79,7 +79,7 @@ Drives a running editor over HTTP (`/api/ops/*`, the same actions its pages run)
| action | what it does | example |
|---|---|---|
| `channel-priority` | channel-priority sets channels' priority, as the /channels deck's tier control does: {"slugs", "tier": "normal" \| "low" \| "paused"} sets the base tier; with "operation" (sync, transcription, download, digest, backfill) it pins that operation's tier, "tier": null clearing it back to the base; "preset": "sync-only" \| "clear" is the deck's two shortcuts and takes no tier. A manual change clears an automatic pause. | `pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'` |
-| `channel-config` | channel-config changes a channel as its Configure form does: {"slug"} and any of "patch" (form field names; "" clears one), "sites" (the WHOLE membership set: \[{"siteId", "groupId"? \| "newGroupName"?}\], \[\] = on no site; an unknown site id is refused), "excludeFromBuild" and "excludeFromCleanup" (set to the value given, not toggled). | `pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'` |
+| `channel-config` | channel-config changes a channel as its Configure form does: {"slug"} and any of "patch" (form field names; "" clears one), "sites" (the WHOLE membership set: \[{"siteId", "groupId"? \| "newGroupName"?}\], \[\] = on no site; an unknown site id is refused), "excludeFromBuild" and "excludeFromCleanup" (set to the value given, not toggled). A VOD mirror's recorded date: "patch": {"recordedDateTitlePattern": "<regex with named groups year, month, day>"} (CHANNEL.md, recordedDate). | `pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}'`<br>`pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}'` |
| `create-channel` | create-channel is the New channel form: {"fields": {"name", "handling": "youtube"\|"transcribe", "url"?, "platform"?, "sourceKind"?, "postFetcher"?, "socialHandle"?, …}} with channel-config's patch keys; "slug"? (else derived from the name), "sites"? (absent = on no site). "fetchPlaylist", "fetchPostsNow" and "prioritizeDownload" are the form's checkboxes, OFF unless true; a job they start comes back as jobId(s), so --wait follows it. | `pnpm ops create-channel --json '{"fields":{"name":"Example (X)","handling":"transcribe","url":"https://x.com/example"}}'` |
| `rename-channel` | rename-channel moves a channel to a new slug, as Danger → Rename does: {"slug", "newSlug"}. Refused while the channel is busy (a job, a lane unit, media in transition) or when the new slug is taken. Old links break. | `pnpm ops rename-channel --json '{"slug":"old-slug","newSlug":"new-slug"}'` |
| `delete-channel` | delete-channel removes a channel's whole directory, as Danger → Delete does: {"slug", "confirm"} — "confirm" must repeat the slug. No undo outside the transcripts/ repo's own history. | `pnpm ops delete-channel --json '{"slug":"x","confirm":"x"}'` |
@@ -145,6 +145,7 @@ Usage: pnpm ops <action> [--json '<body>' | --file <path>] [--wait]
pnpm ops get cleanup <slug>
pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]
pnpm ops get transcript <videoId> [--slug <slug>]
+ pnpm ops get coverage <slug>
pnpm ops list
--wait follows the job's log and survives a poll that fails (a busy
@@ -181,6 +182,12 @@ get cleanup <slug> is one channel's /cleanup row: what each sweep would
reclaim (they overlap — never add them), what holds the rest, and the
failed-transcriptions count. measured: false means unknown, not zero.
+get coverage <slug> is what a channel holds by date: held, dated (by
+ recorded date — the channel's recordedDate title rule — else by upload
+ date), undated, first and last day, per year and month, and every gap
+ over 30 days with nothing held. Off disk; nothing written. (The MCP's
+ channel_coverage takes gap days, a date window and a video list.)
+
get transcript <videoId> [--slug <slug>] reads one video's cues off disk
with no index: a fresh cues.json, else what build-cues would write, else
the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -53,7 +53,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| Variable | Default | What it does | Read by |
|---|---|---|---|
-| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts |
+| `WORKER_TOKEN` | unset (both surfaces off) | Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`; the MCP's `fetch_clip`, `enqueue`, `get_job`, `channel_coverage` and `get_transcript`'s editor fallback). Set the same value on both ends. | common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts |
| `SYNC_HEARTBEAT_SECONDS` | `settings.syncScheduler.heartbeatSeconds` | Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead). | editor/app/scheduler/heartbeat.ts |
| `SYNC_TICK_URL` | `http://127.0.0.1:3001/api/scheduler/tick` | Where `archilyzer sync tick` (cron's heartbeat) posts. | common/bin/sync-tick.ts |
| `SYNC_TICK_TOKEN` | unset (no auth) | Bearer token for the tick endpoint; set on both the editor and the cron job. | common/bin/sync-tick.ts, editor/app/scheduler/auth.ts |
@@ -82,7 +82,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts |
| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts |
| `ARCHILYZER_X_BROWSER` | the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium | The Chromium-family browser /settings' "Connect X account" opens (a path, or a name looked up on PATH), and a forum-thread channel's "Connect forum session" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser. | common/social/xBrowser.ts |
-| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx |
+| `UMTOOL_URL` | unset (no link; the MCP's `notes` off) | umtool's front door; when set, the video page links to it, and the MCP's `notes` tool reads the operator's notes there (e.g. `http://localhost:3050`). | editor/app/channels/[slug]/videos/[id]/page.tsx, mcp/src/archivalTools.ts |
| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts |
| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts |
| `TRANSCRIPT_LOCAL_DIR` | — | MCP server: a composed public dir on disk. | mcp/src/sources.ts |
@@ -93,7 +93,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `ARCHILYZER_INDEX_ALLOW_HELD` | off | `1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel. | common/controller/buildIndex.ts |
| `UV_THREADPOOL_SIZE` | `16` for the editor (`4` is Node's own) | Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive. | Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh) |
| `MCP_IO_STATS` | off | `1` turns on per-call I/O accounting, for `mcp/bench`. | common/lib/archive/io-stats.ts |
-| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's `fetch_clip` talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool, common/bin/migrate-media-tier.ts |
+| `ARCHILYZER_EDITOR_URL` | `http://localhost:3001` | Which editor `pnpm ops` and the MCP's editor-backed tools (`fetch_clip`, `enqueue`, `get_job`, `channel_coverage`, `get_transcript`'s fallback) talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it. | scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts, umtool, common/bin/migrate-media-tier.ts |
| `ARCHILYZER_AGENT` | `cli` | Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`. | scripts/archilyzer-ops.mjs |
| `DIARIZE_ENGINE_KIND` | `sherpa-onnx` | The diarization engine: `sherpa-onnx` or `sortformer`. | scripts/diarize.mjs |
| `DIARIZE_ENGINE_CMD` | the bundled sherpa script | The engine command the wrapper runs. | scripts/diarize.mjs |
diff --git a/OPERATING.md b/OPERATING.md
@@ -174,4 +174,8 @@ pnpm archilyzer publish build example-site --out /srv/example-private
- `/ask` answers a question with citations in the conversation; `/sweep` writes a cited report to a file.
- Search first, then pull only the cited seconds with `fetch_clip`. The editor fetches only for a channel it
already archives.
+- With the editor configured, the MCP also queues archival work (`enqueue`: sync, download-missing, retry-bucket,
+ transcribe-bucket, fetch-posts, import-video), follows it (`get_job`), reports what a channel holds by date and
+ where the gaps are (`channel_coverage`), and reads a video not yet published (`get_transcript`). The operator's
+ notes: `notes` (with `UMTOOL_URL`). Settings, storage and deletes stay `pnpm ops`.
- Tools and their arguments: [mcp/README.md](mcp/README.md).
diff --git a/common/controller/channelCoverage.test.ts b/common/controller/channelCoverage.test.ts
@@ -0,0 +1,95 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/channelCoverage.test.ts
+//
+// Coverage by date (release 19 A9): the recorded date wins when the channel's
+// title rule yields one (release 20 D1's coverageDate), a big metadata file is
+// read at both ends, gaps are the spans past gapDays. Every id is invented.
+
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "coverage-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(process.env.SETTINGS_FILE, "{}\n");
+
+const { channelCoverage, readDateAndTitle, summarizeCoverage } = await import("./channelCoverage");
+const { getPaths } = await import("../lib/paths");
+
+function channel(slug: string, config: Record<string, unknown>, videos: Record<string, Record<string, unknown> | string | null>) {
+ const dir = path.join(ROOT, "channels", slug);
+ mkdirSync(path.join(dir, "data"), { recursive: true });
+ writeFileSync(path.join(dir, "config.json"), JSON.stringify({ handling: "youtube", name: slug, ...config }));
+ for (const [id, meta] of Object.entries(videos)) {
+ mkdirSync(path.join(dir, "data", id), { recursive: true });
+ if (meta === null) continue;
+ writeFileSync(path.join(dir, "data", id, "metadata.info.json"), typeof meta === "string" ? meta : JSON.stringify(meta));
+ }
+}
+
+test("gaps are the spans longer than gapDays, oldest first; months and years counted", () => {
+ const s = summarizeCoverage(
+ [
+ { id: "a", date: "20240101", uploadDate: "20240101" },
+ { id: "b", date: "20240115", uploadDate: "20240115" },
+ { id: "c", date: "20240401", uploadDate: "20240401" },
+ { id: "d", date: "20250101", uploadDate: "20250101" },
+ ],
+ { gapDays: 30 },
+ );
+ assert.equal(s.first, "20240101");
+ assert.equal(s.last, "20250101");
+ assert.deepEqual(s.byYear, { "2024": 3, "2025": 1 });
+ assert.deepEqual(s.byMonth, { "2024-01": 2, "2024-04": 1, "2025-01": 1 });
+ assert.deepEqual(s.gaps, [
+ { after: "20240115", before: "20240401", days: 77 },
+ { after: "20240401", before: "20250101", days: 275 },
+ ]);
+});
+
+test("a big metadata file is read at both ends: title at the head, upload_date at the tail", async () => {
+ const dir = path.join(ROOT, "big");
+ mkdirSync(dir, { recursive: true });
+ const formats = Array.from({ length: 6000 }, (_, i) => ({ format_id: String(i), url: `https://example.invalid/${"x".repeat(40)}` }));
+ writeFileSync(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id: "big", title: "Stream — März 5 \"live\"", formats, upload_date: "20240306" }),
+ );
+ assert.deepEqual(await readDateAndTitle(dir), { uploadDate: "20240306", title: 'Stream — März 5 "live"' });
+ assert.deepEqual(await readDateAndTitle(path.join(ROOT, "nowhere")), {});
+});
+
+test("a mirror's videos are dated by the title rule; the rest by upload; no metadata is undated", async () => {
+ channel(
+ "vod-mirror",
+ { recordedDate: { titlePattern: "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})" } },
+ {
+ m1: { title: "Stream 2023-05-01", upload_date: "20240110" },
+ m2: { title: "Stream 2023-05-03", upload_date: "20240110" },
+ m3: { title: "No date in this one", upload_date: "20240110" },
+ // A title date after the upload is not a recording date.
+ m4: { title: "Preview of 2030-01-01", upload_date: "20240111" },
+ bare: null,
+ },
+ );
+ const c = await channelCoverage(getPaths(), "vod-mirror", { gapDays: 60, list: true });
+ assert.equal(c.titlePattern, "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})");
+ assert.equal(c.held, 5);
+ assert.equal(c.dated, 4);
+ assert.equal(c.byRecordedDate, 2);
+ assert.equal(c.byUploadDate, 2);
+ assert.deepEqual(c.undated, ["bare"]);
+ assert.equal(c.first, "20230501");
+ assert.deepEqual(c.gaps, [{ after: "20230503", before: "20240110", days: 252 }]);
+ assert.deepEqual(
+ c.videos?.map((v) => [v.id, v.date]),
+ [["m1", "20230501"], ["m2", "20230503"], ["m3", "20240110"], ["m4", "20240111"]],
+ );
+ const window = await channelCoverage(getPaths(), "vod-mirror", { from: "20240101" });
+ assert.deepEqual(window.byYear, { "2024": 2 });
+ assert.equal(window.videos, undefined);
+ await assert.rejects(channelCoverage(getPaths(), "no-such"), /Channel "no-such" not found/);
+});
diff --git a/common/controller/channelCoverage.ts b/common/controller/channelCoverage.ts
@@ -0,0 +1,196 @@
+// A CHANNEL'S COVERAGE: WHAT IT HOLDS, BY DATE, AND WHERE THE GAPS ARE
+// (release 19 A9; the MCP's `channel_coverage`, `GET /api/ops/coverage`).
+//
+// Every held video (a data/<id>/ directory) is dated with `coverageDate`
+// (lib/recordedDate.ts, release 20 D1): its RECORDED date when the channel has
+// a title rule (`config.recordedDate`) and the title yields one, else its
+// upload date. A VOD mirror's coverage is then the streams' days, not the
+// copies'. The dates come off disk, not the index, so a video imported an hour
+// ago counts.
+//
+// CHEAP ON A BIG CHANNEL: metadata.info.json can run to megabytes (formats), so
+// a small file is parsed whole and a big one is read at both ends — `title`
+// sits near the head of yt-dlp's JSON, `upload_date` near the tail (the same
+// tail read recencyIndex.ts measured at 0.19 ms a file). A video with neither
+// is `undated`, never guessed. Nothing is written.
+
+import path from "node:path";
+import { open, readFile, stat } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { mapConcurrent } from "../lib/concurrency";
+import { compileRecordedDateRule, coverageDate, deriveRecordedDate } from "../lib/recordedDate";
+import { readChannelConfig } from "./channels";
+import { heldVideoIds } from "./videoCues";
+
+const WHOLE_FILE_BYTES = 256 * 1024;
+const HEAD_BYTES = 16 * 1024;
+const TAIL_BYTES = 8 * 1024;
+
+export const DEFAULT_GAP_DAYS = 30;
+
+export type DatedVideo = {
+ id: string;
+ date: string; // YYYYMMDD
+ uploadDate: string;
+ recordedDate?: string;
+ title?: string;
+};
+
+export type ChannelCoverage = {
+ slug: string;
+ // The channel's recorded-date title rule, when it has one.
+ titlePattern: string | null;
+ held: number;
+ dated: number;
+ // Dated by the title rule / by the upload date.
+ byRecordedDate: number;
+ byUploadDate: number;
+ undated: string[];
+ first: string | null;
+ last: string | null;
+ // Videos per year and per month ("2024", "2024-03"), only those with any.
+ byYear: Record<string, number>;
+ byMonth: Record<string, number>;
+ // Spans longer than `gapDays` with no held video, oldest first: the last
+ // held date before, the first after, and the days between.
+ gapDays: number;
+ gaps: { after: string; before: string; days: number }[];
+ // With `list`: every dated video in the window, oldest first.
+ videos?: DatedVideo[];
+};
+
+const UPLOAD_RE = /"upload_date"\s*:\s*"(\d{8})"/;
+const TITLE_RE = /"title"\s*:\s*("(?:[^"\\]|\\.)*")/;
+
+function fromText(text: string): { uploadDate?: string; title?: string } {
+ const up = UPLOAD_RE.exec(text)?.[1];
+ const rawTitle = TITLE_RE.exec(text)?.[1];
+ let title: string | undefined;
+ if (rawTitle) {
+ try {
+ title = JSON.parse(rawTitle) as string;
+ } catch {
+ /* a title cut by the read boundary is no title */
+ }
+ }
+ return { ...(up ? { uploadDate: up } : {}), ...(title ? { title } : {}) };
+}
+
+// The upload date and title of one video, from its metadata.info.json.
+export async function readDateAndTitle(videoDir: string): Promise<{ uploadDate?: string; title?: string }> {
+ const file = path.join(videoDir, "metadata.info.json");
+ const st = await stat(file).catch(() => null);
+ if (!st?.isFile()) return {};
+ if (st.size <= WHOLE_FILE_BYTES) {
+ try {
+ const j = JSON.parse(await readFile(file, "utf8")) as { upload_date?: unknown; title?: unknown };
+ return {
+ ...(typeof j.upload_date === "string" && /^\d{8}$/.test(j.upload_date) ? { uploadDate: j.upload_date } : {}),
+ ...(typeof j.title === "string" && j.title ? { title: j.title } : {}),
+ };
+ } catch {
+ return {};
+ }
+ }
+ const fh = await open(file, "r");
+ try {
+ const head = Buffer.alloc(HEAD_BYTES);
+ const tail = Buffer.alloc(TAIL_BYTES);
+ await fh.read(head, 0, HEAD_BYTES, 0);
+ await fh.read(tail, 0, TAIL_BYTES, st.size - TAIL_BYTES);
+ // A multi-byte sequence split by a read boundary decodes as U+FFFD; the
+ // regexes only take a whole quoted string, so a cut title is no title.
+ const h = fromText(head.toString("utf8"));
+ const t = fromText(tail.toString("utf8"));
+ const uploadDate = t.uploadDate ?? h.uploadDate;
+ const title = h.title ?? t.title;
+ return { ...(uploadDate ? { uploadDate } : {}), ...(title ? { title } : {}) };
+ } finally {
+ await fh.close();
+ }
+}
+
+function daysBetween(a: string, b: string): number {
+ const toMs = (d: string) => Date.UTC(Number(d.slice(0, 4)), Number(d.slice(4, 6)) - 1, Number(d.slice(6, 8)));
+ return Math.round((toMs(b) - toMs(a)) / 86_400_000);
+}
+
+// The pure half: dated videos → the summary (byYear, byMonth, gaps).
+export function summarizeCoverage(
+ videos: DatedVideo[],
+ opts: { gapDays?: number } = {},
+): Pick<ChannelCoverage, "first" | "last" | "byYear" | "byMonth" | "gapDays" | "gaps"> {
+ const gapDays = opts.gapDays ?? DEFAULT_GAP_DAYS;
+ const sorted = [...videos].sort((a, b) => a.date.localeCompare(b.date) || a.id.localeCompare(b.id));
+ const byYear: Record<string, number> = {};
+ const byMonth: Record<string, number> = {};
+ const gaps: ChannelCoverage["gaps"] = [];
+ for (let i = 0; i < sorted.length; i++) {
+ const d = sorted[i].date;
+ byYear[d.slice(0, 4)] = (byYear[d.slice(0, 4)] ?? 0) + 1;
+ const m = `${d.slice(0, 4)}-${d.slice(4, 6)}`;
+ byMonth[m] = (byMonth[m] ?? 0) + 1;
+ if (i > 0) {
+ const prev = sorted[i - 1].date;
+ const days = daysBetween(prev, d);
+ if (days > gapDays) gaps.push({ after: prev, before: d, days });
+ }
+ }
+ return {
+ first: sorted[0]?.date ?? null,
+ last: sorted.at(-1)?.date ?? null,
+ byYear,
+ byMonth,
+ gapDays,
+ gaps,
+ };
+}
+
+export async function channelCoverage(
+ paths: Paths,
+ slug: string,
+ opts: { gapDays?: number; from?: string; to?: string; list?: boolean } = {},
+): Promise<ChannelCoverage> {
+ const config = await readChannelConfig(paths, slug);
+ if (!config) throw new Error(`Channel "${slug}" not found`);
+ // The text guard: an unmounted or moving drive is not an empty channel.
+ await assertChannelTextReadable(paths, slug, config);
+ const dataDir = path.join(paths.channelsDir, slug, "data");
+ const ids = [...(await heldVideoIds(dataDir))].sort();
+ const rule = config.recordedDate ? compileRecordedDateRule(config.recordedDate) : null;
+ const read = await mapConcurrent(ids, 32, async (id) => ({ id, ...(await readDateAndTitle(path.join(dataDir, id))) }));
+ const dated: DatedVideo[] = [];
+ const undated: string[] = [];
+ let byRecordedDate = 0;
+ for (const r of read) {
+ if (!r.uploadDate) {
+ undated.push(r.id);
+ continue;
+ }
+ const recordedDate = rule && r.title ? deriveRecordedDate(rule, r.title, r.uploadDate) : undefined;
+ if (recordedDate) byRecordedDate++;
+ dated.push({
+ id: r.id,
+ date: coverageDate({ uploadDate: r.uploadDate, ...(recordedDate ? { recordedDate } : {}) }),
+ uploadDate: r.uploadDate,
+ ...(recordedDate ? { recordedDate } : {}),
+ ...(r.title ? { title: r.title } : {}),
+ });
+ }
+ const inWindow = dated.filter((v) => (!opts.from || v.date >= opts.from) && (!opts.to || v.date <= opts.to));
+ const summary = summarizeCoverage(inWindow, { gapDays: opts.gapDays });
+ return {
+ slug,
+ titlePattern: config.recordedDate?.titlePattern ?? null,
+ held: ids.length,
+ dated: dated.length,
+ byRecordedDate,
+ byUploadDate: dated.length - byRecordedDate,
+ undated,
+ ...summary,
+ ...(opts.list
+ ? { videos: [...inWindow].sort((a, b) => a.date.localeCompare(b.date) || a.id.localeCompare(b.id)) }
+ : {}),
+ };
+}
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -89,7 +89,7 @@ const DECLARED: EnvVarDecl[] = [
paths("STAGIT_BIN", "`stagit` on PATH, then `~/.local/bin/stagit`", "stagit, which renders the source's history pages (`/source/git/`: the log and a page per commit with its diff). Optional: without it the source is published without them. See [PUBLISH.md](PUBLISH.md)."),
// ── runtime ────────────────────────────────────────────────────────────
- { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`, the MCP's `fetch_clip`). Set the same value on both ends." },
+ { name: "WORKER_TOKEN", audience: "runtime", default: "unset (both surfaces off)", readBy: "common/lib/workerToken.ts, scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts", doc: "Bearer token for the remote-worker API and for `/api/ops/*` (`pnpm ops`; the MCP's `fetch_clip`, `enqueue`, `get_job`, `channel_coverage` and `get_transcript`'s editor fallback). Set the same value on both ends." },
{ name: "SYNC_HEARTBEAT_SECONDS", audience: "runtime", default: "`settings.syncScheduler.heartbeatSeconds`", readBy: "editor/app/scheduler/heartbeat.ts", doc: "Overrides the editor's in-process sync heartbeat. `0` = no internal timer (tick from cron instead)." },
{ name: "SYNC_TICK_URL", audience: "runtime", default: "`http://127.0.0.1:3001/api/scheduler/tick`", readBy: "common/bin/sync-tick.ts", doc: "Where `archilyzer sync tick` (cron's heartbeat) posts." },
{ name: "SYNC_TICK_TOKEN", audience: "runtime", default: "unset (no auth)", readBy: "common/bin/sync-tick.ts, editor/app/scheduler/auth.ts", doc: "Bearer token for the tick endpoint; set on both the editor and the cron job." },
@@ -118,7 +118,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." },
{ name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." },
{ name: "ARCHILYZER_X_BROWSER", audience: "runtime", default: "the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium", readBy: "common/social/xBrowser.ts", doc: "The Chromium-family browser /settings' \"Connect X account\" opens (a path, or a name looked up on PATH), and a forum-thread channel's \"Connect forum session\" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser." },
- { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." },
+ { name: "UMTOOL_URL", audience: "runtime", default: "unset (no link; the MCP's `notes` off)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx, mcp/src/archivalTools.ts", doc: "umtool's front door; when set, the video page links to it, and the MCP's `notes` tool reads the operator's notes there (e.g. `http://localhost:3050`)." },
{ name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." },
{ name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." },
{ name: "TRANSCRIPT_LOCAL_DIR", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a composed public dir on disk." },
@@ -129,7 +129,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "ARCHILYZER_INDEX_ALLOW_HELD", audience: "runtime", default: "off", readBy: "common/controller/buildIndex.ts", doc: "`1` lets a FULL index rebuild (a schema change, or no index yet) proceed while a channel's media cannot be read; that channel stays out of the index until its media is back and the index is built again. Unset, such a build refuses and names each channel." },
{ name: "UV_THREADPOOL_SIZE", audience: "runtime", default: "`16` for the editor (`4` is Node's own)", readBy: "Node's libuv (set by editor/package.json `start` and docker/entrypoint.sh)", doc: "Threads in Node's pool for filesystem calls. A call on a stalled drive holds one until the drive answers, so the editor starts with 16. It buys time for calls already in flight and isolates nothing: the storage health probe and its gate keep new calls off a stalled drive." },
{ name: "MCP_IO_STATS", audience: "runtime", default: "off", readBy: "common/lib/archive/io-stats.ts", doc: "`1` turns on per-call I/O accounting, for `mcp/bench`." },
- { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, umtool, common/bin/migrate-media-tier.ts", doc: "Which editor `pnpm ops` and the MCP's `fetch_clip` talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it." },
+ { name: "ARCHILYZER_EDITOR_URL", audience: "runtime", default: "`http://localhost:3001`", readBy: "scripts/archilyzer-ops.mjs, mcp/src/fetchClip.ts, mcp/src/editorOps.ts, umtool, common/bin/migrate-media-tier.ts", doc: "Which editor `pnpm ops` and the MCP's editor-backed tools (`fetch_clip`, `enqueue`, `get_job`, `channel_coverage`, `get_transcript`'s fallback) talk to, and whose `/api/pulse` `archilyzer storage migrate-tier` asks before it refuses to run beside it." },
{ name: "ARCHILYZER_AGENT", audience: "runtime", default: "`cli`", readBy: "scripts/archilyzer-ops.mjs", doc: "Who is asking, recorded as the provenance of a curated-tag write through `pnpm ops`." },
{ name: "DIARIZE_ENGINE_KIND", audience: "runtime", default: "`sherpa-onnx`", readBy: "scripts/diarize.mjs", doc: "The diarization engine: `sherpa-onnx` or `sortformer`." },
{ name: "DIARIZE_ENGINE_CMD", audience: "runtime", default: "the bundled sherpa script", readBy: "scripts/diarize.mjs", doc: "The engine command the wrapper runs." },
diff --git a/editor/app/api/ops/coverage/route.test.ts b/editor/app/api/ops/coverage/route.test.ts
@@ -0,0 +1,49 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { setupOpsCorpus, callGet } from "../_testCorpus";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/api/ops/coverage/route.test.ts"
+//
+// GET /api/ops/coverage (release 19 A9, the MCP's channel_coverage): the query
+// refusals and one channel's coverage off a temp corpus. The dating rules are
+// common/controller/channelCoverage.test.ts.
+
+const corpus = await setupOpsCorpus(null);
+await corpus.writeChannelConfig("demo-yt", { url: "https://www.youtube.com/@demo" });
+for (const [id, date] of [["v1", "20240101"], ["v2", "20240301"]]) {
+ const dir = path.join(corpus.transcripts, "channels", "demo-yt", "data", id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify({ id, title: id, upload_date: date }));
+}
+const { GET } = await import("./route");
+test.after(() => corpus.cleanup());
+const get = (q: string) => callGet(GET, `http://localhost/api/ops/coverage?${q}`);
+
+test("coverage refusals: slug, gapDays, dates, list, unknown keys, the token", async () => {
+ const cases: [string, RegExp][] = [
+ ["", /"slug" is required/],
+ ["slug=../x", /not a valid channel slug/],
+ ["slug=demo-yt&gapDays=0", /"gapDays" must be a whole number/],
+ ["slug=demo-yt&from=2024-01-01", /"from" must be a date YYYYMMDD/],
+ ["slug=demo-yt&list=yes", /"list" is 1 or absent/],
+ ["slug=demo-yt&channel=x", /unknown query key\(s\): channel/],
+ ];
+ for (const [q, re] of cases) {
+ const r = await get(q);
+ assert.equal(r.status, 400, q);
+ assert.match(r.body.error ?? "", re, q);
+ }
+ assert.equal((await get("slug=no-such")).status, 404);
+ assert.equal((await callGet(GET, "http://localhost/api/ops/coverage?slug=demo-yt", {}, {})).status, 401);
+});
+
+test("coverage answers the held videos by date and the gaps", async () => {
+ const r = await get("slug=demo-yt&gapDays=30&list=1");
+ assert.equal(r.status, 200, JSON.stringify(r.body));
+ assert.equal(r.body.held, 2);
+ assert.deepEqual(r.body.gaps, [{ after: "20240101", before: "20240301", days: 60 }]);
+ assert.equal((r.body.videos as unknown[]).length, 2);
+});
diff --git a/editor/app/api/ops/coverage/route.ts b/editor/app/api/ops/coverage/route.ts
@@ -0,0 +1,53 @@
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { isValidChannelSlug } from "yt-dlp-transcript-common/controller/channels";
+import { channelCoverage } from "yt-dlp-transcript-common/controller/channelCoverage";
+import { readRoute } from "../_read";
+import { opsFail } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// GET ?slug=<channel>[&gapDays=<n>][&from=YYYYMMDD][&to=YYYYMMDD][&list=1]
+// -> { ok, slug, titlePattern, held, dated, byRecordedDate, byUploadDate,
+// undated: [id], first, last, byYear, byMonth, gapDays,
+// gaps: [{ after, before, days }], videos?: [{ id, date, uploadDate,
+// recordedDate?, title? }] }
+//
+// What a channel holds, by date, and the spans with nothing held
+// (controller/channelCoverage.ts): each held video dated with coverageDate —
+// its recorded date when the channel has a title rule, else its upload date —
+// off disk, so a video imported an hour ago counts. `gapDays` (30) is the
+// shortest span reported as a gap; `from`/`to` narrow the window; `list`
+// adds every dated video. Nothing is written. The MCP's channel_coverage.
+const DAY = /^\d{8}$/;
+
+export async function GET(request: Request) {
+ return readRoute(request, ["slug", "gapDays", "from", "to", "list"], async (q) => {
+ const slug = q.get("slug")?.trim() ?? "";
+ if (!slug) return opsFail('"slug" is required');
+ if (!isValidChannelSlug(slug)) return opsFail(`"${slug}" is not a valid channel slug`);
+ const rawGap = q.get("gapDays");
+ const gapDays = rawGap === null ? undefined : Number(rawGap);
+ if (gapDays !== undefined && (!Number.isInteger(gapDays) || gapDays < 1)) {
+ return opsFail('"gapDays" must be a whole number of days above zero');
+ }
+ const from = q.get("from") ?? undefined;
+ const to = q.get("to") ?? undefined;
+ for (const [k, v] of [["from", from], ["to", to]] as const) {
+ if (v !== undefined && !DAY.test(v)) return opsFail(`"${k}" must be a date YYYYMMDD`);
+ }
+ const list = q.get("list");
+ if (list !== null && list !== "1" && list !== "true") return opsFail('"list" is 1 or absent');
+ try {
+ const coverage = await channelCoverage(getPaths(), slug, {
+ ...(gapDays !== undefined ? { gapDays } : {}),
+ ...(from ? { from } : {}),
+ ...(to ? { to } : {}),
+ list: list !== null,
+ });
+ return { ok: true, ...coverage };
+ } catch (e) {
+ const message = (e as Error).message;
+ return opsFail(message, /not found/.test(message) ? 404 : 503);
+ }
+ });
+}
diff --git a/mcp/README.md b/mcp/README.md
@@ -7,8 +7,13 @@ Desktop, Cursor, and any other MCP client.
It is a **local tool you run yourself**. It does not change the archive: it only
reads the site's already-published static JSON shards (`corpus.json` +
`transcripts/<slug>/…`), either from disk or over HTTP. Nothing is hosted for you.
-The one exception is `fetch_clip`, which asks a local Archilyzer editor to fetch a
-clip window; the MCP itself still writes nothing.
+The exceptions all ask a local Archilyzer editor (`ARCHILYZER_EDITOR_URL` +
+`WORKER_TOKEN`, the editor's own) to do the work — `fetch_clip` fetches a clip
+window, `enqueue` queues archival work, `get_job` / `channel_coverage` read the
+editor's jobs and disk, and `get_transcript` falls back to the editor's disk for a
+video not yet published; `notes` reads umtool (`UMTOOL_URL`). The MCP itself still
+writes nothing. Settings, storage and deletes are not reachable from it (`pnpm ops`
+only).
## Tools
@@ -20,10 +25,14 @@ clip window; the MCP itself still writes nothing.
| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware, pageable, and **filterable** (`states`, `date_from`/`date_to`, `media_type`, `age`, `exclude`, `scopes`). A page that isn't the whole match set is flagged **above** the hits. |
| `enumerate_matches` | A query's **complete** match set as a worklist (id/title/channel/date + batch count) in **one scan**. Takes the **same filters** as `search_transcripts`, so the two can never disagree about coverage. The tool to use whenever you need to count or cover everything. |
| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around one query or up to 8 (`queries`), with per-query counts; or full transcripts without a query. Reads posts too. |
-| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). |
+| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). A video the archive does not hold yet (imported or transcribed since the last build) is read off the **local editor's disk** when one is configured (`GET /api/ops/transcript`: a fresh `transcript.cues.json`, else the raw transcript normalized in memory, else the English VTT), marked as such, with no moment links. |
| `get_post` / `get_thread` | One archived social post, or its whole thread. Posts have no timeline — cite them with no `@ mm:ss`. |
| `get_video_metadata` | Everything known about one video without the transcript body: metadata, plus **view/like counts, cue count and transcript coverage** (`stats/`), **other archived copies of the same recording** with an explicit timings-aligned verdict (`duplicates.json`), and **AI chapters/tags** where they exist (`digests/`). |
| `fetch_clip` | The media behind a cited moment, **fetched by the local editor** (`POST /api/media/fetch-window`) through its paced, cookie-aware, provenanced job — never a yt-dlp run by hand. Needs `ARCHILYZER_EDITOR_URL` (default `http://localhost:3001`) and `WORKER_TOKEN` (the editor's own) in this server's env; without them it says so and fetches nothing. The editor must already archive the cited channel (a channel dir under its `transcripts/`), else it answers 404 `Channel "<slug>" not found`: an MCP pointed at a public site with a fresh editor gets that on every clip. A window is the cited span ± `pad` (default 3 s), at most 15 min, and lands at `channels/<slug>/data/<id>/clips/`; `full: true` fetches the whole recording into the saved-video store (needs a video the editor already knows). `maxHeight` (144–2160) caps the source height: a window is fetched at or under it (default 720); a whole recording at 720 or less is saved as the editor's 720p H.264 preset and above 720 at the original quality (omitted, the channel's source-video quality applies). A file already on disk is returned as it is, never re-fetched for a different cap, and the answer gives its height and says when it is taller than asked. Waits up to `wait_seconds` (default 90, max 300), then returns the job id to resume with `job`; a client with a 60 s default request timeout must raise it or pass `wait_seconds` ≤ 50 — the fetch continues on the editor either way; resume it with `job`, and once it has finished the same request finds it cached. While it waits it sends one progress notification per poll to a client that asked for progress (a `progressToken`), which keeps a reset-on-progress timeout alive. A Rumble embed id is mapped to the editor's slug id through the record's `webpageUrl`, so pass the citing corpus as `source`; a video not in `source` is passed through as cited (known limitation). The file is a read-only corpus artifact. |
+| `get_job` | One editor job's state — kind, channel, status, times, exit code, where it waits in its queue — and the last `tail` (default 40) log lines (`GET /api/ops/job/<id>`). For the job id `enqueue` or `fetch_clip` returned. Needs the editor. |
+| `enqueue` | Queue archival work on a channel the editor archives, as its page's buttons do (`POST /api/ops/<kind>`): `sync` (`full`), `download-missing`, `retry-bucket` (`bucket`, `ids` — the way to download chosen videos), `transcribe-bucket` (`ids`), `fetch-posts` (`full` / `older`), `import-video` (`url`). Answers the job id; the editor runs it on its own queues at each platform's pace. Only when the operator asks. Needs the editor. |
+| `channel_coverage` | What a channel **holds** on the editor's disk by date (`GET /api/ops/coverage`): count, first and last day, per year, and every gap longer than `gap_days` (30); `date_from` / `date_to` narrow it, `list` lists the videos. Each video is dated by its **recorded date** when the channel has a title rule (a VOD mirror, release 20), else its upload date. Needs the editor. |
+| `notes` | The operator's notes in umtool on articles and report-video projects: `list` (every open note, by target), `read` (`target` as `list` names it — the digest with anchors resolved and the source file to edit). `reply` answers with the `umtool notes reply` command, because umtool records a reply sent over HTTP as the operator's. Needs `UMTOOL_URL`. |
| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity) — plan, results and corpus handle in **one** call. `dry_run:true` for the plan alone. |
| `list_sources` | Show the **default** corpus and, with a hub, its member sites as ready-to-paste handles. A site with reports says how many; a cited-only site says it is one. |
| `resolve_source` | Turn a URL or site name into the canonical `source` handle and check it can be read. Changes nothing. |
@@ -496,8 +505,8 @@ nothing to reset).
**Still read-only.** A `source` handle only changes *which* already-published
static shards are read — the same capability the startup flags already grant
-this locally-run tool. This server never writes to any corpus; `fetch_clip` asks
-the editor, and the editor writes.
+this locally-run tool. This server never writes to any corpus; `fetch_clip` and
+`enqueue` ask the editor, and the editor writes.
## Protocol
diff --git a/mcp/src/archivalTools.test.ts b/mcp/src/archivalTools.test.ts
@@ -0,0 +1,204 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { connectWith, fakeEditor, isError, textOf } from "./editorOps.testkit";
+import { enqueueBody, notesTarget, notesTool } from "./archivalTools";
+
+// ─── The archival tools (release 19 A9), through tools/call ───
+//
+// Each maps to one ops route on the fake editor (or umtool's notes routes);
+// what is pinned is the request sent and the words the agent reads.
+
+const EDITOR = "http://editor.test";
+
+test("enqueue: each kind posts to its own route with only the keys it takes", () => {
+ assert.deepEqual(enqueueBody({ kind: "sync", channel: "demo", full: true }), {
+ ok: true,
+ kind: "sync",
+ route: "/api/ops/sync",
+ body: { slug: "demo", full: true },
+ });
+ assert.deepEqual(enqueueBody({ kind: "retry-bucket", channel: "demo", bucket: "noTranscript", ids: ["a1"] }), {
+ ok: true,
+ kind: "retry-bucket",
+ route: "/api/ops/retry-bucket",
+ body: { slug: "demo", bucket: "noTranscript", ids: ["a1"] },
+ });
+ assert.deepEqual(enqueueBody({ kind: "import-video", channel: "demo", url: "https://example.invalid/v/1" }), {
+ ok: true,
+ kind: "import-video",
+ route: "/api/ops/import-video",
+ body: { slug: "demo", url: "https://example.invalid/v/1" },
+ });
+ const refusals: [Record<string, unknown>, RegExp][] = [
+ [{ kind: "delete-channel", channel: "demo" }, /kind must be one of sync, download-missing/],
+ [{ kind: "sync" }, /channel is required/],
+ [{ kind: "sync", channel: "../x" }, /channel is required/],
+ [{ kind: "download-missing", channel: "demo", ids: ["a"] }, /download-missing does not take ids$/],
+ [{ kind: "sync", channel: "demo", url: "https://x" }, /sync does not take url \(it takes full\)/],
+ [{ kind: "retry-bucket", channel: "demo" }, /retry-bucket needs bucket/],
+ [{ kind: "transcribe-bucket", channel: "demo", ids: ["../a"] }, /ids must be a non-empty list of video ids/],
+ [{ kind: "import-video", channel: "demo", url: "ftp://x" }, /import-video needs url/],
+ [{ kind: "fetch-posts", channel: "demo", older: "yes" }, /older must be true or false/],
+ ];
+ for (const [args, re] of refusals) {
+ const r = enqueueBody(args);
+ assert.equal(r.ok, false, JSON.stringify(args));
+ assert.match(!r.ok ? r.error : "", re, JSON.stringify(args));
+ }
+});
+
+test("enqueue through tools/call: the job id comes back with how to follow it", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.endsWith("/api/ops/transcribe-bucket")
+ ? { status: 200, body: { ok: true, jobId: "j-77" } }
+ : { status: 400, body: { ok: false, error: 'Channel "nope" not found' } },
+ );
+ const client = await connectWith(deps);
+ const ok = await client.callTool({ name: "enqueue", arguments: { kind: "transcribe-bucket", channel: "demo" } });
+ assert.equal(isError(ok), false, textOf(ok));
+ assert.equal(sent[0].url, `${EDITOR}/api/ops/transcribe-bucket`);
+ assert.equal(sent[0].method, "POST");
+ assert.deepEqual(sent[0].body, { slug: "demo" });
+ assert.match(textOf(ok), /queued as job j-77.*follow it with get_job \(job: "j-77"\)/s);
+ const refused = await client.callTool({ name: "enqueue", arguments: { kind: "sync", channel: "nope" } });
+ assert.equal(isError(refused), true);
+ assert.match(textOf(refused), /enqueue sync: the editor refused \(HTTP 400\): Channel "nope" not found/);
+ const none = await (await connectWith(fakeEditor(() => ({ status: 200, body: {} }), {}).deps)).callTool({
+ name: "enqueue",
+ arguments: { kind: "sync", channel: "demo" },
+ });
+ assert.match(textOf(none), /no editor configured/);
+});
+
+test("get_job: status, queue place and the log tail", async () => {
+ const { deps, sent } = fakeEditor((s) =>
+ s.url.includes("/api/ops/job/j-1")
+ ? {
+ status: 200,
+ body: {
+ ok: true,
+ job: {
+ id: "j-1",
+ kind: "sync",
+ channelSlug: "demo",
+ status: "queued",
+ queuedAt: Date.UTC(2026, 9, 10),
+ queue: { key: "youtube", position: 2, queued: 3, head: { id: "j-0", kind: "persist-videos", channelSlug: "other" } },
+ },
+ tail: ["line one", "line two"],
+ },
+ }
+ : { status: 404, body: { ok: false, error: "no such job" } },
+ );
+ const client = await connectWith(deps);
+ const res = await client.callTool({ name: "get_job", arguments: { job: "j-1", tail: 2 } });
+ assert.equal(new URL(sent[0].url).pathname, "/api/ops/job/j-1");
+ assert.equal(new URL(sent[0].url).searchParams.get("tail"), "2");
+ const t = textOf(res);
+ assert.match(t, /job j-1: sync on demo — \*\*queued\*\*/);
+ assert.match(t, /waiting on queue "youtube": position 2 of 3, behind persist-videos on other \(j-0\)/);
+ assert.match(t, /Still going: call get_job again later/);
+ assert.match(t, /```\nline one\nline two\n```/);
+ const gone = await client.callTool({ name: "get_job", arguments: { job: "j-9" } });
+ assert.match(textOf(gone), /the editor knows no job j-9/);
+ const bad = await client.callTool({ name: "get_job", arguments: { job: "j-1", tail: -1 } });
+ assert.match(textOf(bad), /tail is a whole number/);
+});
+
+test("channel_coverage: the query sent, and the gaps read back", async () => {
+ const { deps, sent } = fakeEditor(() => ({
+ status: 200,
+ body: {
+ ok: true,
+ slug: "vod-mirror",
+ titlePattern: "(?<year>\\d{4})-(?<month>\\d{2})-(?<day>\\d{2})",
+ held: 3,
+ dated: 2,
+ byRecordedDate: 2,
+ byUploadDate: 0,
+ undated: ["bare"],
+ first: "20230501",
+ last: "20240110",
+ byYear: { "2023": 1, "2024": 1 },
+ byMonth: {},
+ gapDays: 60,
+ gaps: [{ after: "20230501", before: "20240110", days: 254 }],
+ },
+ }));
+ const client = await connectWith(deps);
+ const res = await client.callTool({
+ name: "channel_coverage",
+ arguments: { channel: "vod-mirror", gap_days: 60, date_from: "2023-01-01", list: true },
+ });
+ const q = new URL(sent[0].url).searchParams;
+ assert.equal(new URL(sent[0].url).pathname, "/api/ops/coverage");
+ assert.equal(q.get("slug"), "vod-mirror");
+ assert.equal(q.get("gapDays"), "60");
+ assert.equal(q.get("from"), "20230101");
+ assert.equal(q.get("list"), "1");
+ const t = textOf(res);
+ assert.match(t, /3 held · 2 dated \(2 by recorded date, 0 by upload date\) · 1 undated · 2023-05-01 → 2024-01-10/);
+ assert.match(t, /the recorded date read off each title/);
+ assert.match(t, /- 2023-05-01 → 2024-01-10: 254 days with nothing held/);
+ const bad = await client.callTool({ name: "channel_coverage", arguments: { channel: "x", date_to: "Jan 2024" } });
+ assert.match(textOf(bad), /date_to is a date, YYYY-MM-DD/);
+});
+
+test("notes: list groups open notes by target; read fetches the digest; reply is the CLI command", async () => {
+ const umtool = { UMTOOL_URL: "http://umtool.test/" };
+ const { deps, sent } = fakeEditor((s) => {
+ if (s.url.includes("/api/browse/decisions")) {
+ return {
+ status: 200,
+ body: {
+ items: [
+ { kind: "open-note", project: "sites/demo-site/a-report", projectKind: "article", target: "cite c1", why: "say claimed" },
+ { kind: "cut-without-file", project: "song", target: "x", why: "y" },
+ { kind: "open-note", project: "demo/video-project", projectKind: "report-video", target: "take t2", why: "too long" },
+ ],
+ },
+ };
+ }
+ return { status: 200, body: "unused" };
+ }, umtool);
+ const client = await connectWith(deps);
+ const list = await client.callTool({ name: "notes", arguments: {} });
+ assert.equal(sent[0].url, "http://umtool.test/api/browse/decisions");
+ const t = textOf(list);
+ assert.match(t, /2 open note\(s\) on 2 target\(s\)/);
+ assert.match(t, /## sites\/demo-site\/a-report \(article\)/);
+ assert.match(t, /- \[take t2\] too long/);
+ assert.doesNotMatch(t, /song/);
+
+ assert.deepEqual(notesTarget("sites/demo-site/a-report"), { param: "article", value: "demo-site/a-report" });
+ assert.deepEqual(notesTarget("demo/video-project"), { param: "project", value: "demo/video-project" });
+ assert.equal(notesTarget("sites/../x"), null);
+
+ const digestSent: string[] = [];
+ const read = await notesTool(
+ { action: "read", target: "sites/demo-site/a-report", status: "all" },
+ {
+ env: umtool,
+ fetch: async (url) => {
+ digestSent.push(url);
+ return { status: 200, json: async () => ({}), text: async () => "# Notes on demo-site/a-report\n…" } as never;
+ },
+ },
+ );
+ const q = new URL(digestSent[0]).searchParams;
+ assert.equal(new URL(digestSent[0]).pathname, "/api/notes/context");
+ assert.equal(q.get("article"), "demo-site/a-report");
+ assert.equal(q.get("status"), "all");
+ assert.match(read.text, /^# Notes on demo-site\/a-report/);
+
+ const reply = await client.callTool({
+ name: "notes",
+ arguments: { action: "reply", note_id: "n_abc", text: "Changed it", resolve: true },
+ });
+ assert.equal(isError(reply), true);
+ assert.match(textOf(reply), /umtool notes reply n_abc "Changed it" --resolve/);
+ assert.equal(sent.length, 1, "a reply sends nothing");
+
+ const none = await notesTool({ action: "list" }, { env: {}, fetch: deps.fetch });
+ assert.match(none.text, /no umtool configured\. Set UMTOOL_URL/);
+});
diff --git a/mcp/src/archivalTools.ts b/mcp/src/archivalTools.ts
@@ -0,0 +1,479 @@
+import { describeEditorFailure, editorGet, editorPost, type EditorDeps } from "./editorOps";
+import { describeFetchError, isVideoId, REQUEST_TIMEOUT_MS } from "./fetchClip";
+
+// ─── Archival writes through the editor (release 19 A9) ───
+//
+// The fetch_clip pattern: this server asks the LOCAL editor (its URL and its
+// WORKER_TOKEN) and the editor runs the job, with its queues, pacing, holds
+// and provenance. The MCP writes nothing itself. Each tool maps to ONE
+// existing ops route and nothing else:
+//
+// get_job GET /api/ops/job/<id>?tail=N
+// enqueue POST /api/ops/{sync | download-missing | retry-bucket |
+// transcribe-bucket | fetch-posts | import-video}
+// channel_coverage GET /api/ops/coverage?slug=…
+// notes umtool (UMTOOL_URL): GET /api/browse/decisions (list),
+// GET /api/notes/context (read)
+//
+// Settings, storage and deletes are deliberately absent (operator ruling,
+// release 19): `pnpm ops` / `archilyzer` only. A notes REPLY is not written
+// from here: umtool stamps every write through its HTTP route as the
+// OPERATOR's ("there is no way to claim to be one here", umtool's
+// app/api/notes/route.ts), so an agent reply must go through `umtool notes
+// reply`, which stamps "agent" — the tool answers with that command.
+//
+// Pure apart from the injected deps; server.ts adapts.
+
+export type ToolAnswer = { text: string; isError?: boolean };
+
+const SLUG_RE = /^[a-z0-9][a-z0-9._-]*$/i;
+const trimmed = (v: unknown) => (typeof v === "string" ? v.trim() : "");
+
+// ─── get_job ───
+
+export const DEFAULT_JOB_TAIL = 40;
+export const MAX_JOB_TAIL = 500;
+
+type JobView = {
+ id: string;
+ kind?: string;
+ channelSlug?: string;
+ videoId?: string;
+ status: string;
+ queuedAt?: number;
+ startedAt?: number;
+ endedAt?: number;
+ exitCode?: number;
+ detail?: string;
+ cancelReason?: string;
+ queue?: { key: string; position: number; queued: number; head?: { id: string; kind: string; channelSlug?: string } };
+ progress?: unknown;
+};
+
+const iso = (ms?: number) => (typeof ms === "number" && ms > 0 ? new Date(ms).toISOString() : null);
+
+export function renderJob(job: JobView, tail?: string[]): string {
+ const lines: string[] = [];
+ lines.push(
+ `job ${job.id}: ${job.kind ?? "job"}${job.channelSlug ? ` on ${job.channelSlug}` : ""}${job.videoId ? `/${job.videoId}` : ""} — **${job.status}**`,
+ );
+ if (job.detail) lines.push(`for: ${job.detail}`);
+ const times = [
+ iso(job.queuedAt) ? `queued ${iso(job.queuedAt)}` : "",
+ iso(job.startedAt) ? `started ${iso(job.startedAt)}` : "",
+ iso(job.endedAt) ? `ended ${iso(job.endedAt)}` : "",
+ ].filter(Boolean);
+ if (times.length) lines.push(times.join(" · "));
+ if (typeof job.exitCode === "number") lines.push(`exit code ${job.exitCode}`);
+ if (job.cancelReason) lines.push(`cancelled because: ${job.cancelReason}`);
+ if (job.queue) {
+ const q = job.queue;
+ lines.push(
+ q.position === 0
+ ? `running at the head of queue "${q.key}" (${q.queued} waiting behind)`
+ : `waiting on queue "${q.key}": position ${q.position} of ${q.queued}` +
+ (q.head ? `, behind ${q.head.kind}${q.head.channelSlug ? ` on ${q.head.channelSlug}` : ""} (${q.head.id})` : ""),
+ );
+ }
+ if (job.progress !== undefined) lines.push(`progress: ${JSON.stringify(job.progress)}`);
+ if (job.status === "queued" || job.status === "running") {
+ lines.push("Still going: call get_job again later (a platform queue may hold a job for hours).");
+ }
+ if (tail && tail.length) {
+ lines.push("", `last ${tail.length} log line(s):`, "```", ...tail, "```");
+ }
+ return lines.join("\n");
+}
+
+export async function getJob(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const id = trimmed(args.job);
+ if (!id || !/^[\w.-]+$/.test(id)) return { text: "get_job: job is required (the job id an enqueue or fetch_clip returned)", isError: true };
+ const rawTail = args.tail === undefined ? DEFAULT_JOB_TAIL : Number(args.tail);
+ if (!Number.isInteger(rawTail) || rawTail < 0) return { text: "get_job: tail is a whole number of lines, 0 or more", isError: true };
+ const tail = Math.min(MAX_JOB_TAIL, rawTail);
+ const a = await editorGet(deps, `/api/ops/job/${encodeURIComponent(id)}${tail > 0 ? `?tail=${tail}` : ""}`);
+ if (a.kind === "refused" && a.status === 404) return { text: `get_job: the editor knows no job ${id}`, isError: true };
+ if (a.kind !== "ok") return { text: describeEditorFailure("get_job", a), isError: true };
+ const job = a.body.job as JobView | undefined;
+ if (!job || typeof job.status !== "string") return { text: "get_job: the editor's answer carried no job", isError: true };
+ const lines = Array.isArray(a.body.tail) ? (a.body.tail as unknown[]).map(String) : undefined;
+ return { text: renderJob(job, lines) };
+}
+
+// ─── enqueue ───
+
+export const ENQUEUE_KINDS = [
+ "sync",
+ "download-missing",
+ "retry-bucket",
+ "transcribe-bucket",
+ "fetch-posts",
+ "import-video",
+] as const;
+export type EnqueueKind = (typeof ENQUEUE_KINDS)[number];
+
+// Which tool arguments each kind takes, beyond `channel`. Anything else is
+// refused by name — the ops routes refuse unknown keys too, but saying it here
+// costs no request.
+const KIND_ARGS: Record<EnqueueKind, readonly string[]> = {
+ sync: ["full"],
+ "download-missing": [],
+ "retry-bucket": ["bucket", "ids"],
+ "transcribe-bucket": ["ids"],
+ "fetch-posts": ["full", "older"],
+ "import-video": ["url"],
+};
+const ALL_ARGS = ["full", "older", "bucket", "ids", "url"];
+
+export function enqueueBody(
+ args: Record<string, unknown>,
+): { ok: true; kind: EnqueueKind; route: string; body: Record<string, unknown> } | { ok: false; error: string } {
+ const kind = trimmed(args.kind) as EnqueueKind;
+ if (!ENQUEUE_KINDS.includes(kind)) {
+ return { ok: false, error: `enqueue: kind must be one of ${ENQUEUE_KINDS.join(", ")}` };
+ }
+ const channel = trimmed(args.channel);
+ if (!channel || !SLUG_RE.test(channel)) return { ok: false, error: "enqueue: channel is required (the channel slug)" };
+ const allowed = KIND_ARGS[kind];
+ const stray = ALL_ARGS.filter((k) => args[k] !== undefined && !allowed.includes(k));
+ if (stray.length) {
+ return {
+ ok: false,
+ error: `enqueue: ${kind} does not take ${stray.join(", ")}${allowed.length ? ` (it takes ${allowed.join(", ")})` : ""}`,
+ };
+ }
+ const body: Record<string, unknown> = { slug: channel };
+ for (const k of ["full", "older"] as const) {
+ if (args[k] === undefined) continue;
+ if (typeof args[k] !== "boolean") return { ok: false, error: `enqueue: ${k} must be true or false` };
+ if (args[k]) body[k] = true;
+ }
+ if (args.ids !== undefined) {
+ const ids = args.ids;
+ if (!Array.isArray(ids) || ids.length === 0 || ids.some((i) => typeof i !== "string" || !isVideoId(i))) {
+ return { ok: false, error: "enqueue: ids must be a non-empty list of video ids" };
+ }
+ body.ids = ids;
+ }
+ if (kind === "retry-bucket") {
+ const bucket = trimmed(args.bucket);
+ if (!bucket) return { ok: false, error: "enqueue: retry-bucket needs bucket (a bucket of the channel's report, e.g. noTranscript)" };
+ body.bucket = bucket;
+ }
+ if (kind === "import-video") {
+ const url = trimmed(args.url);
+ if (!/^https?:\/\//i.test(url)) return { ok: false, error: "enqueue: import-video needs url (the video's page)" };
+ body.url = url;
+ }
+ return { ok: true, kind, route: `/api/ops/${kind}`, body };
+}
+
+export async function enqueue(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const plan = enqueueBody(args);
+ if (!plan.ok) return { text: plan.error, isError: true };
+ const a = await editorPost(deps, plan.route, plan.body);
+ if (a.kind !== "ok") return { text: describeEditorFailure(`enqueue ${plan.kind}`, a), isError: true };
+ const ids = Array.isArray(a.body.jobIds)
+ ? (a.body.jobIds as unknown[]).map(String)
+ : typeof a.body.jobId === "string"
+ ? [a.body.jobId]
+ : [];
+ if (ids.length === 0) {
+ return { text: `enqueue ${plan.kind}: the editor accepted it and started no job (${JSON.stringify(a.body)})` };
+ }
+ return {
+ text:
+ `enqueue ${plan.kind} on ${plan.body.slug}: queued as ${ids.map((i) => `job ${i}`).join(", ")}. ` +
+ `It runs on the editor's queue at the platform's pace and may wait behind other work; ` +
+ `follow it with get_job (job: "${ids[0]}").`,
+ };
+}
+
+// ─── channel_coverage ───
+
+type Coverage = {
+ slug: string;
+ titlePattern: string | null;
+ held: number;
+ dated: number;
+ byRecordedDate: number;
+ byUploadDate: number;
+ undated: string[];
+ first: string | null;
+ last: string | null;
+ byYear: Record<string, number>;
+ byMonth: Record<string, number>;
+ gapDays: number;
+ gaps: { after: string; before: string; days: number }[];
+ videos?: { id: string; date: string; title?: string; recordedDate?: string }[];
+};
+
+const day = (d: string | null) => (d && /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : "—");
+const DATE_ARG = /^(\d{4})-?(\d{2})-?(\d{2})$/;
+
+export function renderCoverage(c: Coverage): string {
+ const lines: string[] = [];
+ lines.push(`# Coverage of ${c.slug} (held on the editor's disk)`);
+ lines.push(
+ `${c.held} held · ${c.dated} dated (${c.byRecordedDate} by recorded date, ${c.byUploadDate} by upload date)` +
+ `${c.undated.length ? ` · ${c.undated.length} undated` : ""} · ${day(c.first)} → ${day(c.last)}`,
+ );
+ lines.push(
+ c.titlePattern
+ ? `Dates: the recorded date read off each title (rule \`${c.titlePattern}\`), else the upload date.`
+ : "Dates: upload dates (the channel has no recorded-date title rule).",
+ );
+ const years = Object.entries(c.byYear).sort(([a], [b]) => a.localeCompare(b));
+ if (years.length) lines.push("", "## By year", ...years.map(([y, n]) => `- ${y}: ${n}`));
+ if (c.gaps.length) {
+ lines.push("", `## Gaps longer than ${c.gapDays} days (${c.gaps.length})`);
+ for (const g of c.gaps) lines.push(`- ${day(g.after)} → ${day(g.before)}: ${g.days} days with nothing held`);
+ } else {
+ lines.push("", `No gap longer than ${c.gapDays} days.`);
+ }
+ if (c.undated.length) {
+ const shown = c.undated.slice(0, 20);
+ lines.push("", `Undated (no metadata with an upload date): ${shown.join(", ")}${c.undated.length > 20 ? `, … (+${c.undated.length - 20})` : ""}`);
+ }
+ if (c.videos) {
+ lines.push("", `## Videos (${c.videos.length})`);
+ for (const v of c.videos) lines.push(`- ${day(v.date)} ${v.id}${v.title ? ` — ${v.title}` : ""}${v.recordedDate ? " (recorded)" : ""}`);
+ }
+ return lines.join("\n");
+}
+
+export async function channelCoverageTool(args: Record<string, unknown>, deps: EditorDeps): Promise<ToolAnswer> {
+ const channel = trimmed(args.channel);
+ if (!channel || !SLUG_RE.test(channel)) return { text: "channel_coverage: channel is required (the channel slug)", isError: true };
+ const q = new URLSearchParams({ slug: channel });
+ if (args.gap_days !== undefined) {
+ const n = Number(args.gap_days);
+ if (!Number.isInteger(n) || n < 1) return { text: "channel_coverage: gap_days is a whole number of days above zero", isError: true };
+ q.set("gapDays", String(n));
+ }
+ for (const [arg, key] of [["date_from", "from"], ["date_to", "to"]] as const) {
+ if (args[arg] === undefined) continue;
+ const m = DATE_ARG.exec(trimmed(args[arg]));
+ if (!m) return { text: `channel_coverage: ${arg} is a date, YYYY-MM-DD`, isError: true };
+ q.set(key, `${m[1]}${m[2]}${m[3]}`);
+ }
+ if (args.list === true) q.set("list", "1");
+ const a = await editorGet(deps, `/api/ops/coverage?${q.toString()}`);
+ if (a.kind !== "ok") return { text: describeEditorFailure("channel_coverage", a), isError: true };
+ return { text: renderCoverage(a.body as unknown as Coverage) };
+}
+
+// ─── notes (umtool) ───
+
+export const NO_UMTOOL_TEXT =
+ "notes: no umtool configured. Set UMTOOL_URL (e.g. http://localhost:3050) when " +
+ "registering the MCP server; the notes live in umtool.";
+
+export type NotesDeps = EditorDeps;
+
+function umtoolFrom(env: Record<string, string | undefined>): string | null {
+ const raw = (env.UMTOOL_URL ?? "").trim();
+ return raw ? raw.replace(/\/+$/, "") : null;
+}
+
+async function umtoolGet(
+ deps: NotesDeps,
+ route: string,
+): Promise<{ ok: true; status: number; json?: unknown; text?: string } | { ok: false; error: string }> {
+ const base = umtoolFrom(deps.env);
+ if (!base) return { ok: false, error: NO_UMTOOL_TEXT };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ try {
+ const res = await deps.fetch(`${base}${route}`, { method: "GET", signal: AbortSignal.timeout(timeoutMs) });
+ const body = await res.json().catch(() => null);
+ return { ok: true, status: res.status, json: body };
+ } catch (e) {
+ return { ok: false, error: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}` };
+ }
+}
+
+// A target as notes list names it: `sites/<site>/<report>` is an article,
+// anything else a report-video project id (which may hold a "/" too — the
+// prefix is what tells them apart).
+export function notesTarget(raw: string): { param: "article" | "project"; value: string } | null {
+ const t = raw.trim();
+ if (!t || /\s|\.\.|^\//.test(t)) return null;
+ const article = /^sites\/([a-z0-9][a-z0-9-]*\/[A-Za-z0-9][A-Za-z0-9._-]*)$/.exec(t);
+ if (article) return { param: "article", value: article[1] };
+ if (t.startsWith("sites/")) return null;
+ return { param: "project", value: t };
+}
+
+type Decision = { kind: string; project: string; projectKind?: string; target?: string; why?: string; href?: string };
+
+export async function notesTool(args: Record<string, unknown>, deps: NotesDeps): Promise<ToolAnswer> {
+ const action = trimmed(args.action) || "list";
+ if (action === "reply") {
+ const id = trimmed(args.note_id) || "<note id>";
+ const text = trimmed(args.text) || "<what changed>";
+ return {
+ isError: true,
+ text:
+ "notes: a reply is not written from here. umtool records every write through its HTTP " +
+ "route as the operator's, and an agent's reply must say it is an agent's — so it goes " +
+ "through umtool's CLI, from the repo checkout:\n\n" +
+ ` umtool notes reply ${id} ${JSON.stringify(text)}${args.resolve === true ? " --resolve" : ""}\n\n` +
+ "(`node umtool/bin/umtool.mjs notes reply …` when umtool is not on the PATH.) Edit the " +
+ "note's SOURCE file first (notes read shows it), regenerate, then reply.",
+ };
+ }
+ if (action === "list") {
+ const r = await umtoolGet(deps, "/api/browse/decisions");
+ if (!r.ok) return { text: r.error, isError: true };
+ if (r.status !== 200) return { text: `notes: umtool answered HTTP ${r.status}`, isError: true };
+ const items = ((r.json as { items?: Decision[] } | null)?.items ?? []).filter(
+ (d) => d.kind === "open-note" || d.kind === "unreadable-notes",
+ );
+ if (items.length === 0) return { text: "No open notes." };
+ const byProject = new Map<string, Decision[]>();
+ for (const d of items) byProject.set(d.project, [...(byProject.get(d.project) ?? []), d]);
+ const lines = [`${items.length} open note(s) on ${byProject.size} target(s):`];
+ for (const [project, ds] of byProject) {
+ lines.push("", `## ${project} (${ds[0].projectKind ?? "project"}) — read with notes action "read", target "${project}"`);
+ for (const d of ds) {
+ lines.push(d.kind === "unreadable-notes" ? `- UNREADABLE notes.json: ${d.why ?? ""}` : `- [${d.target ?? "note"}] ${d.why ?? ""}`);
+ }
+ }
+ return { text: lines.join("\n") };
+ }
+ if (action === "read") {
+ const target = notesTarget(trimmed(args.target));
+ if (!target) {
+ return { text: 'notes: read needs target as notes list names it — "sites/<site>/<report>" for an article, else a video project id', isError: true };
+ }
+ const status = trimmed(args.status) || "open";
+ if (!["open", "resolved", "all"].includes(status)) return { text: "notes: status is open, resolved or all", isError: true };
+ const base = umtoolFrom(deps.env);
+ if (!base) return { text: NO_UMTOOL_TEXT, isError: true };
+ const timeoutMs = deps.requestTimeoutMs ?? REQUEST_TIMEOUT_MS;
+ const q = new URLSearchParams({ [target.param]: target.value, status });
+ try {
+ const res = await deps.fetch(`${base}/api/notes/context?${q.toString()}`, {
+ method: "GET",
+ signal: AbortSignal.timeout(timeoutMs),
+ });
+ // The digest is text/plain; an error is JSON.
+ const raw = (res as unknown as { text?: () => Promise<string> }).text
+ ? await (res as unknown as { text: () => Promise<string> }).text()
+ : JSON.stringify(await res.json());
+ if (res.status !== 200) {
+ let error = raw;
+ try {
+ error = (JSON.parse(raw) as { error?: string }).error ?? raw;
+ } catch {
+ /* plain text */
+ }
+ return { text: `notes: umtool refused (HTTP ${res.status}): ${error}`, isError: true };
+ }
+ return { text: raw };
+ } catch (e) {
+ return { text: `notes: umtool at ${base} did not answer: ${describeFetchError(e, timeoutMs)}`, isError: true };
+ }
+ }
+ return { text: 'notes: action is "list", "read" or "reply"', isError: true };
+}
+
+// ─── The tool definitions (server.ts's TOOLS takes them in order) ───
+
+export const ARCHIVAL_TOOLS = [
+ {
+ name: "get_job",
+ description:
+ "One editor job's state — kind, channel, status, times, exit code, where it waits " +
+ "in its queue — and the last lines of its log. For the job id enqueue or fetch_clip " +
+ "returned. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ job: { type: "string", description: "The job id." },
+ tail: {
+ type: "number",
+ description: `Log lines to include (default ${DEFAULT_JOB_TAIL}, at most ${MAX_JOB_TAIL}; 0 for none).`,
+ },
+ },
+ required: ["job"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "enqueue",
+ description:
+ "Ask the local editor to queue archival work on a channel it archives, as the " +
+ "channel page's buttons do: sync (list and download what is new; full: true sweeps " +
+ "the whole listing), download-missing (every listed video not held), retry-bucket " +
+ "(one bucket of the channel's report, or ids within it — the way to download chosen " +
+ "videos), transcribe-bucket (the downloaded, untranscribed videos, or ids among them), " +
+ "fetch-posts (a social channel's new posts; full / older walks), import-video (one " +
+ "video by URL into the channel). The editor runs it on its own queues at each " +
+ "platform's pace and answers a job id — follow it with get_job. Only when the " +
+ "operator asks for it. Settings, storage and deletes are not reachable from here. " +
+ "Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ kind: { type: "string", enum: [...ENQUEUE_KINDS], description: "What to queue." },
+ channel: { type: "string", description: "The channel slug (the editor's)." },
+ full: { type: "boolean", description: "sync: sweep the whole listing now; fetch-posts: re-walk the timeline." },
+ older: { type: "boolean", description: "fetch-posts: walk back below the oldest archived post." },
+ bucket: { type: "string", description: "retry-bucket: the bucket name (e.g. noTranscript, partialDownloads)." },
+ ids: {
+ type: "array",
+ items: { type: "string" },
+ description: "retry-bucket / transcribe-bucket: only these video ids (each must be in the bucket).",
+ },
+ url: { type: "string", description: "import-video: the video's page URL." },
+ },
+ required: ["kind", "channel"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "channel_coverage",
+ description:
+ "What a channel HOLDS on the editor's disk, by date: how many videos, the first and " +
+ "last day, counts per year, and every gap longer than gap_days with nothing held. " +
+ "Each video is dated by its recorded date when the channel has a recorded-date " +
+ "title rule (a VOD mirror), else by its upload date. Counts what is held, not what " +
+ "is published. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ channel: { type: "string", description: "The channel slug (the editor's)." },
+ gap_days: { type: "number", description: "The shortest span reported as a gap (default 30)." },
+ date_from: { type: "string", description: "Only videos on or after this date (YYYY-MM-DD)." },
+ date_to: { type: "string", description: "Only videos on or before this date (YYYY-MM-DD)." },
+ list: { type: "boolean", description: "Also list every dated video (default false)." },
+ },
+ required: ["channel"],
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "notes",
+ description:
+ "The operator's notes in umtool on articles and report-video projects. action " +
+ '"list": every open note, grouped by target. action "read" with target (as list names ' +
+ 'it: "sites/<site>/<report>" for an article, else a project id): each note with its anchor ' +
+ "resolved and the SOURCE file to edit (status: open, resolved or all). action " +
+ '"reply" answers with the `umtool notes reply` command to run — umtool records a ' +
+ "reply sent over HTTP as the operator's, so an agent's reply goes through its CLI. " +
+ "Needs UMTOOL_URL.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ action: { type: "string", enum: ["list", "read", "reply"], description: 'Default "list".' },
+ target: { type: "string", description: 'read: as list names it — "sites/<site>/<report>" (an article) or a project id.' },
+ status: { type: "string", enum: ["open", "resolved", "all"], description: "read: which notes (default open)." },
+ note_id: { type: "string", description: "reply: the note id (n_…)." },
+ text: { type: "string", description: "reply: what changed." },
+ resolve: { type: "boolean", description: "reply: resolve the note too." },
+ },
+ additionalProperties: false,
+ },
+ },
+] as const;
diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts
@@ -160,6 +160,25 @@ function clipStep(ctx: PlanContext): string {
);
}
+// The editor-backed writes and reads (release 19 A9): the same rule as
+// fetch_clip — through the editor, never by hand, and only when asked.
+function archiveStep(): string {
+ return (
+ `**Archive work — through the editor, and only when I ask.** A video the ` +
+ `archive has not published yet is read off the editor's disk by ` +
+ `\`get_transcript\` (it says so; such a video has no moment link — cite its ` +
+ `source URL and time). When I ask for a channel to be synced, downloaded, ` +
+ `transcribed, its posts fetched or a video imported, call \`enqueue\` and ` +
+ `follow the job it names with \`get_job\` — a platform queue may hold it for ` +
+ `hours; report the job, do not wait it out. \`channel_coverage\` answers ` +
+ `what a channel holds by date and where its gaps are (a VOD mirror's videos ` +
+ `by the day they were recorded). The operator's notes on articles and ` +
+ `video projects: \`notes\` (list, read; a reply is \`umtool notes reply\`). ` +
+ `Settings, storage and deletes are not reachable from here — name the ` +
+ `\`pnpm ops\` command instead.`
+ );
+}
+
// The full sweep instructions.
export function buildSweepInstructions(
req: PromptRequest,
@@ -268,6 +287,8 @@ export function buildSweepInstructions(
steps.push(clipStep(ctx));
+ steps.push(archiveStep());
+
steps.push(
`**Finish.** Work to the end of the worklist, then write a summary ` +
`section: the scope swept, **how many of the N you actually read**, ` +
@@ -287,7 +308,8 @@ export function buildSweepInstructions(
`\`${reportPath}\`.\n\n` +
`You are the sweep engine — work the whole match set methodically, using ` +
`the transcript MCP tools for evidence and your own Write/Edit tools for ` +
- `the report. The MCP is read-only; never try to change the archive. The ` +
+ `the report. The MCP never changes the archive itself, and nothing here ` +
+ `asks the editor for a change unless I do. The ` +
`corpus holds video transcripts AND archived social posts.\n\n` +
`**Corpus: \`${ctx.corpus}\`.** Pass \`source: "${ctx.corpus}"\` on every ` +
`single call, including the ones your subagents make. This server has no ` +
@@ -363,6 +385,8 @@ export function buildAskInstructions(
);
steps.push(clipStep(ctx));
+
+ steps.push(archiveStep());
}
const numbered = steps.map((s, i) => `${i + 1}. ${s}`).join("\n\n");
@@ -370,7 +394,8 @@ export function buildAskInstructions(
`Answer this question from the transcript archive: **${question}**\n\n` +
`**Corpus: \`${ctx.corpus}\`.** Pass \`source: "${ctx.corpus}"\` on every ` +
`call — this server has no active source, and a call that omits it reads ` +
- `the server default. The MCP is read-only. The corpus holds video ` +
+ `the server default. The MCP never changes the archive itself; it asks ` +
+ `the local editor only when I ask it to. The corpus holds video ` +
`transcripts AND archived social posts.`;
const notes =
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -45,6 +45,10 @@ const EXPECTED_TOOLS = [
"get_transcripts",
"get_video_metadata",
"fetch_clip",
+ "get_job",
+ "enqueue",
+ "channel_coverage",
+ "notes",
"list_sources",
"resolve_source",
"open_link",
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -89,6 +89,14 @@ import {
type PollProgress,
} from "./fetchClip";
import { editorTranscript, renderEditorTranscript, type EditorDeps } from "./editorOps";
+import {
+ ARCHIVAL_TOOLS,
+ channelCoverageTool,
+ enqueue,
+ getJob,
+ notesTool,
+ type ToolAnswer,
+} from "./archivalTools";
import { extractVideoId } from "yt-dlp-transcript-common/lib/videoId";
import {
decodeShareLink,
@@ -885,6 +893,8 @@ export const TOOLS: Tool[] = [
additionalProperties: false,
},
},
+ // Archival writes through the editor (release 19 A9): archivalTools.ts.
+ ...(ARCHIVAL_TOOLS as unknown as Tool[]),
{
name: "list_sources",
description:
@@ -1160,6 +1170,11 @@ async function handleFetchClip(
return rendered.isError ? errorText(body) : text(body);
}
+// An archival tool's answer (archivalTools.ts) as a tool result.
+function fromAnswer(a: ToolAnswer): ToolResult {
+ return a.isError ? errorText(a.text) : text(a.text);
+}
+
// One MCP progress notification per poll while fetch_clip waits, when the
// client asked for progress (a `progressToken` in the request's _meta). A
// client whose request timeout resets on progress then keeps waiting for as
@@ -1278,6 +1293,14 @@ export function createServer(
fetchClipDeps,
progressNotifier(ctx.mcpReq._meta?.progressToken, ctx.mcpReq.notify),
);
+ case "get_job":
+ return fromAnswer(await getJob(args, fetchClipDeps));
+ case "enqueue":
+ return fromAnswer(await enqueue(args, fetchClipDeps));
+ case "channel_coverage":
+ return fromAnswer(await channelCoverageTool(args, fetchClipDeps));
+ case "notes":
+ return fromAnswer(await notesTool(args, fetchClipDeps));
case "list_sources":
return handleListSources(registry, resolved);
case "resolve_source":
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -17,6 +17,7 @@
// pnpm ops get jobs [--active | --failed] [--kind <kind>] [--slug <slug>] [--limit <n>]
// pnpm ops get remote-listing <slug>
// pnpm ops get transcript <videoId> [--slug <slug>]
+// pnpm ops get coverage <slug>
// pnpm ops job <cancel|drain|promote|force-release|retry> <id>... [--wait]
// pnpm ops job retry-failed | job wait <id>...
// pnpm ops list
@@ -166,6 +167,9 @@ const GETTERS = {
scheduler: () => "/api/ops/scheduler",
// One channel's cleanup row: what each sweep would reclaim, what holds the rest.
cleanup: (slug) => `/api/ops/cleanup/${encodeURIComponent(slug)}`,
+ // What a channel holds, by date, and its gaps (release 19 A9): each video
+ // dated by its recorded date when the channel has a title rule, else upload.
+ coverage: (slug) => `/api/ops/coverage?slug=${encodeURIComponent(slug)}`,
// One video's cues off disk, with no index (release 19 A7): a fresh
// cues.json, else what normalize would write, else the VTT alone. --slug
// names the channel (else the one holding data/<id>/).
@@ -652,6 +656,7 @@ export function usage() {
" pnpm ops get cleanup <slug>",
" pnpm ops get remote-listing <slug> [--wait-timeout <seconds>]",
" pnpm ops get transcript <videoId> [--slug <slug>]",
+ " pnpm ops get coverage <slug>",
" pnpm ops list",
"",
`Actions: ${ACTIONS.join(", ")}`,
@@ -702,6 +707,12 @@ export function usage() {
' (every one held), "force": true to rewrite a fresh one. A job on the',
" channel's queue. The file every reader without an index build wants.",
"",
+ "get coverage <slug> is what a channel holds by date: held, dated (by",
+ " recorded date — the channel's recordedDate title rule — else by upload",
+ " date), undated, first and last day, per year and month, and every gap",
+ " over 30 days with nothing held. Off disk; nothing written. (The MCP's",
+ " channel_coverage takes gap days, a date window and a video list.)",
+ "",
"get transcript <videoId> [--slug <slug>] reads one video's cues off disk",
" with no index: a fresh cues.json, else what build-cues would write, else",
" the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is",
@@ -850,7 +861,9 @@ export function usage() {
' any of "patch" (form field names; "" clears one), "sites" (the WHOLE',
' membership set: [{"siteId", "groupId"? | "newGroupName"?}], [] = on no',
' site; an unknown site id is refused), "excludeFromBuild" and',
- ' "excludeFromCleanup" (set to the value given, not toggled).',
+ ' "excludeFromCleanup" (set to the value given, not toggled). A VOD mirror\'s',
+ ' recorded date: "patch": {"recordedDateTitlePattern": "<regex with named',
+ ' groups year, month, day>"} (CHANNEL.md, recordedDate).',
"",
'create-channel is the New channel form: {"fields": {"name", "handling":',
' "youtube"|"transcribe", "url"?, "platform"?, "sourceKind"?, "postFetcher"?,',
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -822,3 +822,12 @@ test("build-cues is an action; get transcript reads one video's cues, --slug nam
assert.match(parseArgs(["get", "transcript", "v", "--counts"]).error, /does not take --counts/);
assert.match(usage(), /build-cues writes each video's transcript\.cues\.json/);
});
+
+test("get coverage reads one channel's coverage", () => {
+ const r = parseArgs(["get", "coverage", "demo"]);
+ assert.equal(r.method, "GET");
+ assert.equal(r.path, "/api/ops/coverage?slug=demo");
+ assert.match(parseArgs(["get", "coverage"]).error, /needs an argument/);
+ assert.match(usage(), /get coverage <slug> is what a channel holds by date/);
+ assert.match(usage(), /recordedDateTitlePattern/);
+});