commit b31f5d485d6d2b89f38aec9a66db1f2306cc655c
parent 3bea5b510b9e7fdde09b27bbdbedce0fe79c1fcf
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 20 Jul 2026 17:54:38 -0400
MCP: first-class corpus sweeping — sweep prompt, pageable + alias-aware search, batch windowed reads
Turn the mcp/ server into a Claude-Code-driven sweep engine so the browser's
BYO-key corpus sweep can run on plan usage instead. All additive + read-only.
- source.ts: ShardSource.loadAliases() (Local/Remote/Hub, cached) reads the
site's shipped /search-aliases.json via coerceAliasConfig.
- search.ts: buildMatcher() combines plain substring OR fired alias suggestion
regexes (mirrors the browser's buildSearchRoot). searchTranscripts gains
offset/total/hasMore/maxPages/includeSnippets/useAliases + firedAliases.
getWindowedTranscript() windows cues around matches via transcriptWindow.
- server.ts: search_transcripts gains offset/include_snippets/use_aliases/
max_pages + a richer footer (total, has_more, PARTIAL, aliases fired); new
get_transcripts batch tool (<=20, windowed w/ query else full markdown,
missing ids inline); prompts capability + a sweep prompt.
- tsconfig: add dom lib (types-only; transcriptWindow->aiHandoff type-imports a
browser component). package.json: test script. search.test.ts: 11 tests
(unit + end-to-end via InMemoryTransport). README + export/CHANGELOG updated.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
8 files changed, 879 insertions(+), 27 deletions(-)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **MCP server: run the corpus sweep through Claude Code itself — on plan usage, no API key.** The in-browser "corpus sweep" (fold every matching transcript into a running report) needs a BYO AI key, and a free key conks out fast on a real 30k-video corpus. The `mcp/` server now lets **Claude Code be the sweep engine** instead, via a first-class **`sweep` prompt** (a slash command, `/mcp__<name>__sweep query="k cups" channel="chrissie-mayr"`) plus the tools to drive it. `search_transcripts` gains **paging** — it reports the full `total` and `has_more`, so the whole match set can be enumerated with `offset` (and `include_snippets:false` for a cheap worklist) — and is now **alias-aware**: a plain query that matches a curated search alias also searches the alias regex (e.g. `k cups` → also `cake cup`), with the footer naming which aliases fired; reaching the scan cap is surfaced as **PARTIAL** coverage rather than hidden. A new **`get_transcripts`** tool batch-reads up to 20 videos in one call as bounded, timestamped **excerpt windows** around the matches (alias-correct) — or full transcripts without a query — so a sweep stays token-bounded. The `sweep` prompt walks Claude through search → enumerate → plan `ceil(N/batch)` batches → per-batch windowed read + cross-referenced upsert of cited findings (*title + [mm:ss]*) into a markdown report it maintains with its own Write/Edit tools. The MCP stays strictly **read-only**; only the report file is written, in Claude's own working directory. See `mcp/src/{search,source,server}.ts`, `mcp/src/search.test.ts`, and `mcp/README.md`.
- **"Ask AI" now paces itself to your key's rate limit instead of failing.** A whole-corpus sweep on a free-tier key used to fire provider calls as fast as the loop could produce them, blow straight through the per-minute request cap, and stop dead on the first HTTP 429 (and a rate-limit *mid-answer* surfaced as a hard error, because only the sweep path ever handled 429). Now a single client-side limiter sits behind **every** AI call: it **spaces requests** to a conservative, free-tier-safe **requests-per-minute** default chosen per model (e.g. Gemini `*-pro` → 5/min, `*flash` → 10/min; Claude/OpenAI higher), so a paid key runs fast and a free key just runs *slowly* rather than erroring. When a 429 does land, the limiter **honours the provider's own retry hint** (the `Retry-After` header, or Gemini's `RetryInfo` retry-delay) — or an exponential backoff — and **re-issues the request** (safe for streaming: the retry happens before any answer text is emitted), and it **self-lowers** the rate after a 429 so later calls pace slower. Only a 429 that outlasts the retries falls through to the existing **pause/checkpoint** (the right home for a daily-quota cap you resume tomorrow). Provider settings gain an editable **Requests per minute** field (with the effective spacing, e.g. *"≈12s between requests"*, and a reset-to-default), and a paced sweep shows a **"Rate limited — retrying in Ns"** line in its progress strip so it never looks frozen. See `export/app/lib/rateLimit.ts` (the whole mechanism), `export/app/lib/askProvider.ts` (`PausableError` + `acquire`/retry in `askStream`, 429 in `ensureOk`), `export/app/lib/nativeTools/shared.ts` (`acquire`/retry in `postJson`), `export/app/ask/{useAskChat,ProviderSettings,PinnedResultsPanel,AskChat}.tsx`, and `export/e2e/ask-workspace.spec.ts`.
- **"Ask AI" — an integrated grounding workspace: pick your AI target, sweep any size, save & resume chats.** Search and the chat used to behave like two tabs, grounding was all-or-nothing, a sweep had hard-coded caps and couldn't be paused, and a conversation lived in a single unnamed slot that "New chat" wiped. Now they're one surface:
- **Grounding palette (the signature control).** A first-class `Whole search ⇄ Selection` toggle in the chat always states *what the AI is looking at*, alongside an action cluster — **Ask**, **Sweep**, and preset directives (*Summary*, *Contradictions*, *Timeline*) that pre-fill the composer/sweep box. "Selection" grounds in a **hand-picked subset** of results shown as dismissible chips; an empty selection parks on Whole search, so nothing changes for people who never select.
diff --git a/mcp/README.md b/mcp/README.md
@@ -13,10 +13,64 @@ reads the site's already-published static JSON shards (`corpus.json` +
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels (name, slug, video count; site in hub mode). |
-| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Optional `channel` filter. |
+| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Optional `channel` filter. |
+| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. |
| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). |
| `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. |
+### `search_transcripts`
+
+Beyond `query`, `channel`, `regex`, and `limit`:
+
+- **`offset`** (default 0) — skip this many matches. The result footer reports the
+ full `total` and `has_more`, so you can enumerate a query's *entire* match set:
+ page with `offset += limit` until `has_more` is `no`.
+- **`include_snippets`** (default true) — set `false` for a cheap worklist
+ (id / title / channel / date / match count, no cue text). Ideal for the
+ planning pass of a sweep.
+- **`use_aliases`** (default true) — expand the query through the site's curated
+ search aliases. A plain query that matches a curated trigger also searches the
+ alias's regex, so mis-transcribed spellings are caught (e.g. `k cups` also
+ matches `cake cup`). The footer reports which aliases fired, e.g.
+ *"expanded via alias K-Cups → `(k|cake)[ -]?cup`"*. Explicit `regex` queries are
+ used verbatim (no expansion).
+- **`max_pages`** (default 400) — scan cap. If reached (or a very common term
+ passes the 2000-video cap), the footer flags coverage as **PARTIAL** rather
+ than silently truncating.
+
+### `get_transcripts`
+
+Reads a whole batch of videos with one round-trip. `video_ids` (max 20; extras
+dropped and noted). With a **`query`** (alias-aware, like `search_transcripts`;
+optional `regex`/`use_aliases`), each transcript is reduced to bounded windows of
+timestamped lines around the matches (`before`/`after` seconds, default 30) —
+high-signal context for folding a batch into a report. Without a query, each
+video's full transcript comes back as markdown. Missing ids are reported inline.
+
+## The `sweep` prompt
+
+A first-class slash command that turns **Claude Code itself** into the corpus
+sweep engine — the same "batch matching transcripts into a running report" the
+browser does with a BYO AI key, but driven by your Claude **plan usage** (no API
+key) and with the report written to a file.
+
+Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query`
+(required), `channel?`, `directive?` (default *"key claims & contradictions"*),
+`batch_size?` (default 8), `report_path?` (default `./sweep-report.md`).
+
+```
+/mcp__rekietalyzer__sweep query="k cups" channel="chrissie-mayr"
+```
+
+The prompt instructs Claude Code to: **search** (aliases auto-expand) → **enumerate**
+the full worklist by paging with `include_snippets:false` until `has_more` is
+false → **plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts`
+for windowed context, cross-reference against the report so far, and upsert
+findings (claims, contradictions, sources cited as *title + [mm:ss]*) into
+well-titled `##` sections of the report file with its own Write/Edit tools →
+finish with a short summary. The MCP stays read-only; only the report file is
+written, in Claude's working directory.
+
## Data source (pick one)
Resolved from flags or env — precedence hub > remote > local:
diff --git a/mcp/package.json b/mcp/package.json
@@ -9,6 +9,7 @@
},
"scripts": {
"start": "tsx src/index.ts",
+ "test": "tsx --test src/*.test.ts",
"typecheck": "tsc --noEmit -p tsconfig.json"
},
"dependencies": {
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -0,0 +1,279 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import type { ChannelRef, ShardSource } from "./source";
+import { searchTranscripts, getWindowedTranscript, buildMatcher } from "./search";
+import { createServer } from "./server";
+
+// ─── A tiny in-memory ShardSource for the tests ───
+
+function cues(...pairs: [number, string][]): Cue[] {
+ return pairs.map(([start, text]) => ({ start, end: start + 3, text }));
+}
+
+function vid(id: string, title: string, channelSlug: string, cs: Cue[]): TranscriptDetail {
+ return {
+ slug: id,
+ id,
+ channelSlug,
+ title,
+ uploadDate: "20240101",
+ duration: 300,
+ channel: channelSlug === "chan-a" ? "Channel A" : "Channel B",
+ description: "",
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: `https://example.test/${id}`,
+ cues: cs,
+ };
+}
+
+const K_CUPS_ALIAS: SearchAlias = {
+ id: "k-cups",
+ label: "K-Cups",
+ triggers: ["k cups"],
+ suggestion: "(k|cake)[ -]?cup",
+ useRegex: true,
+ enabled: true,
+};
+
+// Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup".
+const CHAN_A: TranscriptDetail[] = [
+ vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"])),
+ vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])),
+ vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"])),
+ vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"])),
+ vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"])),
+ vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])),
+ vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])),
+];
+
+// Channel B: one long transcript for windowing (a single match at 100s).
+const CHAN_B: TranscriptDetail[] = [
+ vid(
+ "b1",
+ "Long one",
+ "chan-b",
+ cues(
+ [0, "intro chatter"],
+ [50, "still warming up"],
+ [98, "right before the moment"],
+ [100, "here is the coffee moment"],
+ [102, "right after the moment"],
+ [160, "much later unrelated"],
+ [220, "the very end"],
+ ),
+ ),
+];
+
+class StubSource implements ShardSource {
+ readonly label = "stub";
+ constructor(private aliases: SearchAlias[] = [K_CUPS_ALIAS]) {}
+
+ async loadAliases(): Promise<SearchAlias[]> {
+ return this.aliases;
+ }
+
+ async listChannels(): Promise<ChannelRef[]> {
+ return [
+ { key: "chan-a", slug: "chan-a", name: "Channel A" },
+ { key: "chan-b", slug: "chan-b", name: "Channel B" },
+ ];
+ }
+
+ private pages(ch: ChannelRef): TranscriptDetail[][] {
+ // One record per page so paging exercises multiple shard pages.
+ const recs = ch.slug === "chan-a" ? CHAN_A : CHAN_B;
+ return recs.map((r) => [r]);
+ }
+
+ async transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest> {
+ const pages = this.pages(ch);
+ const slugToPage: Record<string, number> = {};
+ pages.forEach((p, i) => (slugToPage[p[0].id] = i));
+ return {
+ version: 4,
+ channelSlug: ch.slug,
+ pageCount: pages.length,
+ maxPageBytes: 0,
+ generatedAt: "",
+ slugToPage,
+ };
+ }
+
+ async transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> {
+ return this.pages(ch)[page] ?? [];
+ }
+}
+
+// ─── (a) Paging: offset / limit / total / hasMore ───
+
+test("paging: total is stable and hasMore/offset slice the full set", async () => {
+ const src = new StubSource();
+ // Six "coffee" matches: a1–a5 (Channel A) + b1 (its cue says "coffee moment"),
+ // in corpus order (channel A pages first, then B).
+ const page1 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 0 });
+ assert.equal(page1.total, 6, "six coffee videos total");
+ assert.equal(page1.hits.length, 2);
+ assert.deepEqual(page1.hits.map((h) => h.videoId), ["a1", "a2"]);
+ assert.equal(page1.hasMore, true);
+
+ const page2 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 2 });
+ assert.deepEqual(page2.hits.map((h) => h.videoId), ["a3", "a4"]);
+ assert.equal(page2.hasMore, true);
+
+ const page3 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 4 });
+ assert.deepEqual(page3.hits.map((h) => h.videoId), ["a5", "b1"]);
+ assert.equal(page3.hasMore, false);
+ assert.equal(page3.total, 6, "total unchanged across pages");
+});
+
+// ─── (b) Alias expansion ───
+
+test("alias expansion: 'k cups' matches both the literal and the aliased spelling", async () => {
+ const src = new StubSource();
+ const res = await searchTranscripts(src, { query: "k cups", limit: 20 });
+ const ids = res.hits.map((h) => h.videoId).sort();
+ assert.deepEqual(ids, ["a6", "a7"], "literal 'k cups' and aliased 'cake cup' both match");
+ assert.equal(res.firedAliases.length, 1);
+ assert.equal(res.firedAliases[0].id, "k-cups");
+});
+
+test("alias expansion: use_aliases:false falls back to plain substring only", async () => {
+ const src = new StubSource();
+ const res = await searchTranscripts(src, { query: "k cups", limit: 20, useAliases: false });
+ assert.deepEqual(res.hits.map((h) => h.videoId), ["a6"], "only the literal match survives");
+ assert.equal(res.firedAliases.length, 0);
+});
+
+test("alias expansion: explicit regex disables alias expansion", async () => {
+ const src = new StubSource();
+ const res = await searchTranscripts(src, { query: "cake cup", regex: true, limit: 20 });
+ assert.deepEqual(res.hits.map((h) => h.videoId), ["a7"]);
+ assert.equal(res.firedAliases.length, 0);
+});
+
+// ─── (c) include_snippets ───
+
+test("include_snippets:false omits snippet text but keeps the match count", async () => {
+ const src = new StubSource();
+ const withSnips = await searchTranscripts(src, { query: "coffee", limit: 1 });
+ assert.ok(withSnips.hits[0].snippets.length > 0, "snippets present by default");
+
+ const noSnips = await searchTranscripts(src, {
+ query: "coffee",
+ limit: 1,
+ includeSnippets: false,
+ });
+ assert.equal(noSnips.hits[0].snippets.length, 0, "no snippets");
+ assert.ok(noSnips.hits[0].matches >= 1, "match count still reported");
+});
+
+// ─── (d) getWindowedTranscript ───
+
+test("getWindowedTranscript: windows a bounded region around the match", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "coffee" });
+ const { lines, matchCount } = getWindowedTranscript(rec, match, { before: 30, after: 30 });
+ assert.equal(matchCount, 1);
+ const body = lines.join("\n");
+ // Cues within ±30s of the 100s match are kept…
+ assert.ok(body.includes("right before the moment"));
+ assert.ok(body.includes("here is the coffee moment"));
+ assert.ok(body.includes("right after the moment"));
+ // …and cues far outside the window are dropped.
+ assert.ok(!body.includes("intro chatter"));
+ assert.ok(!body.includes("much later unrelated"));
+ assert.ok(!body.includes("the very end"));
+});
+
+test("getWindowedTranscript: no matches yields no lines", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "banana" });
+ const { lines, matchCount } = getWindowedTranscript(rec, match);
+ assert.equal(matchCount, 0);
+ assert.equal(lines.length, 0);
+});
+
+// ─── End-to-end through the MCP server (tools + prompts + missing ids) ───
+
+async function connectClient(source: ShardSource): Promise<Client> {
+ const server = createServer(source);
+ const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([
+ server.connect(serverTransport),
+ client.connect(clientTransport),
+ ]);
+ return client;
+}
+
+function firstText(res: unknown): string {
+ const content = (res as { content: { type: string; text: string }[] }).content;
+ return content.map((c) => c.text).join("\n");
+}
+
+test("server: search_transcripts footer reports total, has_more, and the alias", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "k cups", limit: 20 },
+ });
+ const out = firstText(res);
+ assert.match(out, /total 2 match/);
+ assert.match(out, /has_more: no/);
+ assert.match(out, /expanded via alias K-Cups/);
+ await client.close();
+});
+
+test("server: get_transcripts windows with a query and reports missing ids", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1", "nope"], query: "coffee" },
+ });
+ const out = firstText(res);
+ assert.match(out, /here is the coffee moment/);
+ assert.ok(!out.includes("intro chatter"), "far cues excluded by the window");
+ assert.match(out, /not found: nope/);
+ await client.close();
+});
+
+test("server: get_transcripts without a query returns full markdown", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"] },
+ });
+ const out = firstText(res);
+ assert.match(out, /## Transcript/);
+ assert.ok(out.includes("intro chatter"), "full transcript includes every cue");
+ assert.ok(out.includes("the very end"));
+ await client.close();
+});
+
+test("server: the sweep prompt lists with its arguments and renders the query", async () => {
+ const client = await connectClient(new StubSource());
+ const list = await client.listPrompts();
+ const sweep = list.prompts.find((p) => p.name === "sweep");
+ assert.ok(sweep, "sweep prompt is listed");
+ const names = (sweep!.arguments ?? []).map((a) => a.name);
+ assert.deepEqual(names.sort(), ["batch_size", "channel", "directive", "query", "report_path"]);
+
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a" },
+ });
+ const msg = got.messages[0].content;
+ assert.equal(msg.type, "text");
+ assert.match((msg as { text: string }).text, /"k cups"/);
+ assert.match((msg as { text: string }).text, /chan-a/);
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -1,5 +1,15 @@
import { formatDuration } from "yt-dlp-transcript-common/lib/format";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import {
+ matchAliases,
+ type SearchAlias,
+} from "yt-dlp-transcript-common/lib/searchAliases";
+import {
+ windowCues,
+ cuesToSnippets,
+ mergeSnippets,
+ type WindowSnippet,
+} from "yt-dlp-transcript-common/lib/transcriptWindow";
import type { ChannelRef, ShardSource } from "./source";
export type Snippet = { clock: string; seconds: number; text: string };
@@ -18,7 +28,19 @@ export type SearchHit = {
export type SearchResult = {
hits: SearchHit[];
+ // Total matched videos found in this scan (up to HARD_VIDEO_CAP). `hits` is
+ // the [offset, offset+limit) slice of that set, so total is stable across
+ // paged calls and lets a caller plan a full sweep.
+ total: number;
+ offset: number;
+ limit: number;
+ hasMore: boolean;
+ // The curated aliases that fired for this query (so the tool can report the
+ // expansion it applied). Empty for a plain or explicit-regex search.
+ firedAliases: SearchAlias[];
scanned: { channels: number; pages: number };
+ // Coverage is partial — the page cap (MAX_PAGES) or the video cap
+ // (HARD_VIDEO_CAP) was reached before the corpus was fully scanned.
truncated: boolean;
};
@@ -26,18 +48,63 @@ export type SearchResult = {
// (or hub-wide) corpus can't run away. Reaching it sets `truncated`.
const MAX_PAGES = 400;
+// A ceiling on the number of matched videos we collect before we stop counting,
+// so `total` stays bounded and stable even for a very common term. Reaching it
+// also sets `truncated` (the true total is higher than reported).
+const HARD_VIDEO_CAP = 2000;
+
+// Cap on windowed excerpt lines emitted per video by getWindowedTranscript, so a
+// video with hundreds of matches can't blow the batch's token budget.
+const WINDOW_LINE_CAP = 200;
+
function clock(seconds: number): string {
const s = Math.max(0, Math.floor(seconds));
return s === 0 ? "0:00" : formatDuration(s);
}
-function makeMatcher(query: string, regex: boolean): (text: string) => boolean {
- if (regex) {
- const re = new RegExp(query, "i");
- return (t) => re.test(t);
+export type Matcher = (text: string) => boolean;
+
+// Build the combined, alias-aware matcher for a query, mirroring the browser's
+// buildSearchRoot OR-of-leaves semantics: a text matches if the plain query
+// substring matches OR any fired alias's suggestion regex matches. An explicit
+// `regex` query is taken verbatim with NO alias expansion (the caller is
+// crafting their own pattern). Returns the fired aliases so the tool can report
+// which curated expansions it applied.
+export function buildMatcher(opts: {
+ query: string;
+ regex?: boolean;
+ useAliases?: boolean;
+ aliases?: SearchAlias[];
+}): { match: Matcher; firedAliases: SearchAlias[] } {
+ if (opts.regex) {
+ const re = new RegExp(opts.query, "i");
+ return { match: (t) => re.test(t), firedAliases: [] };
}
- const needle = query.toLowerCase();
- return (t) => t.toLowerCase().includes(needle);
+ const needle = opts.query.toLowerCase();
+ const plain: Matcher = (t) => t.toLowerCase().includes(needle);
+
+ const useAliases = opts.useAliases !== false;
+ const fired =
+ useAliases && opts.aliases && opts.aliases.length > 0
+ ? matchAliases(opts.query, "transcripts", opts.aliases)
+ : [];
+ if (fired.length === 0) return { match: plain, firedAliases: [] };
+
+ const aliasMatchers: Matcher[] = fired.map((a) => {
+ if (a.useRegex) {
+ try {
+ const re = new RegExp(a.suggestion, "i");
+ return (t: string) => re.test(t);
+ } catch {
+ // malformed suggestion regex — fall back to substring on the literal
+ }
+ }
+ const n = a.suggestion.toLowerCase();
+ return (t: string) => t.toLowerCase().includes(n);
+ });
+
+ const match: Matcher = (t) => plain(t) || aliasMatchers.some((m) => m(t));
+ return { match, firedAliases: fired };
}
function truncate(text: string, max = 240): string {
@@ -45,10 +112,12 @@ function truncate(text: string, max = 240): string {
return t.length > max ? t.slice(0, max - 1) + "…" : t;
}
-// Scan a source's transcript shards for `query`, returning up to `limit` matched
-// videos (each with a few snippet cues + timestamps). A plain server-side scan
-// with the site's own match semantics (substring by default, or a regex) — no
-// browser index needed. Stops early at `limit` and at MAX_PAGES.
+// Scan a source's transcript shards for `query` (alias-aware by default),
+// collecting ALL matched videos up to HARD_VIDEO_CAP so counting is stable, then
+// returning the [offset, offset+limit) slice with a `total`/`hasMore`. Plain
+// substring by default, or a caller-supplied regex; either can be paged. Set
+// `includeSnippets:false` for a cheap worklist (id/title/channel/date/matches,
+// no cue text). Stops early at MAX_PAGES (→ truncated) and HARD_VIDEO_CAP.
export async function searchTranscripts(
source: ShardSource,
opts: {
@@ -56,12 +125,31 @@ export async function searchTranscripts(
channel?: string;
regex?: boolean;
limit?: number;
+ offset?: number;
+ maxPages?: number;
+ includeSnippets?: boolean;
+ useAliases?: boolean;
snippetsPerVideo?: number;
+ aliases?: SearchAlias[];
},
): Promise<SearchResult> {
const limit = opts.limit ?? 20;
+ const offset = Math.max(0, opts.offset ?? 0);
+ const maxPages = opts.maxPages ?? MAX_PAGES;
+ const includeSnippets = opts.includeSnippets !== false;
const snippetsPerVideo = opts.snippetsPerVideo ?? 4;
- const match = makeMatcher(opts.query, opts.regex === true);
+
+ const aliases =
+ opts.aliases ??
+ (opts.useAliases !== false && !opts.regex
+ ? await source.loadAliases()
+ : []);
+ const { match, firedAliases } = buildMatcher({
+ query: opts.query,
+ regex: opts.regex,
+ useAliases: opts.useAliases,
+ aliases,
+ });
let channels = await source.listChannels();
if (opts.channel) {
@@ -74,7 +162,7 @@ export async function searchTranscripts(
);
}
- const hits: SearchHit[] = [];
+ const all: SearchHit[] = [];
let pagesScanned = 0;
let channelsScanned = 0;
let truncated = false;
@@ -88,7 +176,7 @@ export async function searchTranscripts(
}
channelsScanned++;
for (let page = 0; page < manifest.pageCount; page++) {
- if (pagesScanned >= MAX_PAGES) {
+ if (pagesScanned >= maxPages) {
truncated = true;
break outer;
}
@@ -106,7 +194,7 @@ export async function searchTranscripts(
for (const cue of rec.cues ?? []) {
if (!match(cue.text)) continue;
matches++;
- if (snippets.length < snippetsPerVideo) {
+ if (includeSnippets && snippets.length < snippetsPerVideo) {
snippets.push({
clock: clock(cue.start),
seconds: cue.start,
@@ -115,7 +203,7 @@ export async function searchTranscripts(
}
}
if (matches === 0 && !titleHit) continue;
- hits.push({
+ all.push({
videoId: rec.id,
channelSlug: ch.slug,
channelName: ch.name,
@@ -126,18 +214,65 @@ export async function searchTranscripts(
matches: matches || 1,
snippets,
});
- if (hits.length >= limit) break outer;
+ if (all.length >= HARD_VIDEO_CAP) {
+ truncated = true;
+ break outer;
+ }
}
}
}
+ const total = all.length;
+ const hits = all.slice(offset, offset + limit);
return {
hits,
+ total,
+ offset,
+ limit,
+ hasMore: offset + limit < total,
+ firedAliases,
scanned: { channels: channelsScanned, pages: pagesScanned },
truncated,
};
}
+// Render a single record's transcript around the cues that match `matcher`:
+// for each matched cue, take a bounded window of surrounding cues (±`before`/
+// `after` seconds), merge overlapping windows deduped by timestamp (capped), and
+// return timestamped excerpt lines plus the total match count. Bounded and
+// high-signal — the batch read the sweep prompt drives.
+export function getWindowedTranscript(
+ record: TranscriptDetail,
+ matcher: Matcher,
+ opts: {
+ before?: number;
+ after?: number;
+ maxCues?: number;
+ timestamps?: boolean;
+ } = {},
+): { lines: string[]; matchCount: number } {
+ const cues = record.cues ?? [];
+ const timestamps = opts.timestamps !== false;
+ let merged: WindowSnippet[] = [];
+ let matchCount = 0;
+ for (const cue of cues) {
+ if (!matcher(cue.text)) continue;
+ matchCount++;
+ const win = cuesToSnippets(
+ windowCues(cues, cue.start, {
+ before: opts.before,
+ after: opts.after,
+ maxCues: opts.maxCues,
+ }),
+ );
+ merged = mergeSnippets(merged, win, WINDOW_LINE_CAP);
+ }
+ const lines = merged.map((s) =>
+ timestamps ? `[${s.clock}] ${s.text}` : s.text,
+ );
+ return { lines, matchCount };
+}
+
// Locate a single video across the source's channels via each channel's
// slugToPage map, returning the full record + its channel. `channelHint`
// (slug/key/name) short-circuits the scan when the caller knows the channel.
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -2,11 +2,19 @@ import { Server } from "@modelcontextprotocol/sdk/server/index.js";
import {
CallToolRequestSchema,
ListToolsRequestSchema,
+ ListPromptsRequestSchema,
+ GetPromptRequestSchema,
} from "@modelcontextprotocol/sdk/types.js";
import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { ShardSource } from "./source";
-import { searchTranscripts, findVideo } from "./search";
+import {
+ searchTranscripts,
+ findVideo,
+ buildMatcher,
+ getWindowedTranscript,
+} from "./search";
type ToolResult = {
content: { type: "text"; text: string }[];
@@ -35,7 +43,12 @@ const TOOLS = [
"Search transcript captions for a term or phrase and return matching " +
"videos with timestamped snippets. Substring match by default; set regex " +
"to true for a case-insensitive regular expression. Optionally restrict to " +
- "one channel (by slug or name).",
+ "one channel (by slug or name). Alias-aware: a plain query with a curated " +
+ "search alias auto-expands to the alias regex (e.g. 'k cups' also matches " +
+ "'cake cup') and the footer reports which aliases fired. Pageable: returns " +
+ "the total match count and whether more pages exist, so a caller can " +
+ "enumerate a query's full match set with offset. Set include_snippets to " +
+ "false for a cheap worklist (no cue text).",
inputSchema: {
type: "object",
properties: {
@@ -46,11 +59,37 @@ const TOOLS = [
},
regex: {
type: "boolean",
- description: "Treat query as a case-insensitive regex (default false).",
+ description:
+ "Treat query as a case-insensitive regex (default false). " +
+ "Disables alias expansion — the pattern is used verbatim.",
},
limit: {
type: "number",
- description: "Max matching videos to return (default 20).",
+ description: "Max matching videos to return in this page (default 20).",
+ },
+ offset: {
+ type: "number",
+ description:
+ "Skip this many matches before the page (default 0). Page with " +
+ "offset += limit until has_more is false to walk the full set.",
+ },
+ include_snippets: {
+ type: "boolean",
+ description:
+ "Include timestamped snippet lines per video (default true). Set " +
+ "false for a compact worklist (id/title/channel/date/match count).",
+ },
+ use_aliases: {
+ type: "boolean",
+ description:
+ "Expand the query via the site's curated search aliases (default " +
+ "true; ignored when regex is true).",
+ },
+ max_pages: {
+ type: "number",
+ description:
+ "Max shard pages to scan before stopping (default 400). Reaching " +
+ "it marks coverage partial.",
},
},
required: ["query"],
@@ -78,6 +117,61 @@ const TOOLS = [
},
},
{
+ name: "get_transcripts",
+ description:
+ "Batch-read up to 20 videos' transcripts in one call. With a query, each " +
+ "transcript is reduced to bounded, timestamped excerpt windows around the " +
+ "matching lines (alias-aware, like search_transcripts) — high-signal " +
+ "context for folding a batch into a report. Without a query, each video's " +
+ "full transcript is returned as markdown. Missing ids are reported inline. " +
+ "This is the batching workhorse for a sweep.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ video_ids: {
+ type: "array",
+ items: { type: "string" },
+ description: "The video ids to read (max 20; extras are dropped).",
+ },
+ query: {
+ type: "string",
+ description:
+ "Optional term/phrase/regex to window around. When given, only " +
+ "excerpts around matches are returned instead of full transcripts.",
+ },
+ regex: {
+ type: "boolean",
+ description:
+ "Treat query as a case-insensitive regex (default false). Disables " +
+ "alias expansion.",
+ },
+ use_aliases: {
+ type: "boolean",
+ description:
+ "Expand query via curated aliases (default true; ignored with regex).",
+ },
+ channel: {
+ type: "string",
+ description: "Optional owning channel slug/name to speed the lookup.",
+ },
+ timestamps: {
+ type: "boolean",
+ description: "Prefix each line with a timestamp (default true).",
+ },
+ before: {
+ type: "number",
+ description: "Seconds of context before each match (default 30).",
+ },
+ after: {
+ type: "number",
+ description: "Seconds of context after each match (default 30).",
+ },
+ },
+ required: ["video_ids"],
+ additionalProperties: false,
+ },
+ },
+ {
name: "get_video_metadata",
description:
"Fetch one video's metadata (title, channel, upload date, duration, " +
@@ -99,11 +193,22 @@ const TOOLS = [
export function createServer(source: ShardSource): Server {
const server = new Server(
{ name: "yt-dlp-transcript-mcp", version: "0.1.0" },
- { capabilities: { tools: {} } },
+ { capabilities: { tools: {}, prompts: {} } },
);
server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS }));
+ server.setRequestHandler(ListPromptsRequestSchema, async () => ({
+ prompts: PROMPTS,
+ }));
+
+ server.setRequestHandler(GetPromptRequestSchema, async (req) => {
+ if (req.params.name !== "sweep") {
+ throw new Error(`unknown prompt: ${req.params.name}`);
+ }
+ return buildSweepPrompt((req.params.arguments ?? {}) as Record<string, unknown>);
+ });
+
server.setRequestHandler(CallToolRequestSchema, async (req) => {
const name = req.params.name;
const args = (req.params.arguments ?? {}) as Record<string, unknown>;
@@ -115,6 +220,8 @@ export function createServer(source: ShardSource): Server {
return await handleSearch(source, args);
case "get_transcript":
return await handleGetTranscript(source, args);
+ case "get_transcripts":
+ return await handleGetTranscripts(source, args);
case "get_video_metadata":
return await handleGetMetadata(source, args);
default:
@@ -151,12 +258,31 @@ async function handleSearch(
channel: typeof args.channel === "string" ? args.channel : undefined,
regex: args.regex === true,
limit: typeof args.limit === "number" ? args.limit : undefined,
+ offset: typeof args.offset === "number" ? args.offset : undefined,
+ includeSnippets: args.include_snippets !== false,
+ useAliases: args.use_aliases !== false,
+ maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined,
});
+
+ const aliasNote = describeFiredAliases(result.firedAliases);
+ const rangeStart = result.total === 0 ? 0 : result.offset + 1;
+ const rangeEnd = result.offset + result.hits.length;
const footer =
- `\n\n(scanned ${result.scanned.pages} page(s) across ` +
- `${result.scanned.channels} channel(s)${result.truncated ? "; scan truncated at the page cap" : ""})`;
+ `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` +
+ `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` +
+ `page(s) across ${result.scanned.channels} channel(s)` +
+ (result.truncated
+ ? "; coverage PARTIAL — scan hit the page/video cap"
+ : "") +
+ (aliasNote ? `; ${aliasNote}` : "") +
+ ")";
+
if (result.hits.length === 0) {
- return text(`No matches for "${query}".${footer}`);
+ const head =
+ result.total === 0
+ ? `No matches for "${query}".`
+ : `No matches in this page (offset ${result.offset} is past the ${result.total} total).`;
+ return text(`${head}${footer}`);
}
const blocks = result.hits.map((h) => {
const head =
@@ -169,10 +295,103 @@ async function handleSearch(
return snips ? `${head}\n${snips}` : head;
});
return text(
- `${result.hits.length} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}`,
+ `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}`,
);
}
+// A short human note describing the curated aliases a search expanded through,
+// e.g. "expanded via alias K-Cups → `(k|cake)[ -]?cup`". Empty when none fired.
+function describeFiredAliases(fired: SearchAlias[]): string {
+ if (fired.length === 0) return "";
+ const parts = fired.map((a) => `${a.label} → \`${a.suggestion}\``);
+ return `expanded via alias ${parts.join(", ")}`;
+}
+
+async function handleGetTranscripts(
+ source: ShardSource,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const rawIds = Array.isArray(args.video_ids) ? args.video_ids : [];
+ const ids = rawIds
+ .filter((v): v is string => typeof v === "string" && v.trim() !== "")
+ .map((v) => v.trim());
+ if (ids.length === 0) return errorText("video_ids is required (non-empty)");
+
+ const CAP = 20;
+ const dropped = ids.length > CAP ? ids.length - CAP : 0;
+ const batch = ids.slice(0, CAP);
+
+ const query = typeof args.query === "string" ? args.query.trim() : "";
+ const channelHint = typeof args.channel === "string" ? args.channel : undefined;
+ const timestamps = args.timestamps !== false;
+
+ // Build the (alias-aware) matcher once for the whole batch when windowing.
+ let matcher: ReturnType<typeof buildMatcher> | null = null;
+ if (query) {
+ const useAliases = args.use_aliases !== false && args.regex !== true;
+ const aliases = useAliases ? await source.loadAliases() : [];
+ matcher = buildMatcher({
+ query,
+ regex: args.regex === true,
+ useAliases,
+ aliases,
+ });
+ }
+
+ const before = typeof args.before === "number" ? args.before : 30;
+ const after = typeof args.after === "number" ? args.after : 30;
+
+ const blocks: string[] = [];
+ const missing: string[] = [];
+ for (const id of batch) {
+ const found = await findVideo(source, id, channelHint);
+ if (!found) {
+ missing.push(id);
+ continue;
+ }
+ const { record, ch } = found;
+ const head =
+ `## ${record.title || id}\n` +
+ `- video_id: ${id} | channel: ${ch.name}` +
+ (ch.siteTitle ? ` | site: ${ch.siteTitle}` : "") +
+ ` | uploaded: ${formatDate(record.uploadDate)}` +
+ (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "");
+
+ if (matcher) {
+ const { lines, matchCount } = getWindowedTranscript(record, matcher.match, {
+ before,
+ after,
+ timestamps,
+ });
+ const body =
+ matchCount === 0
+ ? "_(no lines matched the query in this transcript)_"
+ : `_(${matchCount} matching line(s), windowed)_\n${lines.join("\n")}`;
+ blocks.push(`${head}\n\n${body}`);
+ } else {
+ const md = transcriptToMarkdown(record, {
+ timestamps,
+ includeTags: true,
+ });
+ blocks.push(md.trim());
+ }
+ }
+
+ const notes: string[] = [];
+ if (query) {
+ const aliasNote = describeFiredAliases(matcher?.firedAliases ?? []);
+ if (aliasNote) notes.push(aliasNote);
+ }
+ if (missing.length > 0) notes.push(`not found: ${missing.join(", ")}`);
+ if (dropped > 0) notes.push(`${dropped} extra id(s) beyond the 20-cap dropped`);
+
+ const footer = notes.length > 0 ? `\n\n(${notes.join("; ")})` : "";
+ if (blocks.length === 0) {
+ return text(`No transcripts read.${footer}`);
+ }
+ return text(`${blocks.join("\n\n---\n\n")}${footer}`);
+}
+
async function handleGetTranscript(
source: ShardSource,
args: Record<string, unknown>,
@@ -213,3 +432,106 @@ async function handleGetMetadata(
),
);
}
+
+// ─── Prompts: the first-class `sweep` entry point ───
+// A single slash command that drives Claude Code to run the browser's
+// "corpus sweep" on plan usage: enumerate a query's full match set, batch the
+// windowed transcripts, and fold findings into a markdown report it maintains
+// with its own Write/Edit tools. The MCP stays read-only; the report is a file
+// in Claude's cwd. Mirrors the browser's accumulationSystemPrompt discipline.
+const PROMPTS = [
+ {
+ name: "sweep",
+ description:
+ "Run a whole-corpus sweep for a query: enumerate every matching video, " +
+ "batch-read the transcripts, and fold cited, cross-referenced findings " +
+ "into a markdown report — driven by Claude Code on plan usage, no API key.",
+ arguments: [
+ { name: "query", description: "Term or phrase to sweep for.", required: true },
+ {
+ name: "channel",
+ description: "Optional channel slug/name to restrict the sweep to.",
+ required: false,
+ },
+ {
+ name: "directive",
+ description:
+ "What to extract (default: 'key claims & contradictions').",
+ required: false,
+ },
+ {
+ name: "batch_size",
+ description: "Videos per batch (default 8).",
+ required: false,
+ },
+ {
+ name: "report_path",
+ description: "Report file to write (default ./sweep-report.md).",
+ required: false,
+ },
+ ],
+ },
+];
+
+function argStr(args: Record<string, unknown>, key: string): string | undefined {
+ const v = args[key];
+ return typeof v === "string" && v.trim() !== "" ? v.trim() : undefined;
+}
+
+function buildSweepPrompt(args: Record<string, unknown>) {
+ const query = argStr(args, "query");
+ if (!query) throw new Error("sweep requires a query argument");
+ const channel = argStr(args, "channel");
+ const directive = argStr(args, "directive") ?? "key claims & contradictions";
+ const batchSize = argStr(args, "batch_size") ?? "8";
+ const reportPath = argStr(args, "report_path") ?? "./sweep-report.md";
+
+ const channelClause = channel
+ ? ` restricted to channel "${channel}"`
+ : " across the whole corpus";
+
+ const text =
+ `Run a **corpus sweep** for the query **"${query}"**${channelClause}, ` +
+ `extracting **${directive}**, and maintain a running markdown report at ` +
+ `\`${reportPath}\`. You are the sweep engine — work through the whole match ` +
+ `set methodically, using the transcript MCP tools for evidence and your own ` +
+ `Write/Edit tools for the report. The MCP is read-only; never try to change ` +
+ `the archive.\n\n` +
+ `Follow these steps:\n\n` +
+ `1. **Search.** Call \`search_transcripts\` with query "${query}"` +
+ (channel ? ` and channel "${channel}"` : "") +
+ `. Curated aliases auto-expand the query — the footer reports which fired ` +
+ `(e.g. mis-transcribed spellings). Treat the *expanded* match set as your ` +
+ `target and mention the expansion in the report if one fired.\n\n` +
+ `2. **Enumerate the full worklist.** Page the complete set with ` +
+ `\`include_snippets: false\` and a rising \`offset\` (offset += limit) until ` +
+ `\`has_more\` is false — this gives you every id/title/channel/date cheaply. ` +
+ `Note the \`total\`. If the footer says coverage is PARTIAL (page/video cap), ` +
+ `say so in the report — the sweep is then a sample, not exhaustive.\n\n` +
+ `3. **Plan.** With N total matches and a batch size of ${batchSize}, that is ` +
+ `\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch count) ` +
+ `before you start.\n\n` +
+ `4. **Per batch**, for each group of up to ${batchSize} video ids:\n` +
+ ` - Call \`get_transcripts\` with those ids **and the query** so each ` +
+ `transcript comes back as bounded, timestamped excerpt windows around the ` +
+ `matches (alias-correct, high-signal).\n` +
+ ` - **Cross-reference** the batch against the report so far. Upsert ` +
+ `findings — claims, and contradictions with earlier claims — into ` +
+ `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` +
+ ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it each ` +
+ `batch), then **drop the raw transcript text** once folded — don't carry it ` +
+ `forward.\n\n` +
+ `5. **Finish.** Repeat to the end of the worklist, then write a short summary ` +
+ `section (how many videos covered, headline findings, any partial-coverage ` +
+ `caveat) and tell me the report path.`;
+
+ return {
+ description: `Corpus sweep for "${query}" → ${reportPath}`,
+ messages: [
+ {
+ role: "user" as const,
+ content: { type: "text" as const, text },
+ },
+ ],
+ };
+}
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -5,6 +5,10 @@ import {
type ChannelTranscriptsManifest,
} from "yt-dlp-transcript-common/lib/manifest";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import {
+ coerceAliasConfig,
+ type SearchAlias,
+} from "yt-dlp-transcript-common/lib/searchAliases";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
// mode (so results can be attributed to the owning member site); `key` is the
@@ -29,6 +33,12 @@ export interface ShardSource {
listChannels(): Promise<ChannelRef[]>;
transcriptsManifest(ch: ChannelRef): Promise<ChannelTranscriptsManifest>;
transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]>;
+ // The site's shipped curated search aliases (the same /search-aliases.json the
+ // viewer reads). Returns [] when the file is absent or malformed. Used to make
+ // caption search alias-aware, so a query for a term with a curated regex
+ // (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed
+ // spellings. Result is cached per source.
+ loadAliases(): Promise<SearchAlias[]>;
}
// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose
@@ -46,10 +56,25 @@ type HubCorpusJson = {
// listing the transcripts/ subdirectories so it works even pre-Layer-1.
export class LocalSource implements ShardSource {
readonly label: string;
+ private aliases?: SearchAlias[];
constructor(private dir: string) {
this.label = `local:${dir}`;
}
+ async loadAliases(): Promise<SearchAlias[]> {
+ if (this.aliases) return this.aliases;
+ try {
+ const raw = await readFile(
+ path.join(this.dir, "search-aliases.json"),
+ "utf8",
+ );
+ this.aliases = coerceAliasConfig(JSON.parse(raw)).aliases;
+ } catch {
+ this.aliases = []; // no/invalid file — search stays plain
+ }
+ return this.aliases;
+ }
+
async listChannels(): Promise<ChannelRef[]> {
try {
const raw = await readFile(path.join(this.dir, "corpus.json"), "utf8");
@@ -106,11 +131,25 @@ export class LocalSource implements ShardSource {
export class RemoteSource implements ShardSource {
readonly label: string;
private base: string;
+ private aliases?: SearchAlias[];
constructor(baseUrl: string) {
this.base = baseUrl.replace(/\/+$/, "");
this.label = `remote:${this.base}`;
}
+ async loadAliases(): Promise<SearchAlias[]> {
+ if (this.aliases) return this.aliases;
+ try {
+ const res = await fetch(`${this.base}/search-aliases.json`);
+ this.aliases = res.ok
+ ? coerceAliasConfig(await res.json()).aliases
+ : [];
+ } catch {
+ this.aliases = [];
+ }
+ return this.aliases;
+ }
+
private async getJson<T>(p: string): Promise<T> {
const res = await fetch(`${this.base}${p}`);
if (!res.ok) {
@@ -146,12 +185,28 @@ export class HubSource implements ShardSource {
readonly label: string;
private hubBase: string;
private members = new Map<string, RemoteSource>(); // siteId -> source
+ private aliases?: SearchAlias[];
constructor(hubUrl: string) {
this.hubBase = hubUrl.replace(/\/+$/, "");
this.label = `hub:${this.hubBase}`;
}
+ // A hub can ship its own /search-aliases.json (the merged federation-wide
+ // dictionary); if it doesn't, aliases are simply off for hub-wide search.
+ async loadAliases(): Promise<SearchAlias[]> {
+ if (this.aliases) return this.aliases;
+ try {
+ const res = await fetch(`${this.hubBase}/search-aliases.json`);
+ this.aliases = res.ok
+ ? coerceAliasConfig(await res.json()).aliases
+ : [];
+ } catch {
+ this.aliases = [];
+ }
+ return this.aliases;
+ }
+
private memberFor(siteId: string): RemoteSource {
const m = this.members.get(siteId);
if (!m) throw new Error(`unknown hub member site: ${siteId}`);
diff --git a/mcp/tsconfig.json b/mcp/tsconfig.json
@@ -1,7 +1,12 @@
{
"extends": "../tsconfig.base.json",
"compilerOptions": {
- "lib": ["esnext"],
+ // "dom" is types-only here — the server runs on Node, but a couple of the
+ // shared common/lib helpers we reuse (transcriptWindow → aiHandoff) carry a
+ // type-only import of a browser component, which pulls DOM ambient globals
+ // (window/indexedDB/…) into the type program. DOM lib satisfies those types
+ // without changing what runs.
+ "lib": ["esnext", "dom"],
"types": ["node"],
"baseUrl": ".",
"paths": {