commit 7f19a8f36afbd224804b58ef602b87ca7519957f
parent 1769513938185ed06752b17eddbbbc1a38d62014
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 20 Jul 2026 22:48:43 -0400
Merge worktree-feat+byo-ai-corpus: MCP runtime-switchable source + group/multi-channel sweep scoping
Read-only MCP additions: list_sources/use_source/reset_source (switch the active
corpus on the fly, persisted across reconnects; single hub member -> RemoteSource
with full groups/aliases, or a federated subset) plus additive channel/group
scoping for search_transcripts, list_channels, and the sweep prompt.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
10 files changed, 1593 insertions(+), 151 deletions(-)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,7 +1,8 @@
# Changelog
## [Unreleased]
-- **MCP server: run the corpus sweep through Claude Code itself — on plan usage, no API key.** The in-browser "corpus sweep" (fold every matching transcript into a running report) needs a BYO AI key, and a free key conks out fast on a real 30k-video corpus. The `mcp/` server now lets **Claude Code be the sweep engine** instead, via a first-class **`sweep` prompt** (a slash command, `/mcp__<name>__sweep query="k cups" channel="chrissie-mayr"`) plus the tools to drive it. `search_transcripts` gains **paging** — it reports the full `total` and `has_more`, so the whole match set can be enumerated with `offset` (and `include_snippets:false` for a cheap worklist) — and is now **alias-aware**: a plain query that matches a curated search alias also searches the alias regex (e.g. `k cups` → also `cake cup`), with the footer naming which aliases fired; reaching the scan cap is surfaced as **PARTIAL** coverage rather than hidden. A new **`get_transcripts`** tool batch-reads up to 20 videos in one call as bounded, timestamped **excerpt windows** around the matches (alias-correct) — or full transcripts without a query — so a sweep stays token-bounded. The `sweep` prompt walks Claude through search → enumerate → plan `ceil(N/batch)` batches → per-batch windowed read + cross-referenced upsert of cited findings (*title + [mm:ss]*) into a markdown report it maintains with its own Write/Edit tools. The MCP stays strictly **read-only**; only the report file is written, in Claude's own working directory. See `mcp/src/{search,source,server}.ts`, `mcp/src/search.test.ts`, and `mcp/README.md`.
+- **MCP server: switch which corpus you're reading on the fly — and it sticks across reconnects.** The MCP server was pinned to a single corpus at launch (`--hub` / `--remote` / `--local` or the `TRANSCRIPT_*` env), so re-aiming it at a different site or a local build meant editing the MCP config and restarting. Now, connected to (say) the **archilyzer hub**, you can retarget it from inside a session with three new **read-only** tools: **`list_sources`** shows the active corpus and, in a hub context, its member sites (`siteId · title · url`); **`use_source`** switches the active corpus — to a single hub member (`site:` — which becomes a plain single-site source and so regains **full channel-group + alias** support), a **federated subset** of the hub (`sites:[…]`, unknown tokens reported not dropped), or an arbitrary `remote:` URL / `local:` dir / `hub:` URL; and **`reset_source`** returns to the startup source. The selection is **persisted** to a small state file (under `$XDG_STATE_HOME/yt-dlp-transcript-mcp`, overridable with `TRANSCRIPT_MCP_STATE_DIR`) keyed by the **startup** source, so it survives a reconnect and two differently-configured servers keep independent selections. Everything the sweep/search tools do reads the current active source. Still strictly **read-only** — this only changes *which* already-published static shards are read, the same capability the startup flags already grant this locally-run tool; nothing is written to any corpus. See `mcp/src/{sources,sourceController,source,server,index}.ts`, `mcp/src/sourceController.test.ts`, and `mcp/README.md`.
+- **MCP server: run the corpus sweep through Claude Code itself — on plan usage, no API key.** The in-browser "corpus sweep" (fold every matching transcript into a running report) needs a BYO AI key, and a free key conks out fast on a real 30k-video corpus. The `mcp/` server now lets **Claude Code be the sweep engine** instead, via a first-class **`sweep` prompt** (a slash command, `/mcp__<name>__sweep query="k cups" channel="chrissie-mayr"`) plus the tools to drive it. `search_transcripts` gains **paging** — it reports the full `total` and `has_more`, so the whole match set can be enumerated with `offset` (and `include_snippets:false` for a cheap worklist) — and is now **alias-aware**: a plain query that matches a curated search alias also searches the alias regex (e.g. `k cups` → also `cake cup`), with the footer naming which aliases fired; reaching the scan cap is surfaced as **PARTIAL** coverage rather than hidden. A new **`get_transcripts`** tool batch-reads up to 20 videos in one call as bounded, timestamped **excerpt windows** around the matches (alias-correct) — or full transcripts without a query — so a sweep stays token-bounded. The `sweep` prompt walks Claude through search → enumerate → plan `ceil(N/batch)` batches → per-batch windowed read + cross-referenced upsert of cited findings (*title + [mm:ss]*) into a markdown report it maintains with its own Write/Edit tools. The sweep is now **group-aware and multi-channel**: `search_transcripts` scope is additive over one-or-more channels (`channel`/`channels`) and/or channel **groups** (`group`/`groups`, matched by group **id or display name** — `group="other"` ≡ `group="Extended Universe"` — and expanded to the group's channels), with the footer naming the resolved scope and flagging any channel/group token that matched nothing so a typo isn't silently a whole-corpus scan; `get_transcripts` takes matching `channels` lookup hints; and `list_channels` now organizes channels under their groups with a compact `id · name · N channels` cheat-sheet. Crucially the `sweep` prompt is now **pick-first**: invoked with **no** scope it makes Claude list the groups/channels and **ask which to sweep — or to confirm the whole corpus — before enumerating**, rather than silently scanning everything (an explicit `search_transcripts` call with no selector still means "all"). (Hub mode defers per-site group resolution — multi-channel scoping still works there.) The MCP stays strictly **read-only**; only the report file is written, in Claude's own working directory. See `mcp/src/{search,source,server}.ts`, `mcp/src/search.test.ts`, and `mcp/README.md`.
- **"Ask AI" now paces itself to your key's rate limit instead of failing.** A whole-corpus sweep on a free-tier key used to fire provider calls as fast as the loop could produce them, blow straight through the per-minute request cap, and stop dead on the first HTTP 429 (and a rate-limit *mid-answer* surfaced as a hard error, because only the sweep path ever handled 429). Now a single client-side limiter sits behind **every** AI call: it **spaces requests** to a conservative, free-tier-safe **requests-per-minute** default chosen per model (e.g. Gemini `*-pro` → 5/min, `*flash` → 10/min; Claude/OpenAI higher), so a paid key runs fast and a free key just runs *slowly* rather than erroring. When a 429 does land, the limiter **honours the provider's own retry hint** (the `Retry-After` header, or Gemini's `RetryInfo` retry-delay) — or an exponential backoff — and **re-issues the request** (safe for streaming: the retry happens before any answer text is emitted), and it **self-lowers** the rate after a 429 so later calls pace slower. Only a 429 that outlasts the retries falls through to the existing **pause/checkpoint** (the right home for a daily-quota cap you resume tomorrow). Provider settings gain an editable **Requests per minute** field (with the effective spacing, e.g. *"≈12s between requests"*, and a reset-to-default), and a paced sweep shows a **"Rate limited — retrying in Ns"** line in its progress strip so it never looks frozen. See `export/app/lib/rateLimit.ts` (the whole mechanism), `export/app/lib/askProvider.ts` (`PausableError` + `acquire`/retry in `askStream`, 429 in `ensureOk`), `export/app/lib/nativeTools/shared.ts` (`acquire`/retry in `postJson`), `export/app/ask/{useAskChat,ProviderSettings,PinnedResultsPanel,AskChat}.tsx`, and `export/e2e/ask-workspace.spec.ts`.
- **"Ask AI" — an integrated grounding workspace: pick your AI target, sweep any size, save & resume chats.** Search and the chat used to behave like two tabs, grounding was all-or-nothing, a sweep had hard-coded caps and couldn't be paused, and a conversation lived in a single unnamed slot that "New chat" wiped. Now they're one surface:
- **Grounding palette (the signature control).** A first-class `Whole search ⇄ Selection` toggle in the chat always states *what the AI is looking at*, alongside an action cluster — **Ask**, **Sweep**, and preset directives (*Summary*, *Contradictions*, *Timeline*) that pre-fill the composer/sweep box. "Selection" grounds in a **hand-picked subset** of results shown as dismissible chips; an empty selection parks on Whole search, so nothing changes for people who never select.
diff --git a/mcp/README.md b/mcp/README.md
@@ -12,16 +12,32 @@ reads the site's already-published static JSON shards (`corpus.json` +
| Tool | What it does |
|------|--------------|
-| `list_channels` | List channels (name, slug, video count; site in hub mode). |
-| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Optional `channel` filter. |
-| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. |
+| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
+| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
+| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). |
| `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. |
+| `list_sources` | Show the **active** corpus (label + kind + target) and, with a hub context, its member sites (`siteId · title · url`, marking the current subset) so you can pick one to switch to. |
+| `use_source` | **Switch which corpus is read**, on the fly — a hub member (`site`), a hub subset (`sites`), or an arbitrary `remote` URL / `local` dir / `hub` URL. Persists across reconnects. |
+| `reset_source` | Return to the source the server was started with and clear the persisted selection. |
### `search_transcripts`
-Beyond `query`, `channel`, `regex`, and `limit`:
-
+Beyond `query`, `regex`, and `limit`:
+
+- **Scope** — restrict the scan to a subset of the corpus. All scope fields are
+ optional and **additive** (the search runs over the union); with none, it
+ covers everything.
+ - **`channel`** / **`channels`** — one or a list of channel slugs/names.
+ - **`group`** / **`groups`** — one or a list of channel groups, matched by
+ **group id or display name** (case-insensitive), expanded to the channels in
+ that group. e.g. `group="other"` and `group="Extended Universe"` resolve the
+ same set. (In hub mode groups are deferred — a group token reports "unknown"
+ while channel scoping still works.)
+
+ The footer names the resolved scope (e.g. *"scope: group Extended Universe (7
+ channels)"* or *"scope: 6 channels"*) and flags any channel/group token that
+ matched nothing — so a typo is surfaced, not silently a whole-corpus scan.
- **`offset`** (default 0) — skip this many matches. The result footer reports the
full `total` and `has_more`, so you can enumerate a query's *entire* match set:
page with `offset += limit` until `has_more` is `no`.
@@ -46,6 +62,9 @@ optional `regex`/`use_aliases`), each transcript is reduced to bounded windows o
timestamped lines around the matches (`before`/`after` seconds, default 30) —
high-signal context for folding a batch into a report. Without a query, each
video's full transcript comes back as markdown. Missing ids are reported inline.
+`channel`/`channels` are optional owning-channel hints that speed the per-id
+lookup when a batch spans several channels (the ids are already scoped by the
+search that produced them).
## The `sweep` prompt
@@ -55,21 +74,35 @@ browser does with a BYO AI key, but driven by your Claude **plan usage** (no API
key) and with the report written to a file.
Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query`
-(required), `channel?`, `directive?` (default *"key claims & contradictions"*),
-`batch_size?` (default 8), `report_path?` (default `./sweep-report.md`).
+(required), `channel?`, `channels?` (comma-separated slugs/names), `group?` (a
+channel group id or name), `directive?` (default *"key claims &
+contradictions"*), `batch_size?` (default 8), `report_path?` (default
+`./sweep-report.md`).
```
/mcp__rekietalyzer__sweep query="k cups" channel="chrissie-mayr"
+/mcp__rekietalyzer__sweep query="k cups" group="other" # a whole group
+/mcp__rekietalyzer__sweep query="k cups" group="Extended Universe" # …by name
+/mcp__rekietalyzer__sweep query="k cups" # pick-first (see below)
```
-The prompt instructs Claude Code to: **search** (aliases auto-expand) → **enumerate**
-the full worklist by paging with `include_snippets:false` until `has_more` is
-false → **plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts`
-for windowed context, cross-reference against the report so far, and upsert
-findings (claims, contradictions, sources cited as *title + [mm:ss]*) into
-well-titled `##` sections of the report file with its own Write/Edit tools →
-finish with a short summary. The MCP stays read-only; only the report file is
-written, in Claude's working directory.
+**Pick-first when no scope is given.** Because a whole-corpus sweep can be a lot
+of work, invoking `sweep` **without** a `channel`/`channels`/`group` makes Claude
+FIRST call `list_channels`, present the groups and their channels, and ask you
+which group(s)/channel(s) to sweep — or to confirm **all** for the whole corpus —
+*before* it enumerates anything. Supplying any scope arg skips the prompt and
+sweeps that scope directly. A group selector accepts either an id or the display
+name (`group="other"` ≡ `group="Extended Universe"`).
+
+Once scoped, the prompt instructs Claude Code to: **search** (aliases
+auto-expand; the footer names the resolved scope) → **enumerate** the full
+worklist by paging with `include_snippets:false` until `has_more` is false →
+**plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts` for
+windowed context, cross-reference against the report so far, and upsert findings
+(claims, contradictions, sources cited as *title + [mm:ss]*) into well-titled
+`##` sections of the report file with its own Write/Edit tools → finish with a
+short summary. The MCP stays read-only; only the report file is written, in
+Claude's working directory.
## Data source (pick one)
@@ -81,6 +114,46 @@ Resolved from flags or env — precedence hub > remote > local:
| Remote | `--remote <url>` | `TRANSCRIPT_SITE_URL` | One deployed site origin. |
| Local | `--local <dir>` | `TRANSCRIPT_LOCAL_DIR` | A composed public dir on disk (default `./export/public`). |
+## Switching the source on the fly
+
+The server is added to your MCP client **once**, pointed at a startup source
+(above). But you don't have to edit the config and restart to aim it elsewhere —
+you can switch the **active corpus** from inside a session with three tools, and
+your choice **persists across reconnects**.
+
+- **`list_sources`** — shows the active source (label + kind + target). When
+ there's a hub in play (you started against a `--hub`, or switched to one) it
+ also lists the hub's member sites as `siteId · title · url`, marking which are
+ in the current subset — so you can see what you can pick.
+- **`use_source`** — switch the active source. Provide **exactly one** target:
+ - **`site: "<siteId | title>"`** — scope to a single hub member (matched by
+ siteId or display title, case-insensitive). This becomes a plain
+ single-site source, so it regains **full channel-group + alias** support.
+ - **`sites: ["<a>", "<b>"]`** — federate a **subset** of the hub's members.
+ Unknown tokens are reported, not silently dropped.
+ - **`remote: "<url>"`** — an arbitrary deployed site origin.
+ - **`local: "<dir>"`** — a composed public dir on disk.
+ - **`hub: "<url>"`** — an arbitrary hub URL to federate over.
+- **`reset_source`** — return to the startup source and clear the persisted
+ selection.
+
+For example, connected to the archilyzer hub: `list_sources` to see the members,
+then `use_source site:"jeralyzer"` to work in that one site (with its groups), or
+`use_source sites:["jeralyzer","rekietalyzer"]` to federate just those two.
+
+**Persistence.** The active selection is written to a small state file so a
+reconnect (or a fresh process with the same startup flags) resumes it. The file
+lives under `$XDG_STATE_HOME/yt-dlp-transcript-mcp` (fallback
+`~/.local/state/yt-dlp-transcript-mcp`), overridable with
+`TRANSCRIPT_MCP_STATE_DIR`. It's keyed by the **startup** source, so two
+differently-configured servers (archilyzer / rekietalyzer / a local build) keep
+independent selections and don't clobber each other.
+
+**Still read-only.** `use_source` only changes *which* already-published static
+shards are read — accepting an arbitrary `remote`/`local`/`hub` at call time is
+the same capability the startup flags already grant this locally-run tool.
+Nothing is ever written to any corpus.
+
## Run it
From the monorepo (cwd is set to the package dir by `pnpm --filter … exec`, so
diff --git a/mcp/src/index.ts b/mcp/src/index.ts
@@ -1,52 +1,15 @@
#!/usr/bin/env tsx
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
-import {
- LocalSource,
- RemoteSource,
- HubSource,
- type ShardSource,
-} from "./source";
+import { resolveSourceSpec } from "./sources";
import { createServer } from "./server";
-
-// Resolve the data source from flags or env. Precedence: hub > remote > local >
-// positional > default. All logging goes to stderr — stdout is the MCP
-// JSON-RPC channel and must not be polluted.
-//
-// --hub <url> | TRANSCRIPT_HUB_URL federate over a hub's member sites
-// --remote <url> | TRANSCRIPT_SITE_URL one deployed site origin
-// --local <dir> | TRANSCRIPT_LOCAL_DIR a composed public dir on disk
-// <positional> an http(s) URL → remote, otherwise a local dir
-// (default) ./export/public
-export function resolveSource(argv: string[]): ShardSource {
- const flag = (name: string): string | undefined => {
- const i = argv.indexOf(name);
- return i >= 0 && i + 1 < argv.length ? argv[i + 1] : undefined;
- };
-
- const hub = flag("--hub") ?? process.env.TRANSCRIPT_HUB_URL;
- if (hub) return new HubSource(hub);
-
- const remote =
- flag("--remote") ?? flag("--url") ?? process.env.TRANSCRIPT_SITE_URL;
- if (remote) return new RemoteSource(remote);
-
- const local = flag("--local") ?? process.env.TRANSCRIPT_LOCAL_DIR;
- if (local) return new LocalSource(local);
-
- const positional = argv.find((a) => !a.startsWith("-"));
- if (positional) {
- return /^https?:\/\//i.test(positional)
- ? new RemoteSource(positional)
- : new LocalSource(positional);
- }
-
- return new LocalSource("export/public");
-}
+import { SourceController } from "./sourceController";
async function main(): Promise<void> {
- const source = resolveSource(process.argv.slice(2));
- console.error(`[yt-dlp-transcript-mcp] source: ${source.label}`);
- const server = createServer(source);
+ const spec = resolveSourceSpec(process.argv.slice(2));
+ const controller = new SourceController(spec);
+ await controller.init();
+ console.error(`[yt-dlp-transcript-mcp] source: ${controller.current.label}`);
+ const server = createServer(controller);
await server.connect(new StdioServerTransport());
console.error("[yt-dlp-transcript-mcp] ready on stdio");
}
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -6,8 +6,14 @@ import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/ma
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
-import type { ChannelRef, ShardSource } from "./source";
-import { searchTranscripts, getWindowedTranscript, buildMatcher } from "./search";
+import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups";
+import type { ChannelGroups, ChannelRef, ShardSource } from "./source";
+import {
+ searchTranscripts,
+ getWindowedTranscript,
+ buildMatcher,
+ resolveSelectedChannels,
+} from "./search";
import { createServer } from "./server";
// ─── A tiny in-memory ShardSource for the tests ───
@@ -73,6 +79,20 @@ const CHAN_B: TranscriptDetail[] = [
),
];
+// Two groups + a (blank-named) default. chan-a is in "other" (Extended
+// Universe); chan-b carries an unknown "ghost" groupId that must fold onto the
+// default group (exactly resolveChannelGroupId's semantics).
+const STUB_GROUPS: ChannelGroup[] = [
+ { id: "default", name: "", selectedByDefault: true, order: 0 },
+ {
+ id: "other",
+ name: "Extended Universe",
+ selectedByDefault: false,
+ order: 1,
+ description: "Related characters",
+ },
+];
+
class StubSource implements ShardSource {
readonly label = "stub";
constructor(private aliases: SearchAlias[] = [K_CUPS_ALIAS]) {}
@@ -81,10 +101,15 @@ class StubSource implements ShardSource {
return this.aliases;
}
+ async loadGroups(): Promise<ChannelGroups> {
+ return { groups: STUB_GROUPS, defaultGroupId: "default" };
+ }
+
async listChannels(): Promise<ChannelRef[]> {
return [
- { key: "chan-a", slug: "chan-a", name: "Channel A" },
- { key: "chan-b", slug: "chan-b", name: "Channel B" },
+ { key: "chan-a", slug: "chan-a", name: "Channel A", groupId: "other" },
+ // "ghost" is not a defined group → folds onto the default group.
+ { key: "chan-b", slug: "chan-b", name: "Channel B", groupId: "ghost" },
];
}
@@ -176,6 +201,81 @@ test("include_snippets:false omits snippet text but keeps the match count", asyn
assert.ok(noSnips.hits[0].matches >= 1, "match count still reported");
});
+// ─── Scope: multi-channel + channel groups ───
+
+test("scope: channels union scopes to both; an unknown channel is reported", async () => {
+ const src = new StubSource();
+ const sel = await resolveSelectedChannels(src, {
+ channels: ["chan-a", "chan-b", "chan-z"],
+ });
+ assert.deepEqual(
+ sel.channels.map((c) => c.key).sort(),
+ ["chan-a", "chan-b"],
+ "the two real channels are selected",
+ );
+ assert.deepEqual(sel.unknownChannels, ["chan-z"], "the typo is surfaced");
+ assert.equal(sel.all, false);
+});
+
+test("scope: group by id and by name expands to member channels", async () => {
+ const src = new StubSource();
+ const byId = await resolveSelectedChannels(src, { group: "other" });
+ assert.deepEqual(byId.channels.map((c) => c.key), ["chan-a"]);
+ assert.deepEqual(byId.matchedGroups.map((g) => g.id), ["other"]);
+
+ const byName = await resolveSelectedChannels(src, { group: "Extended Universe" });
+ assert.deepEqual(byName.channels.map((c) => c.key), ["chan-a"], "name resolves the same group");
+});
+
+test("scope: an unknown/missing channel groupId folds onto the default group", async () => {
+ const src = new StubSource();
+ // chan-b's "ghost" groupId isn't a defined group, so it resolves to default.
+ const def = await resolveSelectedChannels(src, { group: "default" });
+ assert.deepEqual(def.channels.map((c) => c.key), ["chan-b"]);
+});
+
+test("scope: an unknown group is reported, not silently a whole-corpus scan", async () => {
+ const src = new StubSource();
+ const sel = await resolveSelectedChannels(src, { group: "nope" });
+ assert.deepEqual(sel.unknownGroups, ["nope"]);
+ assert.deepEqual(sel.channels, []);
+ assert.equal(sel.all, false);
+});
+
+test("scope: channels + group combine as a deduped union", async () => {
+ const src = new StubSource();
+ // chan-a named explicitly AND a member of group "other" → deduped to one.
+ const overlap = await resolveSelectedChannels(src, {
+ channel: "chan-a",
+ group: "other",
+ });
+ assert.deepEqual(overlap.channels.map((c) => c.key), ["chan-a"]);
+
+ // chan-b explicit + group "other" (chan-a) → the union of both.
+ const union = await resolveSelectedChannels(src, {
+ channels: ["chan-b"],
+ group: "other",
+ });
+ assert.deepEqual(union.channels.map((c) => c.key).sort(), ["chan-a", "chan-b"]);
+});
+
+test("scope: no selector means all channels (unchanged primitive behavior)", async () => {
+ const src = new StubSource();
+ const sel = await resolveSelectedChannels(src, {});
+ assert.equal(sel.all, true);
+ assert.deepEqual(sel.channels.map((c) => c.key).sort(), ["chan-a", "chan-b"]);
+});
+
+test("search: a group scope restricts the scan to its members", async () => {
+ const src = new StubSource();
+ // Group "other" is just chan-a → the five chan-a coffee videos, not b1.
+ const res = await searchTranscripts(src, { query: "coffee", group: "other", limit: 20 });
+ assert.equal(res.total, 5);
+ assert.ok(res.hits.every((h) => h.channelSlug === "chan-a"));
+ assert.equal(res.selection.channelCount, 1);
+ assert.deepEqual(res.selection.matchedGroups.map((g) => g.id), ["other"]);
+});
+
// ─── (d) getWindowedTranscript ───
test("getWindowedTranscript: windows a bounded region around the match", async () => {
@@ -229,10 +329,49 @@ test("server: search_transcripts footer reports total, has_more, and the alias",
const out = firstText(res);
assert.match(out, /total 2 match/);
assert.match(out, /has_more: no/);
+ assert.match(out, /scope: whole corpus/);
assert.match(out, /expanded via alias K-Cups/);
await client.close();
});
+test("server: search_transcripts footer names a group scope and its channel count", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", group: "other", limit: 20 },
+ });
+ const out = firstText(res);
+ assert.match(out, /total 5 match/);
+ assert.match(out, /scope: group Extended Universe \(1 channel/);
+ await client.close();
+});
+
+test("server: search_transcripts footer flags a typo'd group as unknown", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", group: "nope", limit: 20 },
+ });
+ const out = firstText(res);
+ assert.match(out, /unknown group\(s\): nope/);
+ await client.close();
+});
+
+test("server: list_channels groups channels under headers and lists the groups", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({ name: "list_channels", arguments: {} });
+ const out = firstText(res);
+ // Group headers (blank-named default falls back to its id).
+ assert.match(out, /### Extended Universe \(1 channel\(s\); not selected by default\)/);
+ assert.match(out, /### default \(1 channel\(s\)\)/);
+ assert.match(out, /Channel A/);
+ assert.match(out, /Channel B/);
+ // The compact selector cheat-sheet.
+ assert.match(out, /Groups \(select by id or name\)/);
+ assert.match(out, /other · Extended Universe · 1 channel\(s\) · not selected by default/);
+ await client.close();
+});
+
test("server: get_transcripts windows with a query and reports missing ids", async () => {
const client = await connectClient(new StubSource());
const res = await client.callTool({
@@ -265,7 +404,15 @@ test("server: the sweep prompt lists with its arguments and renders the query",
const sweep = list.prompts.find((p) => p.name === "sweep");
assert.ok(sweep, "sweep prompt is listed");
const names = (sweep!.arguments ?? []).map((a) => a.name);
- assert.deepEqual(names.sort(), ["batch_size", "channel", "directive", "query", "report_path"]);
+ assert.deepEqual(names.sort(), [
+ "batch_size",
+ "channel",
+ "channels",
+ "directive",
+ "group",
+ "query",
+ "report_path",
+ ]);
const got = await client.getPrompt({
name: "sweep",
@@ -277,3 +424,28 @@ test("server: the sweep prompt lists with its arguments and renders the query",
assert.match((msg as { text: string }).text, /chan-a/);
await client.close();
});
+
+test("server: the sweep prompt reflects a supplied group scope", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", group: "other" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /scoped to group "other"/);
+ assert.match(text, /group "other"/);
+ await client.close();
+});
+
+test("server: the sweep prompt with no scope instructs a group/channel pick first", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /Choose the scope first/);
+ assert.match(text, /list_channels/);
+ assert.match(text, /confirm \*\*all\*\*/);
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -10,10 +10,120 @@ import {
mergeSnippets,
type WindowSnippet,
} from "yt-dlp-transcript-common/lib/transcriptWindow";
+import {
+ resolveChannelGroupId,
+ type ChannelGroup,
+} from "yt-dlp-transcript-common/lib/channelGroups";
import type { ChannelRef, ShardSource } from "./source";
export type Snippet = { clock: string; seconds: number; text: string };
+// A scope selector for a search/sweep: any mix of channel handles (slug / key /
+// name) and group handles (id / name). All fields are optional and additive —
+// the resolved scope is the UNION of every matched channel and every matched
+// group's members. No tokens at all means "all channels" (the primitive's
+// default).
+export type ChannelSelector = {
+ channel?: string;
+ channels?: string[];
+ group?: string;
+ groups?: string[];
+};
+
+// The outcome of resolving a ChannelSelector against a source: the deduped set
+// of channels to scan, which group tokens matched (for a scope note), and any
+// tokens that matched nothing (surfaced as a warning so a typo isn't silently a
+// whole-corpus scan). `all` is true only when no selector was given.
+export type ResolvedSelection = {
+ channels: ChannelRef[];
+ matchedGroups: ChannelGroup[];
+ unknownChannels: string[];
+ unknownGroups: string[];
+ all: boolean;
+};
+
+function selectorTokens(one: string | undefined, many: string[] | undefined): string[] {
+ return [...(one ? [one] : []), ...(many ?? [])]
+ .map((t) => (typeof t === "string" ? t.trim() : ""))
+ .filter((t) => t !== "");
+}
+
+// Resolve a ChannelSelector to a concrete set of channels. Channel tokens match
+// slug|key|name (case-insensitive); group tokens match a group by id or name
+// (case-insensitive; blank group names are unselectable) and expand to the
+// channels whose resolved groupId is that group — mirroring the browser's
+// group→slugs mapping (resolveChannelGroupId folds unknown/absent groupIds onto
+// the default). The result is the deduped union; unmatched tokens are reported.
+export async function resolveSelectedChannels(
+ source: ShardSource,
+ sel: ChannelSelector,
+): Promise<ResolvedSelection> {
+ const allChannels = await source.listChannels();
+ const channelTokens = selectorTokens(sel.channel, sel.channels);
+ const groupTokens = selectorTokens(sel.group, sel.groups);
+
+ // No selector → all channels (unchanged primitive behavior).
+ if (channelTokens.length === 0 && groupTokens.length === 0) {
+ return {
+ channels: allChannels,
+ matchedGroups: [],
+ unknownChannels: [],
+ unknownGroups: [],
+ all: true,
+ };
+ }
+
+ const selected = new Map<string, ChannelRef>(); // key -> ref (dedupe)
+ const unknownChannels: string[] = [];
+ const unknownGroups: string[] = [];
+ const matchedGroups: ChannelGroup[] = [];
+
+ for (const token of channelTokens) {
+ const want = token.toLowerCase();
+ const matches = allChannels.filter(
+ (c) =>
+ c.slug.toLowerCase() === want ||
+ c.key.toLowerCase() === want ||
+ c.name.toLowerCase() === want,
+ );
+ if (matches.length === 0) {
+ unknownChannels.push(token);
+ continue;
+ }
+ for (const c of matches) selected.set(c.key, c);
+ }
+
+ if (groupTokens.length > 0) {
+ const { groups, defaultGroupId } = await source.loadGroups();
+ for (const token of groupTokens) {
+ const want = token.toLowerCase();
+ const group = groups.find(
+ (g) =>
+ g.id.toLowerCase() === want ||
+ (g.name.trim() !== "" && g.name.toLowerCase() === want),
+ );
+ if (!group) {
+ unknownGroups.push(token);
+ continue;
+ }
+ if (!matchedGroups.some((g) => g.id === group.id)) matchedGroups.push(group);
+ for (const c of allChannels) {
+ if (resolveChannelGroupId(c.groupId, groups, defaultGroupId) === group.id) {
+ selected.set(c.key, c);
+ }
+ }
+ }
+ }
+
+ return {
+ channels: [...selected.values()],
+ matchedGroups,
+ unknownChannels,
+ unknownGroups,
+ all: false,
+ };
+}
+
export type SearchHit = {
videoId: string;
channelSlug: string;
@@ -42,6 +152,16 @@ export type SearchResult = {
// Coverage is partial — the page cap (MAX_PAGES) or the video cap
// (HARD_VIDEO_CAP) was reached before the corpus was fully scanned.
truncated: boolean;
+ // How the scope selector resolved (for the tool's scope note): whether it was
+ // whole-corpus, the channels/groups it matched, and any tokens that matched
+ // nothing (a typo'd channel/group is surfaced, not silently a full scan).
+ selection: {
+ all: boolean;
+ channelCount: number;
+ matchedGroups: ChannelGroup[];
+ unknownChannels: string[];
+ unknownGroups: string[];
+ };
};
// A hard ceiling on shard pages fetched per query so a rare term over a large
@@ -123,6 +243,9 @@ export async function searchTranscripts(
opts: {
query: string;
channel?: string;
+ channels?: string[];
+ group?: string;
+ groups?: string[];
regex?: boolean;
limit?: number;
offset?: number;
@@ -151,16 +274,13 @@ export async function searchTranscripts(
aliases,
});
- let channels = await source.listChannels();
- if (opts.channel) {
- const want = opts.channel.toLowerCase();
- channels = channels.filter(
- (c) =>
- c.slug.toLowerCase() === want ||
- c.key.toLowerCase() === want ||
- c.name.toLowerCase() === want,
- );
- }
+ const selection = await resolveSelectedChannels(source, {
+ channel: opts.channel,
+ channels: opts.channels,
+ group: opts.group,
+ groups: opts.groups,
+ });
+ const channels = selection.channels;
const all: SearchHit[] = [];
let pagesScanned = 0;
@@ -233,6 +353,13 @@ export async function searchTranscripts(
firedAliases,
scanned: { channels: channelsScanned, pages: pagesScanned },
truncated,
+ selection: {
+ all: selection.all,
+ channelCount: selection.channels.length,
+ matchedGroups: selection.matchedGroups,
+ unknownChannels: selection.unknownChannels,
+ unknownGroups: selection.unknownGroups,
+ },
};
}
@@ -275,20 +402,25 @@ export function getWindowedTranscript(
// Locate a single video across the source's channels via each channel's
// slugToPage map, returning the full record + its channel. `channelHint`
-// (slug/key/name) short-circuits the scan when the caller knows the channel.
+// (slug/key/name, or a list of them) short-circuits the scan when the caller
+// knows the channel(s) — it's a lookup accelerator only, so an unmatched hint
+// silently falls back to a full scan.
export async function findVideo(
source: ShardSource,
videoId: string,
- channelHint?: string,
+ channelHint?: string | string[],
): Promise<{ ch: ChannelRef; record: TranscriptDetail } | null> {
let channels = await source.listChannels();
- if (channelHint) {
- const want = channelHint.toLowerCase();
+ const hints = (Array.isArray(channelHint) ? channelHint : channelHint ? [channelHint] : [])
+ .map((h) => (typeof h === "string" ? h.trim().toLowerCase() : ""))
+ .filter((h) => h !== "");
+ if (hints.length > 0) {
+ const want = new Set(hints);
const filtered = channels.filter(
(c) =>
- c.slug.toLowerCase() === want ||
- c.key.toLowerCase() === want ||
- c.name.toLowerCase() === want,
+ want.has(c.slug.toLowerCase()) ||
+ want.has(c.key.toLowerCase()) ||
+ want.has(c.name.toLowerCase()),
);
// Prefer the hinted channel(s), but fall back to a full scan if not found.
if (filtered.length > 0) channels = filtered;
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -8,12 +8,21 @@ import {
import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
-import type { ShardSource } from "./source";
+import {
+ sortGroups,
+ resolveChannelGroupId,
+ FALLBACK_GROUP,
+ type ChannelGroup,
+} from "yt-dlp-transcript-common/lib/channelGroups";
+import { HubSource, type ChannelRef, type HubSite, type ShardSource } from "./source";
+import type { SourceSpec } from "./sources";
+import type { SourceController } from "./sourceController";
import {
searchTranscripts,
findVideo,
buildMatcher,
getWindowedTranscript,
+ type SearchResult,
} from "./search";
type ToolResult = {
@@ -32,9 +41,12 @@ const TOOLS = [
{
name: "list_channels",
description:
- "List the channels in this transcript archive (in hub mode, across every " +
- "federated member site). Returns each channel's display name, slug, video " +
- "count, and owning site.",
+ "List the channels in this transcript archive, organized under their " +
+ "channel groups (in hub mode, across every federated member site). Returns " +
+ "each channel's display name, slug, video count, and owning site, plus a " +
+ "compact list of the groups (id · name · channel count) and whether each " +
+ "is selected by default — so you can pick a group or channels to scope a " +
+ "search or sweep to.",
inputSchema: { type: "object", properties: {}, additionalProperties: false },
},
{
@@ -42,13 +54,17 @@ const TOOLS = [
description:
"Search transcript captions for a term or phrase and return matching " +
"videos with timestamped snippets. Substring match by default; set regex " +
- "to true for a case-insensitive regular expression. Optionally restrict to " +
- "one channel (by slug or name). Alias-aware: a plain query with a curated " +
- "search alias auto-expands to the alias regex (e.g. 'k cups' also matches " +
- "'cake cup') and the footer reports which aliases fired. Pageable: returns " +
- "the total match count and whether more pages exist, so a caller can " +
- "enumerate a query's full match set with offset. Set include_snippets to " +
- "false for a cheap worklist (no cue text).",
+ "to true for a case-insensitive regular expression. Scope is optional and " +
+ "additive: restrict to one or more channels (channel / channels, by slug " +
+ "or name) and/or one or more channel groups (group / groups, by group id " +
+ "or name) — the search runs over the UNION, and with no scope it covers " +
+ "the whole corpus. Alias-aware: a plain query with a curated search alias " +
+ "auto-expands to the alias regex (e.g. 'k cups' also matches 'cake cup') " +
+ "and the footer reports which aliases fired. Pageable: returns the total " +
+ "match count and whether more pages exist, so a caller can enumerate a " +
+ "query's full match set with offset. Set include_snippets to false for a " +
+ "cheap worklist (no cue text). The footer names the resolved scope and " +
+ "flags any channel/group token that matched nothing.",
inputSchema: {
type: "object",
properties: {
@@ -57,6 +73,26 @@ const TOOLS = [
type: "string",
description: "Optional channel slug or name to restrict the search to.",
},
+ channels: {
+ type: "array",
+ items: { type: "string" },
+ description:
+ "Optional list of channel slugs/names to restrict the search to " +
+ "(union with channel/group/groups).",
+ },
+ group: {
+ type: "string",
+ description:
+ "Optional channel group (by id or display name, e.g. 'other' or " +
+ "'Extended Universe') to expand to its member channels.",
+ },
+ groups: {
+ type: "array",
+ items: { type: "string" },
+ description:
+ "Optional list of channel groups (ids or names) to expand to their " +
+ "member channels (union with the other scope fields).",
+ },
regex: {
type: "boolean",
description:
@@ -154,6 +190,13 @@ const TOOLS = [
type: "string",
description: "Optional owning channel slug/name to speed the lookup.",
},
+ channels: {
+ type: "array",
+ items: { type: "string" },
+ description:
+ "Optional owning channel slugs/names to speed the lookup when the " +
+ "batch spans several channels (ids are already scoped by search).",
+ },
timestamps: {
type: "boolean",
description: "Prefix each line with a timestamp (default true).",
@@ -186,11 +229,115 @@ const TOOLS = [
additionalProperties: false,
},
},
+ {
+ name: "list_sources",
+ description:
+ "Show the ACTIVE corpus this server is currently reading (label + kind + " +
+ "target). When there is a hub context, also lists the hub's member sites " +
+ "(siteId · title · url), marking which are in the current subset — so you " +
+ "can pick a site or a subset to switch to with use_source. You can also " +
+ "switch to an arbitrary remote site URL, local dir, or hub URL. The chosen " +
+ "source persists across reconnects. Strictly read-only either way.",
+ inputSchema: { type: "object", properties: {}, additionalProperties: false },
+ },
+ {
+ name: "use_source",
+ description:
+ "Switch which corpus this server reads — on the fly, and the choice " +
+ "persists across reconnects. Provide EXACTLY ONE target: `site` (a hub " +
+ "member by siteId or title — becomes a single-site source with full group/" +
+ "alias support), `sites` (a list of hub members by siteId/title — a " +
+ "federated subset of the hub), `remote` (an arbitrary deployed site " +
+ "origin URL), `local` (a composed public dir on disk), or `hub` (an " +
+ "arbitrary hub URL to federate). Unknown site tokens are reported, not " +
+ "silently dropped. Still strictly read-only — this only changes which " +
+ "already-published static shards are read; nothing is written to any corpus.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ site: {
+ type: "string",
+ description:
+ "A hub member site to scope to, by siteId or title " +
+ "(case-insensitive). Switches to a single-site remote source.",
+ },
+ sites: {
+ type: "array",
+ items: { type: "string" },
+ description:
+ "A list of hub member sites (siteId or title) to federate as a " +
+ "subset of the hub.",
+ },
+ remote: {
+ type: "string",
+ description: "An arbitrary deployed site origin URL to read.",
+ },
+ local: {
+ type: "string",
+ description: "A composed public dir on disk to read.",
+ },
+ hub: {
+ type: "string",
+ description: "An arbitrary hub URL to federate over every member.",
+ },
+ },
+ additionalProperties: false,
+ },
+ },
+ {
+ name: "reset_source",
+ description:
+ "Return to the source this server was started with (its --hub/--remote/" +
+ "--local flag or env) and clear the persisted selection.",
+ inputSchema: { type: "object", properties: {}, additionalProperties: false },
+ },
];
-// Build a configured MCP server over a data source. The same four tools work
-// for local / remote / hub sources — only the ShardSource differs.
-export function createServer(source: ShardSource): Server {
+// The slice of SourceController the server needs. A bare ShardSource is wrapped
+// in a fixed, non-persisting implementation so the existing tools (and tests
+// that pass a plain source) keep working while switching is simply disabled.
+export interface SourceControllerLike {
+ current: ShardSource;
+ activeSpec?: SourceSpec;
+ switchTo(spec: SourceSpec): Promise<void>;
+ reset(): Promise<void>;
+ listHubSites(): Promise<HubSite[]>;
+ hubUrl(): string | undefined;
+}
+
+// Wrap a bare ShardSource so the server always talks to a controller. Switching
+// is disabled (the server was pointed at one fixed source), but list_sources
+// still reports it and a hub source can still enumerate its members.
+function fixedController(source: ShardSource): SourceControllerLike {
+ return {
+ current: source,
+ activeSpec: undefined,
+ async switchTo(): Promise<void> {
+ throw new Error(
+ "this server is pinned to a single source; source switching is disabled",
+ );
+ },
+ async reset(): Promise<void> {
+ // nothing persisted, nothing to return to
+ },
+ async listHubSites(): Promise<HubSite[]> {
+ return source instanceof HubSource ? source.listSites() : [];
+ },
+ hubUrl(): string | undefined {
+ return source instanceof HubSource ? source.hubBase : undefined;
+ },
+ };
+}
+
+// Build a configured MCP server over a data source or a SourceController. The
+// core read tools work for local / remote / hub sources — only the ShardSource
+// differs — and the source can be switched at runtime via use_source when a
+// real controller is supplied.
+export function createServer(sourceOrController: ShardSource | SourceController): Server {
+ const controller: SourceControllerLike =
+ "current" in sourceOrController
+ ? sourceOrController
+ : fixedController(sourceOrController);
const server = new Server(
{ name: "yt-dlp-transcript-mcp", version: "0.1.0" },
{ capabilities: { tools: {}, prompts: {} } },
@@ -212,6 +359,7 @@ export function createServer(source: ShardSource): Server {
server.setRequestHandler(CallToolRequestSchema, async (req) => {
const name = req.params.name;
const args = (req.params.arguments ?? {}) as Record<string, unknown>;
+ const source = controller.current;
try {
switch (name) {
case "list_channels":
@@ -224,6 +372,12 @@ export function createServer(source: ShardSource): Server {
return await handleGetTranscripts(source, args);
case "get_video_metadata":
return await handleGetMetadata(source, args);
+ case "list_sources":
+ return await handleListSources(controller);
+ case "use_source":
+ return await handleUseSource(controller, args);
+ case "reset_source":
+ return await handleResetSource(controller);
default:
return errorText(`unknown tool: ${name}`);
}
@@ -235,16 +389,74 @@ export function createServer(source: ShardSource): Server {
return server;
}
+function channelLine(c: ChannelRef): string {
+ const parts = [`slug: ${c.slug}`];
+ if (c.videoCount != null) parts.push(`${c.videoCount} videos`);
+ if (c.siteTitle) parts.push(`site: ${c.siteTitle}`);
+ return ` - ${c.name} (${parts.join(", ")})`;
+}
+
+// List channels organized under their resolved group, pick-friendly: a header
+// per group (name || id, member count, selectedByDefault), the channels beneath
+// it, then a compact "Groups" line (id · name · N channels) so a human can name
+// a group selector for a scoped search/sweep. When the source has no groups
+// (absent manifest, or hub mode) every channel folds under the fallback group
+// and the Groups line is omitted.
async function handleListChannels(source: ShardSource): Promise<ToolResult> {
const channels = await source.listChannels();
if (channels.length === 0) return text(`No channels found in ${source.label}.`);
- const lines = channels.map((c) => {
- const parts = [`slug: ${c.slug}`];
- if (c.videoCount != null) parts.push(`${c.videoCount} videos`);
- if (c.siteTitle) parts.push(`site: ${c.siteTitle}`);
- return `- ${c.name} (${parts.join(", ")})`;
- });
- return text(`${channels.length} channel(s) in ${source.label}:\n${lines.join("\n")}`);
+ const { groups, defaultGroupId } = await source.loadGroups();
+
+ // Bucket channels by their resolved group id (browser semantics).
+ const byGroup = new Map<string, ChannelRef[]>();
+ for (const c of channels) {
+ const gid = groups.length > 0
+ ? resolveChannelGroupId(c.groupId, groups, defaultGroupId)
+ : FALLBACK_GROUP.id;
+ const bucket = byGroup.get(gid) ?? [];
+ bucket.push(c);
+ byGroup.set(gid, bucket);
+ }
+
+ // Render in group order, only including groups that actually have members.
+ const ordered = groups.length > 0 ? sortGroups(groups) : [FALLBACK_GROUP];
+ const knownIds = new Set(ordered.map((g) => g.id));
+ const rendered: ChannelGroup[] = ordered.filter((g) => (byGroup.get(g.id)?.length ?? 0) > 0);
+ // Any bucket whose id isn't a known group (shouldn't happen after resolve, but
+ // be defensive) is appended under a synthetic header so no channel is dropped.
+ const strayIds = [...byGroup.keys()].filter((id) => !knownIds.has(id));
+
+ const sections: string[] = [];
+ for (const g of rendered) {
+ const members = byGroup.get(g.id) ?? [];
+ const header =
+ `### ${g.name || g.id} (${members.length} channel(s)` +
+ (g.selectedByDefault ? "" : "; not selected by default") +
+ `)` +
+ (g.description ? ` — ${g.description}` : "");
+ sections.push(`${header}\n${members.map(channelLine).join("\n")}`);
+ }
+ for (const id of strayIds) {
+ const members = byGroup.get(id) ?? [];
+ sections.push(`### ${id} (${members.length} channel(s))\n${members.map(channelLine).join("\n")}`);
+ }
+
+ // Compact selector cheat-sheet — only meaningful when real groups exist.
+ const groupsLine =
+ groups.length > 0
+ ? "\n\nGroups (select by id or name):\n" +
+ sortGroups(groups)
+ .map(
+ (g) =>
+ `- ${g.id} · ${g.name || "(unnamed)"} · ${byGroup.get(g.id)?.length ?? 0} channel(s)` +
+ (g.selectedByDefault ? "" : " · not selected by default"),
+ )
+ .join("\n")
+ : "";
+
+ return text(
+ `${channels.length} channel(s) in ${source.label}:\n\n${sections.join("\n\n")}${groupsLine}`,
+ );
}
async function handleSearch(
@@ -256,6 +468,9 @@ async function handleSearch(
const result = await searchTranscripts(source, {
query,
channel: typeof args.channel === "string" ? args.channel : undefined,
+ channels: strArray(args.channels),
+ group: typeof args.group === "string" ? args.group : undefined,
+ groups: strArray(args.groups),
regex: args.regex === true,
limit: typeof args.limit === "number" ? args.limit : undefined,
offset: typeof args.offset === "number" ? args.offset : undefined,
@@ -265,6 +480,7 @@ async function handleSearch(
});
const aliasNote = describeFiredAliases(result.firedAliases);
+ const scopeNote = describeScope(result.selection);
const rangeStart = result.total === 0 ? 0 : result.offset + 1;
const rangeEnd = result.offset + result.hits.length;
const footer =
@@ -274,6 +490,7 @@ async function handleSearch(
(result.truncated
? "; coverage PARTIAL — scan hit the page/video cap"
: "") +
+ (scopeNote ? `; ${scopeNote}` : "") +
(aliasNote ? `; ${aliasNote}` : "") +
")";
@@ -307,6 +524,40 @@ function describeFiredAliases(fired: SearchAlias[]): string {
return `expanded via alias ${parts.join(", ")}`;
}
+// Coerce a tool arg to a trimmed non-empty string[] (drops non-strings/blanks).
+function strArray(v: unknown): string[] | undefined {
+ if (!Array.isArray(v)) return undefined;
+ const out = v
+ .filter((x): x is string => typeof x === "string")
+ .map((x) => x.trim())
+ .filter((x) => x !== "");
+ return out.length > 0 ? out : undefined;
+}
+
+// A human scope note for the search footer: names the resolved scope (whole
+// corpus, a named group + channel count, or an N-channel set) and warns about
+// any channel/group token that matched nothing — so a typo is surfaced, not
+// silently a full-corpus scan.
+function describeScope(sel: SearchResult["selection"]): string {
+ const parts: string[] = [];
+ if (sel.all) {
+ parts.push("scope: whole corpus");
+ } else if (sel.matchedGroups.length > 0) {
+ const names = sel.matchedGroups.map((g) => g.name || g.id).join(", ");
+ const label = sel.matchedGroups.length === 1 ? "group" : "groups";
+ parts.push(`scope: ${label} ${names} (${sel.channelCount} channel(s))`);
+ } else {
+ parts.push(`scope: ${sel.channelCount} channel(s)`);
+ }
+ if (sel.unknownChannels.length > 0) {
+ parts.push(`unknown channel(s): ${sel.unknownChannels.join(", ")}`);
+ }
+ if (sel.unknownGroups.length > 0) {
+ parts.push(`unknown group(s): ${sel.unknownGroups.join(", ")}`);
+ }
+ return parts.join("; ");
+}
+
async function handleGetTranscripts(
source: ShardSource,
args: Record<string, unknown>,
@@ -322,7 +573,10 @@ async function handleGetTranscripts(
const batch = ids.slice(0, CAP);
const query = typeof args.query === "string" ? args.query.trim() : "";
- const channelHint = typeof args.channel === "string" ? args.channel : undefined;
+ const channelHint = [
+ ...(typeof args.channel === "string" ? [args.channel] : []),
+ ...(strArray(args.channels) ?? []),
+ ];
const timestamps = args.timestamps !== false;
// Build the (alias-aware) matcher once for the whole batch when windowing.
@@ -433,6 +687,215 @@ async function handleGetMetadata(
);
}
+// ─── Source switching: list_sources / use_source / reset_source ───
+
+// A one-line human description of a source spec's target (kind + where it
+// points), for the list_sources / use_source reports.
+function describeSpec(spec: SourceSpec | undefined): string {
+ if (!spec) return "(pinned source)";
+ switch (spec.kind) {
+ case "hub":
+ return spec.sites && spec.sites.length > 0
+ ? `hub ${spec.url} (subset: ${spec.sites.join(", ")})`
+ : `hub ${spec.url} (all members)`;
+ case "remote":
+ return `remote ${spec.url}`;
+ case "local":
+ return `local ${spec.dir}`;
+ }
+}
+
+// A cheap post-switch summary: channel count, and group count when the new
+// source actually ships channel groups. Tolerant of a source that can't be
+// reached (reports the switch anyway).
+async function sourceSummary(source: ShardSource): Promise<string> {
+ try {
+ const channels = await source.listChannels();
+ const { groups } = await source.loadGroups();
+ const groupNote =
+ groups.length > 0 ? `, ${groups.length} group(s)` : "";
+ return `${channels.length} channel(s)${groupNote}`;
+ } catch (e) {
+ return `(could not read the new source: ${(e as Error).message})`;
+ }
+}
+
+// Report the active source and, when a hub is in play, its member sites so a
+// human can pick one (or a subset) to switch to.
+async function handleListSources(
+ controller: SourceControllerLike,
+): Promise<ToolResult> {
+ const lines: string[] = [
+ `Active source: ${controller.current.label}`,
+ ` target: ${describeSpec(controller.activeSpec)}`,
+ ];
+
+ let sites: HubSite[] = [];
+ try {
+ sites = await controller.listHubSites();
+ } catch (e) {
+ lines.push(`\n(could not list hub members: ${(e as Error).message})`);
+ }
+
+ if (sites.length > 0) {
+ // Which siteIds are in the current subset (only when the active source is a
+ // hub scoped to a subset); otherwise every member is "in scope".
+ const spec = controller.activeSpec;
+ const subset =
+ spec && spec.kind === "hub" && spec.sites && spec.sites.length > 0
+ ? new Set(spec.sites)
+ : undefined;
+ const rows = sites.map((s) => {
+ const inScope = !subset || subset.has(s.siteId);
+ const mark = subset ? (inScope ? "✓ " : " ") : "";
+ return ` - ${mark}${s.siteId} · ${s.title} · ${s.url}`;
+ });
+ lines.push(
+ `\nHub member sites (${sites.length})` +
+ (subset ? ` — ✓ = in the current subset` : "") +
+ `:\n${rows.join("\n")}`,
+ );
+ }
+
+ lines.push(
+ `\nSwitch with use_source: site:"<siteId|title>" (one member), ` +
+ `sites:["<a>","<b>"] (a subset), or an arbitrary remote:"<url>" / ` +
+ `local:"<dir>" / hub:"<url>". reset_source returns to the startup source. ` +
+ `Read-only throughout.`,
+ );
+
+ return text(lines.join("\n"));
+}
+
+// Switch the active source. Exactly one target family is accepted; a hub member
+// token (site/sites) is resolved against the hub's member list.
+async function handleUseSource(
+ controller: SourceControllerLike,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const site = typeof args.site === "string" ? args.site.trim() : "";
+ const sites = strArray(args.sites);
+ const remote = typeof args.remote === "string" ? args.remote.trim() : "";
+ const local = typeof args.local === "string" ? args.local.trim() : "";
+ const hub = typeof args.hub === "string" ? args.hub.trim() : "";
+
+ const families = [
+ site ? "site" : "",
+ sites ? "sites" : "",
+ remote ? "remote" : "",
+ local ? "local" : "",
+ hub ? "hub" : "",
+ ].filter((f) => f !== "");
+ if (families.length === 0) {
+ return errorText(
+ "use_source needs exactly one target: site, sites, remote, local, or hub.",
+ );
+ }
+ if (families.length > 1) {
+ return errorText(
+ `use_source takes exactly one target; got ${families.join(", ")}.`,
+ );
+ }
+
+ let spec: SourceSpec;
+ if (remote) {
+ spec = { kind: "remote", url: remote };
+ } else if (local) {
+ spec = { kind: "local", dir: local };
+ } else if (hub) {
+ spec = { kind: "hub", url: hub };
+ } else {
+ // site / sites — resolve against the hub member list.
+ const hubUrl = controller.hubUrl();
+ if (!hubUrl) {
+ return errorText(
+ "no hub context to resolve a site token against — start the server " +
+ "against a hub, or first switch with hub:\"<url>\" (or use remote/local).",
+ );
+ }
+ let members: HubSite[];
+ try {
+ members = await controller.listHubSites();
+ } catch (e) {
+ return errorText(`could not read hub members: ${(e as Error).message}`);
+ }
+ const resolve = (token: string): HubSite | undefined => {
+ const t = token.toLowerCase();
+ return members.find(
+ (m) => m.siteId.toLowerCase() === t || m.title.toLowerCase() === t,
+ );
+ };
+
+ if (site) {
+ const member = resolve(site);
+ if (!member) {
+ return errorText(
+ `unknown hub site "${site}". Known: ` +
+ members.map((m) => m.siteId).join(", "),
+ );
+ }
+ // A single member becomes a plain remote source → full group/alias support.
+ spec = { kind: "remote", url: member.url };
+ } else {
+ const resolved: string[] = [];
+ const unknown: string[] = [];
+ for (const token of sites!) {
+ const member = resolve(token);
+ if (member) resolved.push(member.siteId);
+ else unknown.push(token);
+ }
+ if (resolved.length === 0) {
+ return errorText(
+ `no known hub sites in [${sites!.join(", ")}]. Known: ` +
+ members.map((m) => m.siteId).join(", "),
+ );
+ }
+ spec = { kind: "hub", url: hubUrl, sites: resolved };
+ if (unknown.length > 0) {
+ // Switch to the resolvable subset but surface the typos.
+ try {
+ await controller.switchTo(spec);
+ } catch (e) {
+ return errorText(`use_source failed: ${(e as Error).message}`);
+ }
+ const summary = await sourceSummary(controller.current);
+ return text(
+ `Switched to ${controller.current.label} — ${summary}.\n` +
+ ` target: ${describeSpec(spec)}\n` +
+ `Unknown site token(s) skipped: ${unknown.join(", ")}.`,
+ );
+ }
+ }
+ }
+
+ try {
+ await controller.switchTo(spec);
+ } catch (e) {
+ return errorText(`use_source failed: ${(e as Error).message}`);
+ }
+ const summary = await sourceSummary(controller.current);
+ return text(
+ `Switched to ${controller.current.label} — ${summary}.\n` +
+ ` target: ${describeSpec(spec)}`,
+ );
+}
+
+// Return to the startup source and clear the persisted selection.
+async function handleResetSource(
+ controller: SourceControllerLike,
+): Promise<ToolResult> {
+ try {
+ await controller.reset();
+ } catch (e) {
+ return errorText(`reset_source failed: ${(e as Error).message}`);
+ }
+ const summary = await sourceSummary(controller.current);
+ return text(
+ `Reset to the startup source ${controller.current.label} — ${summary}.\n` +
+ ` target: ${describeSpec(controller.activeSpec)}`,
+ );
+}
+
// ─── Prompts: the first-class `sweep` entry point ───
// A single slash command that drives Claude Code to run the browser's
// "corpus sweep" on plan usage: enumerate a query's full match set, batch the
@@ -443,9 +906,11 @@ const PROMPTS = [
{
name: "sweep",
description:
- "Run a whole-corpus sweep for a query: enumerate every matching video, " +
- "batch-read the transcripts, and fold cited, cross-referenced findings " +
- "into a markdown report — driven by Claude Code on plan usage, no API key.",
+ "Sweep a query across the corpus (or a chosen group/channels): enumerate " +
+ "every matching video, batch-read the transcripts, and fold cited, " +
+ "cross-referenced findings into a markdown report — driven by Claude Code " +
+ "on plan usage, no API key. With no scope arg it lists the channel groups " +
+ "and asks you to pick a group/channels (or confirm 'all') before sweeping.",
arguments: [
{ name: "query", description: "Term or phrase to sweep for.", required: true },
{
@@ -454,6 +919,19 @@ const PROMPTS = [
required: false,
},
{
+ name: "channels",
+ description:
+ "Optional comma-separated channel slugs/names to restrict the sweep to.",
+ required: false,
+ },
+ {
+ name: "group",
+ description:
+ "Optional channel group (id or name, e.g. 'other' or 'Extended " +
+ "Universe') to restrict the sweep to its member channels.",
+ required: false,
+ },
+ {
name: "directive",
description:
"What to extract (default: 'key claims & contradictions').",
@@ -482,48 +960,104 @@ function buildSweepPrompt(args: Record<string, unknown>) {
const query = argStr(args, "query");
if (!query) throw new Error("sweep requires a query argument");
const channel = argStr(args, "channel");
+ const group = argStr(args, "group");
+ const channelsRaw = argStr(args, "channels");
+ const channels = channelsRaw
+ ? channelsRaw.split(",").map((s) => s.trim()).filter((s) => s !== "")
+ : [];
const directive = argStr(args, "directive") ?? "key claims & contradictions";
const batchSize = argStr(args, "batch_size") ?? "8";
const reportPath = argStr(args, "report_path") ?? "./sweep-report.md";
- const channelClause = channel
- ? ` restricted to channel "${channel}"`
- : " across the whole corpus";
+ // Human-readable scope clauses + the literal search_transcripts scope args to
+ // pass. When none is given, the sweep must pick-first (list + ask) rather than
+ // silently scanning the whole corpus.
+ const scopeClauses: string[] = [];
+ if (channel) scopeClauses.push(`channel "${channel}"`);
+ if (channels.length > 0)
+ scopeClauses.push(`channels [${channels.map((c) => `"${c}"`).join(", ")}]`);
+ if (group) scopeClauses.push(`group "${group}"`);
+ const hasScope = scopeClauses.length > 0;
+ const scopeArgsText = scopeClauses.join(" and ");
+
+ const introScope = hasScope
+ ? ` scoped to ${scopeArgsText}`
+ : " over a scope you will confirm with me first (see step 1)";
+
+ // Build the numbered steps. With an explicit scope we search directly; with
+ // none we insert a pick-first step and the Search step uses the chosen scope.
+ const steps: string[] = [];
+
+ if (!hasScope) {
+ steps.push(
+ `**Choose the scope first — do NOT default to the whole corpus.** No ` +
+ `channel/channels/group was supplied. Call \`list_channels\`, present ` +
+ `the channel groups and their channels to me, and ask which group(s) or ` +
+ `channel(s) to sweep — or to confirm **all** for the whole corpus. Wait ` +
+ `for my choice before enumerating anything. Only sweep everything if I ` +
+ `explicitly choose "all". Use my choice as the \`channel\`/\`channels\`/` +
+ `\`group\` scope in every \`search_transcripts\` call below.`,
+ );
+ }
+
+ steps.push(
+ `**Search.** Call \`search_transcripts\` with query "${query}"` +
+ (hasScope
+ ? ` and ${scopeArgsText}`
+ : ` and the scope I chose in step 1`) +
+ `. Curated aliases auto-expand the query — the footer reports which fired ` +
+ `(e.g. mis-transcribed spellings) and names the resolved scope (and warns ` +
+ `about any channel/group token that matched nothing — fix a typo before ` +
+ `continuing). Treat the *expanded* match set as your target and mention ` +
+ `the expansion and the scope in the report.`,
+ );
+
+ steps.push(
+ `**Enumerate the full worklist.** Page the complete set with ` +
+ `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` +
+ `until \`has_more\` is false — this gives you every id/title/channel/date ` +
+ `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` +
+ `(page/video cap), say so in the report — the sweep is then a sample, not ` +
+ `exhaustive.`,
+ );
+
+ steps.push(
+ `**Plan.** With N total matches and a batch size of ${batchSize}, that is ` +
+ `\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch ` +
+ `count) before you start.`,
+ );
+
+ steps.push(
+ `**Per batch**, for each group of up to ${batchSize} video ids:\n` +
+ ` - Call \`get_transcripts\` with those ids **and the query** so each ` +
+ `transcript comes back as bounded, timestamped excerpt windows around the ` +
+ `matches (alias-correct, high-signal).\n` +
+ ` - **Cross-reference** the batch against the report so far. Upsert ` +
+ `findings — claims, and contradictions with earlier claims — into ` +
+ `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` +
+ ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it ` +
+ `each batch), then **drop the raw transcript text** once folded — don't ` +
+ `carry it forward.`,
+ );
+
+ steps.push(
+ `**Finish.** Repeat to the end of the worklist, then write a short summary ` +
+ `section (the scope swept, how many videos covered, headline findings, any ` +
+ `partial-coverage caveat) and tell me the report path.`,
+ );
+
+ const numbered = steps
+ .map((s, i) => `${i + 1}. ${s}`)
+ .join("\n\n");
const text =
- `Run a **corpus sweep** for the query **"${query}"**${channelClause}, ` +
+ `Run a **corpus sweep** for the query **"${query}"**${introScope}, ` +
`extracting **${directive}**, and maintain a running markdown report at ` +
`\`${reportPath}\`. You are the sweep engine — work through the whole match ` +
`set methodically, using the transcript MCP tools for evidence and your own ` +
`Write/Edit tools for the report. The MCP is read-only; never try to change ` +
`the archive.\n\n` +
- `Follow these steps:\n\n` +
- `1. **Search.** Call \`search_transcripts\` with query "${query}"` +
- (channel ? ` and channel "${channel}"` : "") +
- `. Curated aliases auto-expand the query — the footer reports which fired ` +
- `(e.g. mis-transcribed spellings). Treat the *expanded* match set as your ` +
- `target and mention the expansion in the report if one fired.\n\n` +
- `2. **Enumerate the full worklist.** Page the complete set with ` +
- `\`include_snippets: false\` and a rising \`offset\` (offset += limit) until ` +
- `\`has_more\` is false — this gives you every id/title/channel/date cheaply. ` +
- `Note the \`total\`. If the footer says coverage is PARTIAL (page/video cap), ` +
- `say so in the report — the sweep is then a sample, not exhaustive.\n\n` +
- `3. **Plan.** With N total matches and a batch size of ${batchSize}, that is ` +
- `\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch count) ` +
- `before you start.\n\n` +
- `4. **Per batch**, for each group of up to ${batchSize} video ids:\n` +
- ` - Call \`get_transcripts\` with those ids **and the query** so each ` +
- `transcript comes back as bounded, timestamped excerpt windows around the ` +
- `matches (alias-correct, high-signal).\n` +
- ` - **Cross-reference** the batch against the report so far. Upsert ` +
- `findings — claims, and contradictions with earlier claims — into ` +
- `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` +
- ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it each ` +
- `batch), then **drop the raw transcript text** once folded — don't carry it ` +
- `forward.\n\n` +
- `5. **Finish.** Repeat to the end of the worklist, then write a short summary ` +
- `section (how many videos covered, headline findings, any partial-coverage ` +
- `caveat) and tell me the report path.`;
+ `Follow these steps:\n\n${numbered}`;
return {
description: `Corpus sweep for "${query}" → ${reportPath}`,
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -9,20 +9,53 @@ import {
coerceAliasConfig,
type SearchAlias,
} from "yt-dlp-transcript-common/lib/searchAliases";
+import {
+ parseChannelGroups,
+ resolveDefaultGroupId,
+ DEFAULT_GROUP_FALLBACK_ID,
+ type ChannelGroup,
+} from "yt-dlp-transcript-common/lib/channelGroups";
// A channel the source can serve. `siteId`/`siteUrl` are only populated in hub
// mode (so results can be attributed to the owning member site); `key` is the
// stable, source-unique handle a tool passes back to fetch this channel's data.
+// `groupId` is the channel's raw group membership as shipped in corpus.json (it
+// may be unknown/absent — resolve it against loadGroups() with
+// resolveChannelGroupId before using it).
export type ChannelRef = {
key: string;
slug: string;
name: string;
videoCount?: number;
+ groupId?: string;
siteId?: string;
siteTitle?: string;
siteUrl?: string;
};
+// The site's channel-group definitions, as read from summaries/manifest.json —
+// the same groups the viewer's channel filter renders. `defaultGroupId` is the
+// bucket unknown/absent channel groupIds fold onto (resolveChannelGroupId).
+export type ChannelGroups = {
+ groups: ChannelGroup[];
+ defaultGroupId: string;
+};
+
+// What a source returns when it has no group definitions (absent/malformed
+// manifest, or hub mode where per-site groups are a different model).
+const EMPTY_GROUPS: ChannelGroups = {
+ groups: [],
+ defaultGroupId: DEFAULT_GROUP_FALLBACK_ID,
+};
+
+// Parse a summaries/manifest.json blob into channel-group defs, tolerating any
+// missing/malformed shape (→ empty fallback).
+function parseGroupsManifest(raw: unknown): ChannelGroups {
+ const m = (raw ?? {}) as { groups?: unknown; defaultGroupId?: unknown };
+ const groups = parseChannelGroups(m.groups);
+ return { groups, defaultGroupId: resolveDefaultGroupId(m.defaultGroupId, groups) };
+}
+
// A read-only view over a transcript corpus's paginated JSON shards. Three
// implementations (local dir / remote origin / federated hub) all speak the
// same three-call contract, which mirrors the documented shard scheme in
@@ -39,17 +72,30 @@ export interface ShardSource {
// (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed
// spellings. Result is cached per source.
loadAliases(): Promise<SearchAlias[]>;
+ // The site's channel-group definitions (summaries/manifest.json). Returns the
+ // empty fallback when absent/malformed, or in hub mode (federated per-site
+ // groups are a different model — deferred). Cached per source.
+ loadGroups(): Promise<ChannelGroups>;
}
// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose
-// — we only need slug/name/count.
-type CorpusJsonChannel = { slug: string; name?: string; videoCount?: number };
+// — we only need slug/name/count/group.
+type CorpusJsonChannel = {
+ slug: string;
+ name?: string;
+ videoCount?: number;
+ groupId?: string;
+};
type SiteCorpusJson = { channels?: CorpusJsonChannel[] };
type HubCorpusJson = {
kind?: string;
sites?: { siteId: string; title: string; url: string }[];
};
+// One member site of a hub, as listed in the hub's corpus.json. Exposed so the
+// source controller can resolve a `site`/`sites` token to a member origin.
+export type HubSite = { siteId: string; title: string; url: string };
+
// ─── Local: read composed shards from a directory on disk ───
// `dir` is a composed public dir (or any dir containing transcripts/<slug>/…).
// Prefers corpus.json for the channel list (names + counts); falls back to
@@ -57,6 +103,7 @@ type HubCorpusJson = {
export class LocalSource implements ShardSource {
readonly label: string;
private aliases?: SearchAlias[];
+ private groups?: ChannelGroups;
constructor(private dir: string) {
this.label = `local:${dir}`;
}
@@ -75,6 +122,20 @@ export class LocalSource implements ShardSource {
return this.aliases;
}
+ async loadGroups(): Promise<ChannelGroups> {
+ if (this.groups) return this.groups;
+ try {
+ const raw = await readFile(
+ path.join(this.dir, "summaries", "manifest.json"),
+ "utf8",
+ );
+ this.groups = parseGroupsManifest(JSON.parse(raw));
+ } catch {
+ this.groups = EMPTY_GROUPS; // no/invalid manifest — groups off
+ }
+ return this.groups;
+ }
+
async listChannels(): Promise<ChannelRef[]> {
try {
const raw = await readFile(path.join(this.dir, "corpus.json"), "utf8");
@@ -85,6 +146,7 @@ export class LocalSource implements ShardSource {
slug: c.slug,
name: c.name ?? c.slug,
videoCount: c.videoCount,
+ groupId: c.groupId,
}));
}
} catch {
@@ -132,6 +194,7 @@ export class RemoteSource implements ShardSource {
readonly label: string;
private base: string;
private aliases?: SearchAlias[];
+ private groups?: ChannelGroups;
constructor(baseUrl: string) {
this.base = baseUrl.replace(/\/+$/, "");
this.label = `remote:${this.base}`;
@@ -150,6 +213,19 @@ export class RemoteSource implements ShardSource {
return this.aliases;
}
+ async loadGroups(): Promise<ChannelGroups> {
+ if (this.groups) return this.groups;
+ try {
+ const res = await fetch(`${this.base}/summaries/manifest.json`);
+ this.groups = res.ok
+ ? parseGroupsManifest(await res.json())
+ : EMPTY_GROUPS;
+ } catch {
+ this.groups = EMPTY_GROUPS;
+ }
+ return this.groups;
+ }
+
private async getJson<T>(p: string): Promise<T> {
const res = await fetch(`${this.base}${p}`);
if (!res.ok) {
@@ -165,6 +241,7 @@ export class RemoteSource implements ShardSource {
slug: c.slug,
name: c.name ?? c.slug,
videoCount: c.videoCount,
+ groupId: c.groupId,
siteUrl: this.base,
}));
}
@@ -183,13 +260,37 @@ export class RemoteSource implements ShardSource {
// stay unique, and manifest/page calls dispatch to the owning member.
export class HubSource implements ShardSource {
readonly label: string;
- private hubBase: string;
+ readonly hubBase: string;
private members = new Map<string, RemoteSource>(); // siteId -> source
private aliases?: SearchAlias[];
+ // Optional subset allowlist of member siteIds. Undefined = federate every
+ // member; a set restricts listChannels() to those members (site discovery via
+ // listSites() stays unfiltered so a picker can still see all members).
+ private allowSiteIds?: Set<string>;
- constructor(hubUrl: string) {
+ constructor(hubUrl: string, allowSiteIds?: string[]) {
this.hubBase = hubUrl.replace(/\/+$/, "");
- this.label = `hub:${this.hubBase}`;
+ this.allowSiteIds =
+ allowSiteIds && allowSiteIds.length > 0
+ ? new Set(allowSiteIds)
+ : undefined;
+ this.label = this.allowSiteIds
+ ? `hub:${this.hubBase} (${this.allowSiteIds.size} site(s))`
+ : `hub:${this.hubBase}`;
+ }
+
+ // Fetch the hub's corpus.json and return its member sites — UNFILTERED (the
+ // full membership), even when this source is scoped to a subset, so a picker
+ // (list_sources / use_source) can show every member.
+ async listSites(): Promise<HubSite[]> {
+ const res = await fetch(`${this.hubBase}/corpus.json`);
+ if (!res.ok) {
+ throw new Error(
+ `GET ${this.hubBase}/corpus.json -> ${res.status} ${res.statusText}`,
+ );
+ }
+ const hub = (await res.json()) as HubCorpusJson;
+ return hub.sites ?? [];
}
// A hub can ship its own /search-aliases.json (the merged federation-wide
@@ -207,6 +308,14 @@ export class HubSource implements ShardSource {
return this.aliases;
}
+ // In hub mode each group is itself a federated member site (a different model
+ // — accent-per-origin), so hub-wide channel-group tokens are deferred: return
+ // the empty fallback. Multi-channel scoping still works (member groupIds carry
+ // through listChannels, they just don't resolve against hub-level groups).
+ async loadGroups(): Promise<ChannelGroups> {
+ return EMPTY_GROUPS;
+ }
+
private memberFor(siteId: string): RemoteSource {
const m = this.members.get(siteId);
if (!m) throw new Error(`unknown hub member site: ${siteId}`);
@@ -214,14 +323,9 @@ export class HubSource implements ShardSource {
}
async listChannels(): Promise<ChannelRef[]> {
- const res = await fetch(`${this.hubBase}/corpus.json`);
- if (!res.ok) {
- throw new Error(
- `GET ${this.hubBase}/corpus.json -> ${res.status} ${res.statusText}`,
- );
- }
- const hub = (await res.json()) as HubCorpusJson;
- const sites = hub.sites ?? [];
+ const sites = (await this.listSites()).filter(
+ (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId),
+ );
const all: ChannelRef[] = [];
// Sequential member fetches keep it simple and polite; the channel count is
// small. A failing member is skipped rather than failing the whole list.
diff --git a/mcp/src/sourceController.test.ts b/mcp/src/sourceController.test.ts
@@ -0,0 +1,257 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { Client } from "@modelcontextprotocol/sdk/client/index.js";
+import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
+import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import type { ChannelGroups, ChannelRef, HubSite, ShardSource } from "./source";
+import type { SourceSpec } from "./sources";
+import { SourceController } from "./sourceController";
+import { createServer } from "./server";
+
+// ─── A build factory over in-memory stub sources (no HTTP) ───
+//
+// Each spec maps to a labelled stub. Hub specs additionally expose listSites()
+// returning a fixed two-member roster, so `site`/`sites` resolution and the
+// list_sources member listing can be exercised with no network.
+
+const HUB_SITES: HubSite[] = [
+ { siteId: "alpha", title: "Alpha Site", url: "https://alpha.example" },
+ { siteId: "beta", title: "Beta Site", url: "https://beta.example" },
+];
+
+function chanRef(key: string, name: string): ChannelRef {
+ return { key, slug: key, name };
+}
+
+// A minimal ShardSource whose only interesting behaviour is a distinct label
+// and a channel count, plus (for hub specs) listSites().
+class FakeSource implements ShardSource {
+ readonly label: string;
+ readonly listSites?: () => Promise<HubSite[]>;
+ constructor(
+ label: string,
+ private channels: ChannelRef[],
+ isHub: boolean,
+ ) {
+ this.label = label;
+ if (isHub) this.listSites = async () => HUB_SITES;
+ }
+ async loadAliases(): Promise<SearchAlias[]> {
+ return [];
+ }
+ async loadGroups(): Promise<ChannelGroups> {
+ return { groups: [], defaultGroupId: "default" };
+ }
+ async listChannels(): Promise<ChannelRef[]> {
+ return this.channels;
+ }
+ async transcriptsManifest(): Promise<ChannelTranscriptsManifest> {
+ throw new Error("not used in these tests");
+ }
+ async transcriptPage(): Promise<TranscriptDetail[]> {
+ return [];
+ }
+}
+
+function labelFor(spec: SourceSpec): string {
+ switch (spec.kind) {
+ case "hub":
+ return spec.sites && spec.sites.length > 0
+ ? `hub:${spec.url} (${spec.sites.length} site(s))`
+ : `hub:${spec.url}`;
+ case "remote":
+ return `remote:${spec.url}`;
+ case "local":
+ return `local:${spec.dir}`;
+ }
+}
+
+// The injected factory: FakeSource per spec, hub specs get listSites().
+function fakeBuild(spec: SourceSpec): ShardSource {
+ const channels =
+ spec.kind === "hub"
+ ? [chanRef("h1", "Hub Chan 1"), chanRef("h2", "Hub Chan 2")]
+ : [chanRef("c1", "Chan 1")];
+ return new FakeSource(labelFor(spec), channels, spec.kind === "hub");
+}
+
+async function tempStateFile(): Promise<{ file: string; dir: string }> {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "mcp-src-state-"));
+ return { file: path.join(dir, "state.json"), dir };
+}
+
+function newController(startup: SourceSpec, stateFile: string): SourceController {
+ return new SourceController(startup, { build: fakeBuild, stateFile });
+}
+
+// ─── Client helpers ───
+
+async function connect(controller: SourceController): Promise<Client> {
+ const server = createServer(controller);
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+ return client;
+}
+
+function firstText(res: unknown): string {
+ const content = (res as { content: { type: string; text: string }[] }).content;
+ return content.map((c) => c.text).join("\n");
+}
+
+const HUB_SPEC: SourceSpec = { kind: "hub", url: "https://hub.example" };
+
+// ─── (a) list_sources shows the active source and hub members ───
+
+test("list_sources shows the active source and, with a hub, its members", async () => {
+ const { file, dir } = await tempStateFile();
+ const controller = newController(HUB_SPEC, file);
+ await controller.init();
+ const client = await connect(controller);
+
+ const out = firstText(await client.callTool({ name: "list_sources", arguments: {} }));
+ assert.match(out, /Active source: hub:https:\/\/hub\.example/);
+ assert.match(out, /alpha · Alpha Site · https:\/\/alpha\.example/);
+ assert.match(out, /beta · Beta Site · https:\/\/beta\.example/);
+
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+});
+
+// ─── (b) use_source remote swaps; reset returns to startup ───
+
+test("use_source remote swaps the active source; reset_source returns to startup", async () => {
+ const { file, dir } = await tempStateFile();
+ const controller = newController(HUB_SPEC, file);
+ await controller.init();
+ const client = await connect(controller);
+
+ const sw = firstText(
+ await client.callTool({
+ name: "use_source",
+ arguments: { remote: "https://solo.example" },
+ }),
+ );
+ assert.match(sw, /Switched to remote:https:\/\/solo\.example/);
+ // A following list_channels reflects the new (remote → single) source.
+ const chans = firstText(await client.callTool({ name: "list_channels", arguments: {} }));
+ assert.match(chans, /remote:https:\/\/solo\.example/);
+ assert.match(chans, /Chan 1/);
+ assert.ok(!chans.includes("Hub Chan"), "no longer the hub's channels");
+
+ const reset = firstText(await client.callTool({ name: "reset_source", arguments: {} }));
+ assert.match(reset, /Reset to the startup source hub:https:\/\/hub\.example/);
+ const back = firstText(await client.callTool({ name: "list_channels", arguments: {} }));
+ assert.match(back, /Hub Chan 1/);
+
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+});
+
+// ─── (c) use_source site resolves a hub member; unknown is reported ───
+
+test("use_source site resolves a hub member to a single remote source", async () => {
+ const { file, dir } = await tempStateFile();
+ const controller = newController(HUB_SPEC, file);
+ await controller.init();
+ const client = await connect(controller);
+
+ // By title (case-insensitive) → the member's remote origin.
+ const byTitle = firstText(
+ await client.callTool({ name: "use_source", arguments: { site: "alpha site" } }),
+ );
+ assert.match(byTitle, /Switched to remote:https:\/\/alpha\.example/);
+ assert.equal(controller.activeSpec.kind, "remote");
+
+ const unknown = firstText(
+ await client.callTool({ name: "use_source", arguments: { site: "gamma" } }),
+ );
+ assert.match(unknown, /unknown hub site "gamma"/);
+
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+});
+
+// ─── (d) use_source sites builds a subset federation ───
+
+test("use_source sites builds a hub subset federation and reports unknown tokens", async () => {
+ const { file, dir } = await tempStateFile();
+ const controller = newController(HUB_SPEC, file);
+ await controller.init();
+ const client = await connect(controller);
+
+ const out = firstText(
+ await client.callTool({
+ name: "use_source",
+ arguments: { sites: ["alpha", "beta", "ghost"] },
+ }),
+ );
+ assert.match(out, /Switched to hub:https:\/\/hub\.example \(2 site\(s\)\)/);
+ assert.match(out, /Unknown site token\(s\) skipped: ghost/);
+ assert.equal(controller.activeSpec.kind, "hub");
+ assert.deepEqual(
+ controller.activeSpec.kind === "hub" ? controller.activeSpec.sites : null,
+ ["alpha", "beta"],
+ );
+
+ await client.close();
+ await rm(dir, { recursive: true, force: true });
+});
+
+// ─── (e) persistence across a fresh controller; reset clears it ───
+
+test("a switched source persists to a fresh controller; reset clears the file", async () => {
+ const { file, dir } = await tempStateFile();
+ const first = newController(HUB_SPEC, file);
+ await first.init();
+ await first.switchTo({ kind: "remote", url: "https://persisted.example" });
+
+ // The state file exists and names the switched spec.
+ const raw = JSON.parse(await readFile(file, "utf8"));
+ assert.equal(raw.activeSpec.url, "https://persisted.example");
+ assert.equal(raw.hubRef, "https://hub.example", "hub context is remembered");
+
+ // A brand-new controller over the same startup + state file resumes it.
+ const second = newController(HUB_SPEC, file);
+ await second.init();
+ assert.equal(second.current.label, "remote:https://persisted.example");
+ assert.equal(second.activeSpec.kind, "remote");
+ // hubRef survives, so site/sites still resolve.
+ assert.equal(second.hubUrl(), "https://hub.example");
+
+ // reset clears persistence; a third controller starts from the startup spec.
+ await second.reset();
+ await assert.rejects(readFile(file, "utf8"), "state file deleted");
+ const third = newController(HUB_SPEC, file);
+ await third.init();
+ assert.equal(third.current.label, "hub:https://hub.example");
+
+ await rm(dir, { recursive: true, force: true });
+});
+
+// ─── (f) a bare ShardSource still drives createServer (switching disabled) ───
+
+test("createServer accepts a bare ShardSource; list_sources works, use_source is disabled", async () => {
+ const bare = new FakeSource("local:/tmp/x", [chanRef("c1", "Chan 1")], false);
+ const server = createServer(bare);
+ const [ct, st] = InMemoryTransport.createLinkedPair();
+ const client = new Client({ name: "test", version: "0" }, { capabilities: {} });
+ await Promise.all([server.connect(st), client.connect(ct)]);
+
+ const listed = firstText(await client.callTool({ name: "list_sources", arguments: {} }));
+ assert.match(listed, /Active source: local:\/tmp\/x/);
+
+ const res = await client.callTool({
+ name: "use_source",
+ arguments: { remote: "https://nope.example" },
+ });
+ assert.equal((res as { isError?: boolean }).isError, true);
+ assert.match(firstText(res), /pinned to a single source/);
+
+ await client.close();
+});
diff --git a/mcp/src/sourceController.ts b/mcp/src/sourceController.ts
@@ -0,0 +1,140 @@
+import { mkdir, readFile, writeFile, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { HubSource, type HubSite, type ShardSource } from "./source";
+import { buildSource, type SourceSpec } from "./sources";
+
+// What we persist to the state file: the active source spec plus a remembered
+// hub URL (so `site`/`sites` tokens still resolve after switching to a single
+// member or an arbitrary target). Kept minimal and forward-tolerant.
+type PersistedState = {
+ activeSpec: SourceSpec;
+ hubRef?: string;
+};
+
+// The directory selections are persisted under. TRANSCRIPT_MCP_STATE_DIR wins;
+// else $XDG_STATE_HOME/yt-dlp-transcript-mcp (fallback ~/.local/state/…).
+function stateDir(): string {
+ const override = process.env.TRANSCRIPT_MCP_STATE_DIR;
+ if (override && override.trim() !== "") return override;
+ const xdg = process.env.XDG_STATE_HOME;
+ const base =
+ xdg && xdg.trim() !== "" ? xdg : path.join(os.homedir(), ".local", "state");
+ return path.join(base, "yt-dlp-transcript-mcp");
+}
+
+// A filesystem-safe slug of a source label, used to key the state file so two
+// differently-configured servers (archilyzer / rekietalyzer / a local build)
+// keep independent selections and don't clobber each other.
+function slugify(label: string): string {
+ return (
+ label
+ .toLowerCase()
+ .replace(/[^a-z0-9]+/g, "-")
+ .replace(/^-+|-+$/g, "") || "default"
+ );
+}
+
+// A hub URL a `site`/`sites` token can resolve against: the remembered hubRef,
+// or the active spec's url when the active source is itself a hub.
+function hubUrlFrom(activeSpec: SourceSpec, hubRef?: string): string | undefined {
+ if (hubRef) return hubRef;
+ if (activeSpec.kind === "hub") return activeSpec.url;
+ return undefined;
+}
+
+// Holds the mutable "active source" for the server and persists the selection
+// so it survives reconnects. The `build` factory is injectable so tests can
+// swap in stub sources with no HTTP. Everything the server does per call reads
+// `controller.current`.
+export class SourceController {
+ current: ShardSource;
+ activeSpec: SourceSpec;
+ hubRef?: string;
+ readonly startupSpec: SourceSpec;
+ readonly stateFile: string;
+ private build: (spec: SourceSpec) => ShardSource;
+
+ constructor(
+ startupSpec: SourceSpec,
+ opts: {
+ build?: (spec: SourceSpec) => ShardSource;
+ stateFile?: string;
+ } = {},
+ ) {
+ this.startupSpec = startupSpec;
+ this.build = opts.build ?? buildSource;
+ this.activeSpec = startupSpec;
+ this.hubRef = startupSpec.kind === "hub" ? startupSpec.url : undefined;
+ this.current = this.build(startupSpec);
+ this.stateFile =
+ opts.stateFile ?? path.join(stateDir(), `${slugify(this.current.label)}.json`);
+ }
+
+ // Adopt a persisted selection if one exists and parses (persist across
+ // reconnects); otherwise stay on the startup source. Never throws — a
+ // missing/corrupt state file just means "start fresh".
+ async init(): Promise<void> {
+ try {
+ const raw = await readFile(this.stateFile, "utf8");
+ const state = JSON.parse(raw) as PersistedState;
+ if (state && state.activeSpec && typeof state.activeSpec.kind === "string") {
+ this.activeSpec = state.activeSpec;
+ this.hubRef = state.hubRef ?? this.hubRef;
+ this.current = this.build(state.activeSpec);
+ }
+ } catch {
+ // no/invalid state file — keep the startup source
+ }
+ }
+
+ // Switch the active source, remembering a hub URL when the target is a hub,
+ // and persist the new selection.
+ async switchTo(spec: SourceSpec): Promise<void> {
+ this.current = this.build(spec);
+ this.activeSpec = spec;
+ if (spec.kind === "hub") this.hubRef = spec.url;
+ await this.persist();
+ }
+
+ // Return to the startup source and clear the persisted selection.
+ async reset(): Promise<void> {
+ this.activeSpec = this.startupSpec;
+ this.hubRef =
+ this.startupSpec.kind === "hub" ? this.startupSpec.url : undefined;
+ this.current = this.build(this.startupSpec);
+ try {
+ await rm(this.stateFile, { force: true });
+ } catch {
+ // best-effort — nothing to clean up
+ }
+ }
+
+ private async persist(): Promise<void> {
+ const state: PersistedState = {
+ activeSpec: this.activeSpec,
+ hubRef: this.hubRef,
+ };
+ await mkdir(path.dirname(this.stateFile), { recursive: true });
+ await writeFile(this.stateFile, JSON.stringify(state, null, 2), "utf8");
+ }
+
+ // The hub URL a `site`/`sites` token resolves against (remembered hubRef, or
+ // the active spec's url when it's a hub), or undefined when there's no hub
+ // context at all.
+ hubUrl(): string | undefined {
+ return hubUrlFrom(this.activeSpec, this.hubRef);
+ }
+
+ // List the member sites of the hub context (unfiltered), or [] when there is
+ // no hub in play. Built through the same factory so tests can stub it.
+ async listHubSites(): Promise<HubSite[]> {
+ const url = this.hubUrl();
+ if (!url) return [];
+ const hub = this.build({ kind: "hub", url });
+ if (hub instanceof HubSource) return hub.listSites();
+ // A stubbed factory may return a non-HubSource that still exposes listSites.
+ const maybe = hub as unknown as { listSites?: () => Promise<HubSite[]> };
+ return typeof maybe.listSites === "function" ? maybe.listSites() : [];
+ }
+}
diff --git a/mcp/src/sources.ts b/mcp/src/sources.ts
@@ -0,0 +1,66 @@
+import {
+ LocalSource,
+ RemoteSource,
+ HubSource,
+ type ShardSource,
+} from "./source";
+
+// A serializable description of a data source — the shape we persist, pass to
+// use_source, and rebuild a live ShardSource from. `sites` on a hub spec is a
+// subset allowlist of member siteIds (undefined = federate every member).
+export type SourceSpec =
+ | { kind: "hub"; url: string; sites?: string[] }
+ | { kind: "remote"; url: string }
+ | { kind: "local"; dir: string };
+
+// Resolve the data-source spec from flags or env. Precedence: hub > remote >
+// local > positional > default. All logging goes to stderr — stdout is the MCP
+// JSON-RPC channel and must not be polluted.
+//
+// --hub <url> | TRANSCRIPT_HUB_URL federate over a hub's member sites
+// --remote <url> | TRANSCRIPT_SITE_URL one deployed site origin
+// --local <dir> | TRANSCRIPT_LOCAL_DIR a composed public dir on disk
+// <positional> an http(s) URL → remote, otherwise a local dir
+// (default) ./export/public
+export function resolveSourceSpec(argv: string[]): SourceSpec {
+ const flag = (name: string): string | undefined => {
+ const i = argv.indexOf(name);
+ return i >= 0 && i + 1 < argv.length ? argv[i + 1] : undefined;
+ };
+
+ const hub = flag("--hub") ?? process.env.TRANSCRIPT_HUB_URL;
+ if (hub) return { kind: "hub", url: hub };
+
+ const remote =
+ flag("--remote") ?? flag("--url") ?? process.env.TRANSCRIPT_SITE_URL;
+ if (remote) return { kind: "remote", url: remote };
+
+ const local = flag("--local") ?? process.env.TRANSCRIPT_LOCAL_DIR;
+ if (local) return { kind: "local", dir: local };
+
+ const positional = argv.find((a) => !a.startsWith("-"));
+ if (positional) {
+ return /^https?:\/\//i.test(positional)
+ ? { kind: "remote", url: positional }
+ : { kind: "local", dir: positional };
+ }
+
+ return { kind: "local", dir: "export/public" };
+}
+
+// Build a live ShardSource from a spec.
+export function buildSource(spec: SourceSpec): ShardSource {
+ switch (spec.kind) {
+ case "hub":
+ return new HubSource(spec.url, spec.sites);
+ case "remote":
+ return new RemoteSource(spec.url);
+ case "local":
+ return new LocalSource(spec.dir);
+ }
+}
+
+// Thin compatibility wrapper: resolve a spec and build it in one call.
+export function resolveSource(argv: string[]): ShardSource {
+ return buildSource(resolveSourceSpec(argv));
+}