commit 6bb828a78f356d9e52af8cc1bba8ea44cb3f9167
parent cb05a9059cc98e4a3f414ad350e0c5677954bba2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 21 Jul 2026 14:42:09 -0400
MCP: cite-and-link output, subagent-compacted sweeps, and open_link
Feature 1 — clickable moment links by default. New pure common/lib/momentUrl.ts
builds an archilyzer viewer deep link (?v=<slug>&t=<sec>) when the source has a
public origin, else a per-platform watch link (YouTube &t=<sec>s, Odysee/Twitch
equivalents, Rumble/Kick bare). ShardSource.publicOrigin() exposes the origin;
search_transcripts / get_transcripts / get_transcript now render every [mm:ss]
as a Markdown link (via transcriptToMarkdown.linkForCue + getWindowedTranscript
link builders).
Feature 2 — subagent-compacted sweeps (prompt-only). The sweep per-batch step is
a map-reduce: a Task subagent reads each batch and returns only a compact,
linked, cited fragment, so the heavy transcript text stays out of the
orchestrator's context; inline fallback when no subagent tool is available.
Citations are now clickable [title @ mm:ss](url).
Feature 3 — open_link + sweep link=. Paste an archilyzer viewer share link and
re-run that exact search at full fidelity, via a preview -> adjust -> apply flow.
- mcp/src/shareLink.ts decodes origin + qt= query tree (or legacy q/re) +
fc/ft/fa/fav/fdf/fdt filters (fk decoded but reported ignored), with structured
overrides and readable plan rendering.
- mcp/src/search.ts gains runSearchSpec: a per-record query-tree evaluator
(transcripts/metadata/description/tags/chat scopes, AND/OR/NOT) plus a filter
predicate mirroring the browser, keeping paging/total/truncation. The plain
searchTranscripts scanner is kept intact so existing callers are unaffected.
- mcp/src/source.ts adds lazy subs (chat scope) + availability (fav filter)
fetching from the shipped shards.
- open_link previews a plan (no source switch, no search), honors overrides to
adjust in natural language, and on apply switches source + returns linked
results.
Tests: momentUrl, shareLink decode/override, and runSearchSpec per-scope /
tree-algebra / filter / paging suites (74 mcp tests green); typecheck clean.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
11 files changed, 2339 insertions(+), 81 deletions(-)
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -0,0 +1,110 @@
+// Build a timestamped deep link for a cited moment in a video.
+//
+// Two shapes, in preference order:
+//
+// 1. **Archilyzer viewer link** — `${siteOrigin}/?v=<slug>&t=<seconds>`. This
+// is the exact scheme `common/components/urlState.ts` reads: `v` is the
+// video slug (`<channelSlug>/<id>`, the value the modal compares against
+// `detail.slug`) and `t` is whole seconds. Opening it lands in the archive's
+// transcript modal at the cited moment. Used whenever we know the public
+// origin of the viewer that owns the video (a deployed site, or a hub
+// member's url).
+//
+// 2. **Platform link** — the video's own `webpageUrl` plus a per-platform time
+// parameter, mirroring the seek patterns the in-app players use
+// (`common/components/{Odysee,Rumble,Twitch}Player.tsx`). Used as a fallback
+// when there is no viewer origin (e.g. an MCP pointed at a local build).
+//
+// Pure and dependency-light so it can be reused by the MCP server, the browser
+// report-citation UI, and build tools alike.
+
+import { detectPlatform, type Platform } from "./platform";
+
+export type MomentUrlInput = {
+ // Public origin of the archilyzer viewer that owns this video (a RemoteSource
+ // base, or a hub member's url). When present and non-empty, we build a viewer
+ // deep link. Null/undefined ⇒ fall back to a platform link.
+ siteOrigin?: string | null;
+ // The video's slug as the viewer's `?v=` param expects it (`channelSlug/id`).
+ slug?: string | null;
+ // The cited moment, in seconds.
+ seconds: number;
+ // Fallback building blocks when there is no viewer origin.
+ webpageUrl?: string | null;
+ platform?: Platform | null;
+};
+
+// Twitch watch/VOD URLs take an `XhYmZs` time token, not raw seconds.
+function twitchTime(totalSeconds: number): string {
+ const s = Math.max(0, Math.floor(totalSeconds));
+ const h = Math.floor(s / 3600);
+ const m = Math.floor((s % 3600) / 60);
+ return `${h}h${m}m${s % 60}s`;
+}
+
+// Best-effort deep link into a platform's own watch page at `seconds`. Mirrors
+// the per-platform time params the in-app players build. Platforms whose watch
+// page has no reliable start param (Rumble, Kick) get the bare `webpageUrl`.
+// Returns null only when there is no `webpageUrl` to work from.
+export function platformMomentUrl(
+ webpageUrl: string | null | undefined,
+ platform: Platform | null | undefined,
+ seconds: number,
+): string | null {
+ if (!webpageUrl) return null;
+ const secs = Math.max(0, Math.floor(seconds || 0));
+ if (secs <= 0) return webpageUrl;
+ const plat = platform ?? detectPlatform(webpageUrl);
+ let u: URL;
+ try {
+ u = new URL(webpageUrl);
+ } catch {
+ return webpageUrl;
+ }
+ switch (plat) {
+ case "youtube":
+ u.searchParams.set("t", `${secs}s`);
+ return u.toString();
+ case "odysee":
+ u.searchParams.set("t", String(secs));
+ return u.toString();
+ case "twitch":
+ u.searchParams.set("t", twitchTime(secs));
+ return u.toString();
+ // Rumble / Kick / unknown: the watch page has no dependable start param —
+ // return the plain webpage URL rather than an invalid seek.
+ default:
+ return webpageUrl;
+ }
+}
+
+// The archilyzer viewer deep link, or null when `siteOrigin`/`slug` are missing
+// or the origin can't be parsed.
+export function viewerMomentUrl(
+ siteOrigin: string | null | undefined,
+ slug: string | null | undefined,
+ seconds: number,
+): string | null {
+ if (!siteOrigin || !slug) return null;
+ const secs = Math.max(0, Math.floor(seconds || 0));
+ let u: URL;
+ try {
+ u = new URL(siteOrigin);
+ } catch {
+ return null;
+ }
+ u.pathname = "/";
+ u.search = "";
+ u.hash = "";
+ u.searchParams.set("v", slug);
+ if (secs > 0) u.searchParams.set("t", String(secs));
+ return u.toString();
+}
+
+// The preferred moment link: viewer deep link when we have an origin+slug,
+// else the platform fallback. Null when neither can be built.
+export function momentUrl(input: MomentUrlInput): string | null {
+ const viewer = viewerMomentUrl(input.siteOrigin, input.slug, input.seconds);
+ if (viewer) return viewer;
+ return platformMomentUrl(input.webpageUrl, input.platform, input.seconds);
+}
diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts
@@ -31,6 +31,11 @@ export type TranscriptMarkdownOptions = {
// Cap the number of cue lines emitted (for fitting a context window). When
// truncated, a marker line is appended. Default: no cap.
maxCues?: number;
+ // Optional builder turning a cue's start seconds into a deep link. When it
+ // returns a URL, the timestamp is rendered as a Markdown link
+ // (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when
+ // `timestamps` is false or when it returns null. See `momentUrl`.
+ linkForCue?: (seconds: number) => string | null;
};
// [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so
@@ -53,6 +58,7 @@ export function transcriptToMarkdown(
includeDescription = true,
includeTags = false,
maxCues,
+ linkForCue,
} = options;
const lines: string[] = [];
@@ -97,7 +103,13 @@ export function transcriptToMarkdown(
const cue = cues[i];
const text = cue.text.trim();
if (!text) continue;
- lines.push(timestamps ? `[${stamp(cue.start)}] ${text}` : text);
+ if (!timestamps) {
+ lines.push(text);
+ continue;
+ }
+ const url = linkForCue ? linkForCue(cue.start) : null;
+ const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start);
+ lines.push(`[${label}] ${text}`);
}
if (limit < cues.length) {
lines.push("");
diff --git a/mcp/README.md b/mcp/README.md
@@ -13,14 +13,22 @@ reads the site's already-published static JSON shards (`corpus.json` +
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
-| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
-| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
-| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). |
+| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment**. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
+| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`), or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
+| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). |
| `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. |
+| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity). **Previews** a plan by default; **applies** it (switch source + search, linked results) on `apply:true`. Adjust in natural language via `overrides`. |
| `list_sources` | Show the **active** corpus (label + kind + target) and, with a hub context, its member sites (`siteId · title · url`, marking the current subset) so you can pick one to switch to. |
| `use_source` | **Switch which corpus is read**, on the fly — a hub member (`site`), a hub subset (`sites`), or an arbitrary `remote` URL / `local` dir / `hub` URL. Persists across reconnects. |
| `reset_source` | Return to the source the server was started with and clear the persisted selection. |
+**Clickable moment links.** Every timestamp the read tools emit is a Markdown link
+to the exact second — an **archilyzer viewer** deep link (`…/?v=<slug>&t=<sec>`,
+opening the transcript modal at the moment) when the source has a public origin,
+otherwise the video's platform watch page with a per-platform time param
+(YouTube `&t=<sec>s`, Odysee/Twitch equivalents). A `--local` source with no
+origin falls back to platform links.
+
### `search_transcripts`
Beyond `query`, `regex`, and `limit`:
@@ -66,6 +74,40 @@ video's full transcript comes back as markdown. Missing ids are reported inline.
lookup when a batch spans several channels (the ids are already scoped by the
search that produced them).
+### `open_link` — paste a viewer share link, at full fidelity
+
+The archilyzer viewer's **Share** button produces a URL that encodes the whole
+search: the origin, a `qt=` composite **query tree** (any mix of scopes —
+transcripts, live-chat, title/channel, description, tags — combined with
+AND/OR/NOT), and the filter block (`fc` channels, `ft` type, `fa` audience,
+`fav` availability, `fdf`/`fdt` upload-date range). `open_link` re-runs that exact
+search here — no manual source-switching or query reconstruction, nothing lost.
+
+It is a **preview → adjust → apply** flow:
+
+- **Preview** (default, `apply:false`) — decode the link and return a *plan*: the
+ resolved source (origin, hub vs single-site, auto-probed from `corpus.json`),
+ the query tree rendered readably, every active filter, the channel scope
+ validated against the live corpus, and anything ignored or warned (the `fk`
+ subtitle-track token is vestigial in the composite share model — decoded and
+ reported as ignored). **No source switch, no search yet.**
+- **Adjust** — re-call with structured `overrides` to honor a natural-language
+ edit: `clear_availability` (drop the availability filter), `clear_type`,
+ `clear_age`, `clear_dates`, `clear_filters`, `clear_channels`,
+ `channels:[…]` (re-scope), `date_from`/`date_to`, or `query`/`regex`/
+ `query_scope` (replace the search). The plan updates in place.
+- **Apply** (`apply:true`) — switch the active source to the link's origin (a
+ single-site remote, or a hub when the origin federates), run the search at full
+ fidelity (the query tree + filters, honoring the chat and availability scopes),
+ and return the first page of **linked** results. The applied plan is echoed at
+ the top for transparency. Pageable via `limit`/`offset`. Still read-only.
+
+```
+open_link link="https://rekietalyzer.pages.dev/?qt=…&fv=1&fc=Rekieta%20Law&ft=v&fav=a"
+open_link link="…" overrides={ "clear_availability": true } # "drop the availability filter"
+open_link link="…" apply=true # switch source + search
+```
+
## The `sweep` prompt
A first-class slash command that turns **Claude Code itself** into the corpus
@@ -74,8 +116,9 @@ browser does with a BYO AI key, but driven by your Claude **plan usage** (no API
key) and with the report written to a file.
Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query`
-(required), `channel?`, `channels?` (comma-separated slugs/names), `group?` (a
-channel group id or name), `directive?` (default *"key claims &
+(required *unless* a `link` is given), `link?` (an archilyzer viewer share URL to
+seed the sweep from), `channel?`, `channels?` (comma-separated slugs/names),
+`group?` (a channel group id or name), `directive?` (default *"key claims &
contradictions"*), `batch_size?` (default 8), `report_path?` (default
`./sweep-report.md`).
@@ -84,8 +127,15 @@ contradictions"*), `batch_size?` (default 8), `report_path?` (default
/mcp__rekietalyzer__sweep query="k cups" group="other" # a whole group
/mcp__rekietalyzer__sweep query="k cups" group="Extended Universe" # …by name
/mcp__rekietalyzer__sweep query="k cups" # pick-first (see below)
+/mcp__rekietalyzer__sweep link="https://…/?qt=…&fv=1&fc=…" # seed from a share link
```
+**Seed from a share link (`link=`).** Pass a viewer share URL instead of a
+`query` and the sweep starts by driving `open_link`: it previews the decoded plan
+(source, query tree, filters, scope, ignored bits), lets you confirm or adjust in
+natural language, then applies it — switching source and enumerating the full
+tree+filter match set — before batching. Everything the link encodes is honored.
+
**Pick-first when no scope is given.** Because a whole-corpus sweep can be a lot
of work, invoking `sweep` **without** a `channel`/`channels`/`group` makes Claude
FIRST call `list_channels`, present the groups and their channels, and ask you
@@ -97,13 +147,24 @@ name (`group="other"` ≡ `group="Extended Universe"`).
Once scoped, the prompt instructs Claude Code to: **search** (aliases
auto-expand; the footer names the resolved scope) → **enumerate** the full
worklist by paging with `include_snippets:false` until `has_more` is false →
-**plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts` for
-windowed context, cross-reference against the report so far, and upsert findings
-(claims, contradictions, sources cited as *title + [mm:ss]*) into well-titled
-`##` sections of the report file with its own Write/Edit tools → finish with a
-short summary. The MCP stays read-only; only the report file is written, in
+**plan** `ceil(N / batch_size)` batches → **per batch, map-reduce** → finish with
+a short summary. The MCP stays read-only; only the report file is written, in
Claude's working directory.
+**Subagent-compacted batches.** Each batch is handed to a **subagent** (Claude
+Code's Task tool) that calls `get_transcripts`, extracts findings for the
+directive, and returns *only a compact fragment of cited, linked findings* — the
+heavy transcript text lives and dies inside the subagent, so the orchestrator's
+context keeps just the distilled report. The orchestrator merges each fragment
+into the report and discards it; batches are independent, so several can run in
+parallel. If no subagent tool is available, the batch is processed inline and the
+raw text dropped after folding (the original behaviour).
+
+**Linked citations.** Every source is cited as a clickable
+`[title @ mm:ss](<moment url>)` link — the moment URL comes straight from the
+`[mm:ss](url)` links in the `get_transcripts` output, so a click seeks to the
+exact second (see *Clickable moment links* above).
+
## Data source (pick one)
Resolved from flags or env — precedence hub > remote > local:
diff --git a/mcp/src/momentUrl.test.ts b/mcp/src/momentUrl.test.ts
@@ -0,0 +1,127 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ momentUrl,
+ viewerMomentUrl,
+ platformMomentUrl,
+} from "yt-dlp-transcript-common/lib/momentUrl";
+
+// ─── Archilyzer viewer link (preferred) ───
+
+test("viewer link: origin + slug + seconds → ?v=<slug>&t=<sec>", () => {
+ const url = viewerMomentUrl(
+ "https://rekietalyzer.pages.dev",
+ "rekietalaw/0eYLR6CbR78",
+ 754,
+ );
+ assert.ok(url);
+ const u = new URL(url!);
+ assert.equal(u.origin, "https://rekietalyzer.pages.dev");
+ assert.equal(u.pathname, "/");
+ // The slug round-trips through URL decoding to exactly the modal's ?v= value.
+ assert.equal(u.searchParams.get("v"), "rekietalaw/0eYLR6CbR78");
+ assert.equal(u.searchParams.get("t"), "754");
+});
+
+test("viewer link: an origin with its own path/query is normalized to /", () => {
+ const url = viewerMomentUrl("https://site.example/search?q=x", "ch/vid", 10);
+ const u = new URL(url!);
+ assert.equal(u.pathname, "/");
+ assert.equal(u.searchParams.get("q"), null);
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("viewer link: t is omitted at second 0", () => {
+ const url = viewerMomentUrl("https://site.example", "ch/vid", 0);
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), null);
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("viewer link: null without an origin or a slug", () => {
+ assert.equal(viewerMomentUrl(null, "ch/vid", 10), null);
+ assert.equal(viewerMomentUrl("https://site.example", null, 10), null);
+});
+
+// ─── Platform fallback ───
+
+test("platform link: YouTube gets &t=<sec>s appended", () => {
+ const url = platformMomentUrl(
+ "https://www.youtube.com/watch?v=abc123",
+ "youtube",
+ 90,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("v"), "abc123");
+ assert.equal(u.searchParams.get("t"), "90s");
+});
+
+test("platform link: platform is detected from the URL when not given", () => {
+ const url = platformMomentUrl("https://www.youtube.com/watch?v=abc123", null, 5);
+ assert.match(url!, /t=5s/);
+});
+
+test("platform link: Odysee gets ?t=<sec> (raw seconds)", () => {
+ const url = platformMomentUrl(
+ "https://odysee.com/@chan/video",
+ "odysee",
+ 42,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), "42");
+});
+
+test("platform link: Twitch gets ?t=<XhYmZs>", () => {
+ const url = platformMomentUrl(
+ "https://www.twitch.tv/videos/12345",
+ "twitch",
+ 3661,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), "1h1m1s");
+});
+
+test("platform link: Rumble has no reliable start param → bare webpage url", () => {
+ const url = platformMomentUrl("https://rumble.com/v123-title.html", "rumble", 60);
+ assert.equal(url, "https://rumble.com/v123-title.html");
+});
+
+test("platform link: second 0 → bare webpage url; no webpage url → null", () => {
+ assert.equal(
+ platformMomentUrl("https://www.youtube.com/watch?v=abc", "youtube", 0),
+ "https://www.youtube.com/watch?v=abc",
+ );
+ assert.equal(platformMomentUrl(null, "youtube", 30), null);
+});
+
+// ─── momentUrl: prefers the viewer link, falls back to the platform link ───
+
+test("momentUrl: viewer link wins when an origin is present", () => {
+ const url = momentUrl({
+ siteOrigin: "https://site.example",
+ slug: "ch/vid",
+ seconds: 12,
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ const u = new URL(url!);
+ assert.equal(u.origin, "https://site.example");
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("momentUrl: falls back to the platform link with no origin (local source)", () => {
+ const url = momentUrl({
+ siteOrigin: null,
+ slug: "ch/vid",
+ seconds: 12,
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.match(url!, /youtube\.com/);
+ assert.match(url!, /t=12s/);
+});
+
+test("momentUrl: null when neither a viewer nor a platform link can be built", () => {
+ const url = momentUrl({ siteOrigin: null, slug: "ch/vid", seconds: 12 });
+ assert.equal(url, null);
+});
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -2,29 +2,60 @@ import { test } from "node:test";
import assert from "node:assert/strict";
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
-import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type {
+ ChannelTranscriptsManifest,
+ ChannelSubsManifest,
+} from "yt-dlp-transcript-common/lib/manifest";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups";
-import type { ChannelGroups, ChannelRef, ShardSource } from "./source";
+import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery";
+import type {
+ ChannelGroups,
+ ChannelRef,
+ ShardSource,
+ VideoAvailability,
+} from "./source";
import {
searchTranscripts,
getWindowedTranscript,
buildMatcher,
resolveSelectedChannels,
+ runSearchSpec,
+ type SearchFilters,
} from "./search";
import { createServer } from "./server";
+// Every filter facet kept (the "no-op" filter) — override fields per test.
+const KEEP_ALL: SearchFilters = {
+ videos: true,
+ livestreams: true,
+ allAges: true,
+ restricted: true,
+ available: true,
+ unlisted: true,
+ deleted: true,
+};
+
// ─── A tiny in-memory ShardSource for the tests ───
function cues(...pairs: [number, string][]): Cue[] {
return pairs.map(([start, text]) => ({ start, end: start + 3, text }));
}
-function vid(id: string, title: string, channelSlug: string, cs: Cue[]): TranscriptDetail {
+function vid(
+ id: string,
+ title: string,
+ channelSlug: string,
+ cs: Cue[],
+ extra: Partial<TranscriptDetail> = {},
+): TranscriptDetail {
return {
- slug: id,
+ // Real records carry a channel-prefixed slug (`<channelSlug>/<id>`); mirror
+ // that so viewer-link + availability joins are exercised.
+ slug: `${channelSlug}/${id}`,
id,
channelSlug,
title,
@@ -38,6 +69,7 @@ function vid(id: string, title: string, channelSlug: string, cs: Cue[]): Transcr
platform: "youtube",
webpageUrl: `https://example.test/${id}`,
cues: cs,
+ ...extra,
};
}
@@ -51,16 +83,40 @@ const K_CUPS_ALIAS: SearchAlias = {
};
// Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup".
+// Extra metadata on a1/a3/a4/a5 exercises the description/tags scopes and the
+// type/age/date filters without changing what matches the "coffee" cue search.
const CHAN_A: TranscriptDetail[] = [
- vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"])),
+ vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), {
+ description: "a pour over brewing guide",
+ tags: ["espresso", "beans"],
+ }),
vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])),
- vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"])),
- vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"])),
- vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"])),
+ vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), {
+ isLivestream: true,
+ }),
+ vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), {
+ ageRestricted: true,
+ }),
+ vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"]), {
+ uploadDate: "20250601",
+ }),
vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])),
vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])),
];
+// Live-chat cues, keyed by video slug — served via the subs shard methods so a
+// `chat`-scope leaf has something to match. Only b1 has a chat track.
+const CHAT: Record<string, Cue[]> = {
+ "chan-b/b1": cues([12, "@fan: banana bread anyone"], [15, "@mod: stay on topic"]),
+};
+
+// Availability overrides, keyed by video slug (defaults: available). a1 deleted,
+// a2 unlisted — the rest available.
+const AVAILABILITY: Record<string, VideoAvailability> = {
+ "chan-a/a1": { isDeleted: true, isUnlisted: false },
+ "chan-a/a2": { isDeleted: false, isUnlisted: true },
+};
+
// Channel B: one long transcript for windowing (a single match at 100s).
const CHAN_B: TranscriptDetail[] = [
vid(
@@ -136,6 +192,52 @@ class StubSource implements ShardSource {
async transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> {
return this.pages(ch)[page] ?? [];
}
+
+ publicOrigin(): string | null {
+ return null; // a stub has no viewer origin
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ // Only chan-b ships a (single-page) subs shard, with b1 on page 0.
+ if (ch.slug !== "chan-b") return null;
+ return {
+ version: 1,
+ channelSlug: ch.slug,
+ pageCount: 1,
+ maxPageBytes: 0,
+ generatedAt: "",
+ tracks: ["live_chat"],
+ slugToPage: { b1: 0 },
+ };
+ }
+
+ async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ if (ch.slug !== "chan-b" || page !== 0) return [];
+ const slug = "chan-b/b1";
+ return [
+ {
+ slug,
+ id: "b1",
+ channelSlug: "chan-b",
+ title: "Long one",
+ uploadDate: "20240101",
+ date: "2024-01-01",
+ duration: "5:00",
+ channel: "Channel B",
+ isLivestream: false,
+ ageRestricted: false,
+ isDeleted: false,
+ isUnlisted: false,
+ platform: "youtube",
+ webpageUrl: "https://example.test/b1",
+ tracks: { live_chat: CHAT[slug] },
+ },
+ ];
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map(Object.entries(AVAILABILITY));
+ }
}
// ─── (a) Paging: offset / limit / total / hasMore ───
@@ -302,6 +404,173 @@ test("getWindowedTranscript: no matches yields no lines", async () => {
assert.equal(lines.length, 0);
});
+// ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ───
+
+async function allChannels(src: StubSource): Promise<ChannelRef[]> {
+ return (await resolveSelectedChannels(src, {})).channels;
+}
+
+function leafTree(
+ query: string,
+ scope: "transcripts" | "chat" | "metadata" | "description" | "tags",
+ extra: Record<string, unknown> = {},
+) {
+ return newGroup({ children: [newLeaf({ query, scope, ...extra })] });
+}
+
+async function specIds(
+ src: StubSource,
+ tree: ReturnType<typeof newGroup>,
+ spec: { filters?: SearchFilters | null; aliases?: SearchAlias[] } = {},
+): Promise<string[]> {
+ const res = await runSearchSpec(src, await allChannels(src), { tree, ...spec }, { limit: 50 });
+ return res.hits.map((h) => h.videoId).sort();
+}
+
+test("spec transcripts leaf: matches cue text (same set as the plain scanner)", async () => {
+ const src = new StubSource();
+ assert.deepEqual(await specIds(src, leafTree("coffee", "transcripts")), [
+ "a1", "a2", "a3", "a4", "a5", "b1",
+ ]);
+});
+
+test("spec metadata leaf: matches title/channel, not cues", async () => {
+ const src = new StubSource();
+ // "kcup" is in a6's title but no cue (cue says "k cups" with a space).
+ assert.deepEqual(await specIds(src, leafTree("kcup", "metadata")), ["a6"]);
+});
+
+test("spec description leaf: matches the description only", async () => {
+ const src = new StubSource();
+ // "brewing" is only in a1's description, in no cue.
+ assert.deepEqual(await specIds(src, leafTree("brewing", "description")), ["a1"]);
+});
+
+test("spec tags leaf: matches the joined tags", async () => {
+ const src = new StubSource();
+ assert.deepEqual(await specIds(src, leafTree("espresso", "tags")), ["a1"]);
+});
+
+test("spec chat leaf: matches live_chat cues fetched from subs shards", async () => {
+ const src = new StubSource();
+ // "banana" only appears in b1's live chat, in no transcript cue.
+ const res = await runSearchSpec(src, await allChannels(src), {
+ tree: leafTree("banana", "chat"),
+ }, { limit: 50 });
+ assert.deepEqual(res.hits.map((h) => h.videoId), ["b1"]);
+ assert.equal(res.hits[0].snippets[0].scope, "chat");
+ assert.equal(res.hits[0].snippets[0].track, "live_chat");
+});
+
+test("spec AND: transcripts AND tags narrows to the intersection", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags" }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a1"]);
+});
+
+test("spec OR: description OR chat unions the two", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "OR",
+ children: [
+ newLeaf({ query: "brewing", scope: "description" }),
+ newLeaf({ query: "banana", scope: "chat" }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a1", "b1"]);
+});
+
+test("spec negate: coffee AND NOT (espresso tag) drops a1", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags", negate: true }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a2", "a3", "a4", "a5", "b1"]);
+});
+
+test("spec filter ft: livestreams:false drops the livestream (a3)", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, livestreams: false },
+ });
+ assert.deepEqual(ids, ["a1", "a2", "a4", "a5", "b1"]);
+});
+
+test("spec filter fa: keep only age-restricted → a4", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, allAges: false },
+ });
+ assert.deepEqual(ids, ["a4"]);
+});
+
+test("spec filter fav: available-only drops deleted a1 + unlisted a2", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, unlisted: false, deleted: false },
+ });
+ assert.deepEqual(ids, ["a3", "a4", "a5", "b1"]);
+});
+
+test("spec filter fav: deleted+unlisted-only keeps a1 + a2", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, available: false },
+ });
+ assert.deepEqual(ids, ["a1", "a2"]);
+});
+
+test("spec filter dates: fdf bound keeps only the later upload (a5)", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, dateFrom: "20250101" },
+ });
+ assert.deepEqual(ids, ["a5"]);
+});
+
+test("spec paging/total: stable total with an offset/limit slice", async () => {
+ const src = new StubSource();
+ const chans = await allChannels(src);
+ const tree = leafTree("coffee", "transcripts");
+ const page = await runSearchSpec(src, chans, { tree }, { limit: 2, offset: 2 });
+ assert.equal(page.total, 6);
+ assert.deepEqual(page.hits.map((h) => h.videoId), ["a3", "a4"]);
+ assert.equal(page.hasMore, true);
+});
+
+test("spec aliases: a transcripts leaf expands via curated aliases", async () => {
+ const src = new StubSource();
+ const res = await runSearchSpec(
+ src,
+ await allChannels(src),
+ { tree: leafTree("k cups", "transcripts"), aliases: await src.loadAliases() },
+ { limit: 50 },
+ );
+ assert.deepEqual(res.hits.map((h) => h.videoId).sort(), ["a6", "a7"]);
+ assert.equal(res.firedAliases.length, 1);
+ assert.equal(res.firedAliases[0].id, "k-cups");
+});
+
+test("spec hits carry slug + platform for moment links", async () => {
+ const src = new StubSource();
+ const res = await runSearchSpec(src, await allChannels(src), {
+ tree: leafTree("coffee", "transcripts"),
+ }, { limit: 1 });
+ assert.equal(res.hits[0].slug, "chan-a/a1");
+ assert.equal(res.hits[0].platform, "youtube");
+ assert.ok(res.hits[0].snippets[0].seconds >= 0);
+});
+
// ─── End-to-end through the MCP server (tools + prompts + missing ids) ───
async function connectClient(source: ShardSource): Promise<Client> {
@@ -410,6 +679,7 @@ test("server: the sweep prompt lists with its arguments and renders the query",
"channels",
"directive",
"group",
+ "link",
"query",
"report_path",
]);
@@ -449,3 +719,35 @@ test("server: the sweep prompt with no scope instructs a group/channel pick firs
assert.match(text, /confirm \*\*all\*\*/);
await client.close();
});
+
+test("server: the sweep prompt mandates linked citations and subagent batches", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ // Feature 1: linked citation form (not the old "title + [mm:ss]").
+ assert.match(text, /\[title @ mm:ss\]\(<moment url>\)/);
+ assert.ok(!/title \+ \[mm:ss\]/.test(text), "old bare citation form is gone");
+ // Feature 2: subagent map-reduce + inline fallback.
+ assert.match(text, /Spawn a subagent/);
+ assert.match(text, /Task tool/);
+ assert.match(text, /Fallback/);
+ await client.close();
+});
+
+test("server: the sweep prompt accepts a link= seed and drives open_link", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { link: "https://site.example/?q=coffee" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /open_link/);
+ assert.match(text, /apply:true/);
+ assert.match(text, /https:\/\/site\.example\/\?q=coffee/);
+ // A link-only sweep needs no query argument.
+ assert.match(text, /share link/);
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -1,5 +1,16 @@
import { formatDuration } from "yt-dlp-transcript-common/lib/format";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import type { Platform } from "yt-dlp-transcript-common/lib/platform";
+import {
+ forEachLeaf,
+ isLeaf,
+ isLeafActive,
+ isNodeActive,
+ type GroupNode,
+ type LayerScope,
+ type QueryNode,
+} from "yt-dlp-transcript-common/lib/searchQuery";
import {
matchAliases,
type SearchAlias,
@@ -14,7 +25,7 @@ import {
resolveChannelGroupId,
type ChannelGroup,
} from "yt-dlp-transcript-common/lib/channelGroups";
-import type { ChannelRef, ShardSource } from "./source";
+import type { ChannelRef, ShardSource, VideoAvailability } from "./source";
export type Snippet = { clock: string; seconds: number; text: string };
@@ -126,9 +137,17 @@ export async function resolveSelectedChannels(
export type SearchHit = {
videoId: string;
+ // The video's slug (`<channelSlug>/<id>`) — the archilyzer viewer's `?v=`
+ // value, used to build a moment deep link (momentUrl).
+ slug: string;
channelSlug: string;
channelName: string;
siteTitle?: string;
+ // The public origin of the site that owns this video (a member url in hub
+ // mode, the site base for a single remote). Used with `slug` for a viewer
+ // link; absent for a local source (→ platform-link fallback).
+ siteUrl?: string;
+ platform?: Platform;
title: string;
uploadDate: string;
webpageUrl?: string;
@@ -325,9 +344,12 @@ export async function searchTranscripts(
if (matches === 0 && !titleHit) continue;
all.push({
videoId: rec.id,
+ slug: rec.slug,
channelSlug: ch.slug,
channelName: ch.name,
...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}),
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(rec.platform ? { platform: rec.platform } : {}),
title: rec.title,
uploadDate: rec.uploadDate,
webpageUrl: rec.webpageUrl,
@@ -376,6 +398,9 @@ export function getWindowedTranscript(
after?: number;
maxCues?: number;
timestamps?: boolean;
+ // Optional builder turning a line's start seconds into a moment deep link;
+ // when it returns a URL the timestamp is rendered as a Markdown link.
+ link?: (seconds: number) => string | null;
} = {},
): { lines: string[]; matchCount: number } {
const cues = record.cues ?? [];
@@ -394,9 +419,12 @@ export function getWindowedTranscript(
);
merged = mergeSnippets(merged, win, WINDOW_LINE_CAP);
}
- const lines = merged.map((s) =>
- timestamps ? `[${s.clock}] ${s.text}` : s.text,
- );
+ const lines = merged.map((s) => {
+ if (!timestamps) return s.text;
+ const url = opts.link ? opts.link(s.seconds) : null;
+ const stamp = url ? `[${s.clock}](${url})` : s.clock;
+ return `[${stamp}] ${s.text}`;
+ });
return { lines, matchCount };
}
@@ -445,3 +473,419 @@ export async function findVideo(
}
return null;
}
+
+// ─── Full-fidelity spec engine: query tree + filter predicate ───
+//
+// `runSearchSpec` is the per-record evaluator that backs the `open_link` /
+// `sweep link=` flows: it honors everything an archilyzer share link can carry
+// (a composite `qt=` query tree over any mix of scopes, plus the `fc/ft/fa/fav/
+// fdf/fdt` filters), while keeping the same paging / total / truncation
+// contract as `searchTranscripts`. The plain `search_transcripts` path stays on
+// the simpler `searchTranscripts` scanner above (a trivial single-leaf query),
+// so today's callers/tests are unaffected.
+//
+// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate
+// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean
+// instead of over slug sets — and the per-scope text extraction of
+// searchPipeline.ts.
+
+// The positive share-filter selection (parseShareV1), as a predicate input.
+export type SearchFilters = {
+ // ft — video / livestream types kept.
+ videos: boolean;
+ livestreams: boolean;
+ // fa — all-ages / age-restricted kept.
+ allAges: boolean;
+ restricted: boolean;
+ // fav — availability three-way (available / unlisted / deleted) kept.
+ available: boolean;
+ unlisted: boolean;
+ deleted: boolean;
+ // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD".
+ dateFrom?: string;
+ dateTo?: string;
+};
+
+export type SearchSpec = {
+ // The root of the composite query (a `qt=` tree, or a synthesized single leaf
+ // for a legacy `q`/`re` link).
+ tree: GroupNode;
+ // The decoded share filters, or null for an unfiltered search.
+ filters?: SearchFilters | null;
+ // Curated aliases to expand transcripts-scope leaves through (as in the plain
+ // path). Empty/omitted → no expansion.
+ aliases?: SearchAlias[];
+ // Expand transcripts leaves via aliases (default true; skipped per-leaf for a
+ // regex leaf).
+ useAliases?: boolean;
+};
+
+// A snippet tagged with the scope it came from, carrying the seconds needed to
+// build a moment link. `seconds` is 0 for the non-timed scopes (metadata /
+// description / tags).
+export type ScopedSnippet = {
+ scope: LayerScope;
+ track?: string;
+ clock: string;
+ seconds: number;
+ text: string;
+};
+
+export type SpecHit = {
+ videoId: string;
+ slug: string;
+ channelSlug: string;
+ channelName: string;
+ siteTitle?: string;
+ siteUrl?: string;
+ platform?: Platform;
+ title: string;
+ uploadDate: string;
+ webpageUrl?: string;
+ matches: number;
+ snippets: ScopedSnippet[];
+};
+
+export type SpecResult = {
+ hits: SpecHit[];
+ total: number;
+ offset: number;
+ limit: number;
+ hasMore: boolean;
+ firedAliases: SearchAlias[];
+ scanned: { channels: number; pages: number };
+ truncated: boolean;
+};
+
+type LeafMatcher = { scope: LayerScope; test: Matcher };
+
+// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless
+// they're regex); every other scope matches plain-substring / regex only. Fired
+// aliases are unioned for the caller to report.
+function buildLeafMatchers(
+ root: QueryNode,
+ aliases: SearchAlias[],
+ useAliases: boolean,
+): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } {
+ const matchers = new Map<string, LeafMatcher>();
+ const fired: SearchAlias[] = [];
+ forEachLeaf(root, (leaf) => {
+ if (!isLeafActive(leaf)) return;
+ const aliasAware =
+ leaf.scope === "transcripts" && !leaf.useRegex && useAliases;
+ const built = buildMatcher({
+ query: leaf.query,
+ regex: leaf.useRegex,
+ useAliases: aliasAware,
+ aliases: aliasAware ? aliases : [],
+ });
+ matchers.set(leaf.id, { scope: leaf.scope, test: built.match });
+ for (const a of built.firedAliases) {
+ if (!fired.some((x) => x.id === a.id)) fired.push(a);
+ }
+ });
+ return { matchers, fired };
+}
+
+// Per-record context the tree evaluates against.
+type RecordCtx = {
+ title: string;
+ channel: string;
+ description: string;
+ tags: string;
+ cues: Cue[];
+ chatCues: Cue[];
+ snippetsPerVideo: number;
+ includeSnippets: boolean;
+};
+
+type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] };
+
+function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome {
+ if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] };
+ const hits: ScopedSnippet[] = [];
+ const push = (s: ScopedSnippet): void => {
+ if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s);
+ };
+ let count = 0;
+ switch (m.scope) {
+ case "transcripts":
+ for (const cue of ctx.cues) {
+ if (!m.test(cue.text)) continue;
+ count++;
+ push({
+ scope: "transcripts",
+ clock: clock(cue.start),
+ seconds: cue.start,
+ text: truncate(cue.text),
+ });
+ }
+ break;
+ case "chat":
+ for (const cue of ctx.chatCues) {
+ if (!m.test(cue.text)) continue;
+ count++;
+ push({
+ scope: "chat",
+ track: "live_chat",
+ clock: clock(cue.start),
+ seconds: cue.start,
+ text: truncate(cue.text),
+ });
+ }
+ break;
+ case "metadata": {
+ const titleHit = m.test(ctx.title);
+ const channelHit = m.test(ctx.channel);
+ if (titleHit || channelHit) {
+ count++;
+ if (titleHit) {
+ push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) });
+ } else {
+ push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` });
+ }
+ }
+ break;
+ }
+ case "description":
+ if (ctx.description && m.test(ctx.description)) {
+ count++;
+ push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) });
+ }
+ break;
+ case "tags":
+ if (ctx.tags && m.test(ctx.tags)) {
+ count++;
+ push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) });
+ }
+ break;
+ }
+ return { matched: count > 0, count, hits };
+}
+
+type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] };
+
+// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate,
+// per record. A negated node contributes no hits (like the browser's `diff`).
+// An inactive (empty) subtree is identity (matches, no hits).
+function evalNode(
+ node: QueryNode,
+ matchers: Map<string, LeafMatcher>,
+ ctx: RecordCtx,
+): NodeOutcome {
+ if (!isNodeActive(node)) return { match: true, count: 0, hits: [] };
+ if (isLeaf(node)) {
+ const m = matchers.get(node.id);
+ if (!m) return { match: true, count: 0, hits: [] };
+ const r = evalLeaf(node, m, ctx);
+ if (node.negate) return { match: !r.matched, count: 0, hits: [] };
+ return {
+ match: r.matched,
+ count: node.contributeHits ? r.count : 0,
+ hits: node.contributeHits ? r.hits : [],
+ };
+ }
+ const active = node.children.filter(isNodeActive);
+ if (active.length === 0) return { match: true, count: 0, hits: [] };
+ if (node.op === "AND") {
+ let allMatch = true;
+ let count = 0;
+ const hits: ScopedSnippet[] = [];
+ for (const c of active) {
+ const r = evalNode(c, matchers, ctx);
+ if (!r.match) {
+ allMatch = false;
+ break;
+ }
+ count += r.count;
+ hits.push(...r.hits);
+ }
+ const match = node.negate ? !allMatch : allMatch;
+ return match && !node.negate
+ ? { match, count, hits }
+ : { match, count: 0, hits: [] };
+ }
+ // OR
+ let any = false;
+ let count = 0;
+ const hits: ScopedSnippet[] = [];
+ for (const c of active) {
+ const r = evalNode(c, matchers, ctx);
+ if (r.match) {
+ any = true;
+ count += r.count;
+ hits.push(...r.hits);
+ }
+ }
+ const match = node.negate ? !any : any;
+ return match && !node.negate
+ ? { match, count, hits }
+ : { match, count: 0, hits: [] };
+}
+
+// The share-filter predicate over a transcript record + its availability.
+// Mirrors SearchSessionContext.tsx `passesFilter` exactly.
+function passesFilters(
+ rec: TranscriptDetail,
+ f: SearchFilters,
+ avail: VideoAvailability | undefined,
+): boolean {
+ // ft — type
+ if (rec.isLivestream ? !f.livestreams : !f.videos) return false;
+ // fa — audience
+ if (rec.ageRestricted ? !f.restricted : !f.allAges) return false;
+ // fav — availability three-way
+ if (avail?.isDeleted) {
+ if (!f.deleted) return false;
+ } else if (avail?.isUnlisted) {
+ if (!f.unlisted) return false;
+ } else if (!f.available) {
+ return false;
+ }
+ // fdf / fdt — upload-date range (lexicographic on YYYYMMDD)
+ if (f.dateFrom && rec.uploadDate < f.dateFrom) return false;
+ if (f.dateTo && rec.uploadDate > f.dateTo) return false;
+ return true;
+}
+
+// True when the fav filter could exclude something (so availability must be
+// fetched). If every availability bucket is kept, there's nothing to look up.
+function needsAvailability(f: SearchFilters | null | undefined): boolean {
+ return !!f && (!f.available || !f.unlisted || !f.deleted);
+}
+
+// A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a
+// chat-scope leaf only pulls the subs shards it actually touches.
+function makeChatFetcher(source: ShardSource) {
+ const manifests = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsManifest"]>>>>();
+ const pages = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsPage"]>>>>();
+ return async function chatCues(ch: ChannelRef, rec: TranscriptDetail): Promise<Cue[]> {
+ let mp = manifests.get(ch.key);
+ if (!mp) {
+ mp = source.subsManifest(ch).catch(() => null);
+ manifests.set(ch.key, mp);
+ }
+ const manifest = await mp;
+ if (!manifest) return [];
+ const page = manifest.slugToPage[rec.id];
+ if (page === undefined) return [];
+ const pk = `${ch.key}:${page}`;
+ let pp = pages.get(pk);
+ if (!pp) {
+ pp = source.subsPage(ch, page).catch(() => []);
+ pages.set(pk, pp);
+ }
+ const records = await pp;
+ const found = records.find((r) => r.slug === rec.slug || r.id === rec.id);
+ const lc = found?.tracks?.live_chat;
+ return Array.isArray(lc) ? lc : [];
+ };
+}
+
+// Run a full-fidelity spec over the given (already-resolved) channel set. Same
+// paging/total/truncation semantics as searchTranscripts.
+export async function runSearchSpec(
+ source: ShardSource,
+ channels: ChannelRef[],
+ spec: SearchSpec,
+ opts: {
+ offset?: number;
+ limit?: number;
+ includeSnippets?: boolean;
+ maxPages?: number;
+ snippetsPerVideo?: number;
+ } = {},
+): Promise<SpecResult> {
+ const limit = opts.limit ?? 20;
+ const offset = Math.max(0, opts.offset ?? 0);
+ const maxPages = opts.maxPages ?? MAX_PAGES;
+ const includeSnippets = opts.includeSnippets !== false;
+ const snippetsPerVideo = opts.snippetsPerVideo ?? 4;
+ const filters = spec.filters ?? null;
+
+ const { matchers, fired } = buildLeafMatchers(
+ spec.tree,
+ spec.aliases ?? [],
+ spec.useAliases !== false,
+ );
+ const wantsChat = [...matchers.values()].some((m) => m.scope === "chat");
+ const chatCuesFor = wantsChat ? makeChatFetcher(source) : null;
+ const availability = needsAvailability(filters)
+ ? await source.availabilityMap()
+ : null;
+
+ const all: SpecHit[] = [];
+ let pagesScanned = 0;
+ let channelsScanned = 0;
+ let truncated = false;
+
+ outer: for (const ch of channels) {
+ let manifest;
+ try {
+ manifest = await source.transcriptsManifest(ch);
+ } catch {
+ continue;
+ }
+ channelsScanned++;
+ for (let page = 0; page < manifest.pageCount; page++) {
+ if (pagesScanned >= maxPages) {
+ truncated = true;
+ break outer;
+ }
+ let records: TranscriptDetail[];
+ try {
+ records = await source.transcriptPage(ch, page);
+ } catch {
+ continue;
+ }
+ pagesScanned++;
+ for (const rec of records) {
+ if (filters && !passesFilters(rec, filters, availability?.get(rec.slug))) {
+ continue;
+ }
+ const ctx: RecordCtx = {
+ title: rec.title ?? "",
+ channel: rec.channel ?? ch.name,
+ description: rec.description ?? "",
+ tags: (rec.tags ?? []).join(", "),
+ cues: rec.cues ?? [],
+ chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [],
+ snippetsPerVideo,
+ includeSnippets,
+ };
+ const r = evalNode(spec.tree, matchers, ctx);
+ if (!r.match) continue;
+ all.push({
+ videoId: rec.id,
+ slug: rec.slug,
+ channelSlug: ch.slug,
+ channelName: ch.name,
+ ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}),
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(rec.platform ? { platform: rec.platform } : {}),
+ title: rec.title,
+ uploadDate: rec.uploadDate,
+ webpageUrl: rec.webpageUrl,
+ matches: r.count || 1,
+ snippets: r.hits,
+ });
+ if (all.length >= HARD_VIDEO_CAP) {
+ truncated = true;
+ break outer;
+ }
+ }
+ }
+ }
+
+ const total = all.length;
+ return {
+ hits: all.slice(offset, offset + limit),
+ total,
+ offset,
+ limit,
+ hasMore: offset + limit < total,
+ firedAliases: fired,
+ scanned: { channels: channelsScanned, pages: pagesScanned },
+ truncated,
+ };
+}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -7,6 +7,8 @@ import {
} from "@modelcontextprotocol/sdk/types.js";
import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
+import { momentUrl } from "yt-dlp-transcript-common/lib/momentUrl";
+import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import {
sortGroups,
@@ -14,7 +16,13 @@ import {
FALLBACK_GROUP,
type ChannelGroup,
} from "yt-dlp-transcript-common/lib/channelGroups";
-import { HubSource, type ChannelRef, type HubSite, type ShardSource } from "./source";
+import {
+ HubSource,
+ RemoteSource,
+ type ChannelRef,
+ type HubSite,
+ type ShardSource,
+} from "./source";
import type { SourceSpec } from "./sources";
import type { SourceController } from "./sourceController";
import {
@@ -22,8 +30,56 @@ import {
findVideo,
buildMatcher,
getWindowedTranscript,
+ runSearchSpec,
type SearchResult,
+ type SpecHit,
+ type ScopedSnippet,
} from "./search";
+import {
+ decodeShareLink,
+ applyLinkOverrides,
+ renderQueryTree,
+ describeFilters,
+ type DecodedLink,
+ type LinkOverrides,
+} from "./shareLink";
+
+// The building blocks momentUrl needs from a hit (viewer origin comes from the
+// per-video siteUrl, else the source's single public origin).
+type LinkableHit = {
+ slug: string;
+ siteUrl?: string;
+ webpageUrl?: string;
+ platform?: Platform;
+};
+
+// A moment deep link for one cited second of a hit, or null when neither a
+// viewer nor a platform link can be built (e.g. a local source + no webpage url).
+function momentLinkFor(
+ source: ShardSource,
+ h: LinkableHit,
+ seconds: number,
+): string | null {
+ return momentUrl({
+ siteOrigin: h.siteUrl ?? source.publicOrigin(),
+ slug: h.slug,
+ seconds,
+ webpageUrl: h.webpageUrl,
+ platform: h.platform,
+ });
+}
+
+// A `[m:ss]` timestamp rendered as a Markdown link when we can build one, else
+// bare. Used inside a `- [ … ] text` snippet line → `- [[m:ss](url)] text`.
+function stampMarkup(
+ source: ShardSource,
+ h: LinkableHit,
+ clock: string,
+ seconds: number,
+): string {
+ const url = momentLinkFor(source, h, seconds);
+ return url ? `[${clock}](${url})` : clock;
+}
type ToolResult = {
content: { type: "text"; text: string }[];
@@ -291,6 +347,103 @@ const TOOLS = [
"--local flag or env) and clear the persisted selection.",
inputSchema: { type: "object", properties: {}, additionalProperties: false },
},
+ {
+ name: "open_link",
+ description:
+ "Paste an archilyzer viewer **share link** (origin + a `qt=` query tree + " +
+ "`fc/ft/fa/fav/fdf/fdt` filters) to re-run that exact search here — at full " +
+ "fidelity, with clickable timestamped result links. Two-phase: by default " +
+ "(apply:false) it PREVIEWS — decodes the link and returns a plan (resolved " +
+ "source, the query tree rendered readably, every active filter, the channel " +
+ "scope validated against the live corpus, and anything ignored, e.g. the " +
+ "vestigial `fk` tracks) WITHOUT switching source or searching. Adjust in " +
+ "natural language by re-calling with `overrides` (e.g. clear_availability " +
+ "to drop the availability filter, channels to re-scope, query to replace " +
+ "the search). When the plan looks right, call again with apply:true: it " +
+ "switches the active source to the link's origin (hub or single-site, " +
+ "auto-detected) and returns the first page of linked results. Read-only.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ link: {
+ type: "string",
+ description: "The archilyzer viewer share URL to decode and run.",
+ },
+ apply: {
+ type: "boolean",
+ description:
+ "false (default) previews the plan only; true switches source and " +
+ "runs the search.",
+ },
+ overrides: {
+ type: "object",
+ description:
+ "Structured edits to the decoded search, so a natural-language " +
+ "adjustment can be honored by re-calling with the changed facet.",
+ properties: {
+ clear_filters: { type: "boolean", description: "Drop all filters." },
+ clear_availability: {
+ type: "boolean",
+ description: "Remove the availability (deleted/unlisted) filter.",
+ },
+ clear_type: {
+ type: "boolean",
+ description: "Remove the video/livestream type filter.",
+ },
+ clear_age: {
+ type: "boolean",
+ description: "Remove the all-ages/age-restricted filter.",
+ },
+ clear_dates: {
+ type: "boolean",
+ description: "Remove the upload-date range filter.",
+ },
+ clear_channels: {
+ type: "boolean",
+ description: "Widen the channel scope to the whole corpus.",
+ },
+ channels: {
+ type: "array",
+ items: { type: "string" },
+ description: "Replace the channel scope with these channel names.",
+ },
+ date_from: {
+ type: "string",
+ description: "Set the inclusive lower upload-date bound (YYYYMMDD).",
+ },
+ date_to: {
+ type: "string",
+ description: "Set the inclusive upper upload-date bound (YYYYMMDD).",
+ },
+ query: {
+ type: "string",
+ description: "Replace the whole query with this single term.",
+ },
+ regex: {
+ type: "boolean",
+ description: "Treat the override query as a regex.",
+ },
+ query_scope: {
+ type: "string",
+ enum: ["transcripts", "chat", "metadata", "description", "tags"],
+ description: "Scope for the override query (default transcripts).",
+ },
+ },
+ additionalProperties: false,
+ },
+ limit: {
+ type: "number",
+ description: "Results per page when apply:true (default 20).",
+ },
+ offset: {
+ type: "number",
+ description: "Skip this many matches when apply:true (default 0).",
+ },
+ },
+ required: ["link"],
+ additionalProperties: false,
+ },
+ },
];
// The slice of SourceController the server needs. A bare ShardSource is wrapped
@@ -378,6 +531,8 @@ export function createServer(sourceOrController: ShardSource | SourceController)
return await handleUseSource(controller, args);
case "reset_source":
return await handleResetSource(controller);
+ case "open_link":
+ return await handleOpenLink(controller, args);
default:
return errorText(`unknown tool: ${name}`);
}
@@ -508,7 +663,9 @@ async function handleSearch(
(h.siteTitle ? ` | site: ${h.siteTitle}` : "") +
` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` +
(h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "");
- const snips = h.snippets.map((s) => ` - [${s.clock}] ${s.text}`).join("\n");
+ const snips = h.snippets
+ .map((s) => ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`)
+ .join("\n");
return snips ? `${head}\n${snips}` : head;
});
return text(
@@ -604,6 +761,14 @@ async function handleGetTranscripts(
continue;
}
const { record, ch } = found;
+ const link: LinkableHit = {
+ slug: record.slug,
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
+ ...(record.platform ? { platform: record.platform } : {}),
+ };
+ const linkForSeconds = (seconds: number): string | null =>
+ momentLinkFor(source, link, seconds);
const head =
`## ${record.title || id}\n` +
`- video_id: ${id} | channel: ${ch.name}` +
@@ -616,6 +781,7 @@ async function handleGetTranscripts(
before,
after,
timestamps,
+ link: linkForSeconds,
});
const body =
matchCount === 0
@@ -626,6 +792,7 @@ async function handleGetTranscripts(
const md = transcriptToMarkdown(record, {
timestamps,
includeTags: true,
+ linkForCue: linkForSeconds,
});
blocks.push(md.trim());
}
@@ -658,9 +825,17 @@ async function handleGetTranscript(
typeof args.channel === "string" ? args.channel : undefined,
);
if (!found) return errorText(`video not found: ${videoId}`);
- const md = transcriptToMarkdown(found.record, {
+ const { record, ch } = found;
+ const link: LinkableHit = {
+ slug: record.slug,
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
+ ...(record.platform ? { platform: record.platform } : {}),
+ };
+ const md = transcriptToMarkdown(record, {
timestamps: args.timestamps !== false,
includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
});
return text(md);
}
@@ -896,6 +1071,271 @@ async function handleResetSource(
);
}
+// ─── open_link: paste a share link → preview, adjust, apply ───
+
+// Map the snake_cased tool `overrides` object onto the LinkOverrides shape.
+function parseOverrides(raw: unknown): LinkOverrides | undefined {
+ if (!raw || typeof raw !== "object") return undefined;
+ const o = raw as Record<string, unknown>;
+ const bool = (k: string): boolean | undefined =>
+ o[k] === true ? true : undefined;
+ const str = (k: string): string | undefined =>
+ typeof o[k] === "string" ? (o[k] as string) : undefined;
+ const scope = str("query_scope");
+ return {
+ clearFilters: bool("clear_filters"),
+ clearAvailability: bool("clear_availability"),
+ clearType: bool("clear_type"),
+ clearAge: bool("clear_age"),
+ clearDates: bool("clear_dates"),
+ clearChannels: bool("clear_channels"),
+ channels: strArray(o.channels),
+ dateFrom: str("date_from"),
+ dateTo: str("date_to"),
+ query: str("query"),
+ regex: o.regex === true ? true : undefined,
+ queryScope:
+ scope === "transcripts" ||
+ scope === "chat" ||
+ scope === "metadata" ||
+ scope === "description" ||
+ scope === "tags"
+ ? scope
+ : undefined,
+ };
+}
+
+type OriginProbe = { kind: "hub" | "remote"; siteCount?: number; channelCount?: number };
+
+// Probe an origin's corpus.json to classify it as a federated hub (has a
+// `sites[]` array) or a single site (has `channels[]`). Transient — no source is
+// committed. Defaults to single-site when the probe can't be read.
+async function probeOrigin(origin: string): Promise<OriginProbe> {
+ try {
+ const res = await fetch(`${origin}/corpus.json`);
+ if (res.ok) {
+ const j = (await res.json()) as {
+ sites?: unknown[];
+ channels?: unknown[];
+ };
+ if (Array.isArray(j.sites)) return { kind: "hub", siteCount: j.sites.length };
+ if (Array.isArray(j.channels)) {
+ return { kind: "remote", channelCount: j.channels.length };
+ }
+ }
+ } catch {
+ // unreachable — fall through to the single-site default
+ }
+ return { kind: "remote" };
+}
+
+type ChannelScope = {
+ channels: ChannelRef[];
+ all: boolean;
+ matched: string[];
+ unknown: string[];
+};
+
+// Resolve a decoded link's `fc` channel NAMES against a live source's channels
+// (by display name or slug, case-insensitive). No channel filter → whole corpus.
+async function resolveLinkChannels(
+ source: ShardSource,
+ decoded: DecodedLink,
+): Promise<ChannelScope> {
+ const all = await source.listChannels();
+ if (!decoded.hasChannelFilter) {
+ return { channels: all, all: true, matched: [], unknown: [] };
+ }
+ const want = decoded.channelNames.map((n) => n.toLowerCase());
+ const matches = (c: ChannelRef): boolean =>
+ want.includes(c.name.toLowerCase()) || want.includes(c.slug.toLowerCase());
+ const channels = all.filter(matches);
+ const matched = channels.map((c) => c.name);
+ const unknown = decoded.channelNames.filter(
+ (n) =>
+ !all.some(
+ (c) =>
+ c.name.toLowerCase() === n.toLowerCase() ||
+ c.slug.toLowerCase() === n.toLowerCase(),
+ ),
+ );
+ return { channels, all: false, matched, unknown };
+}
+
+// Render the human-readable preview/echo plan for a decoded link.
+function buildLinkPlan(
+ decoded: DecodedLink,
+ probe: OriginProbe,
+ scope: ChannelScope,
+ applied: boolean,
+): string {
+ const lines: string[] = [];
+ lines.push(applied ? "## Applied share link" : "## Share-link preview");
+ const kindNote =
+ probe.kind === "hub"
+ ? `hub (${probe.siteCount ?? "?"} member site(s))`
+ : `single site${probe.channelCount != null ? ` (${probe.channelCount} channel(s))` : ""}`;
+ lines.push(`- Source: ${decoded.origin} — ${kindNote}`);
+ lines.push(`- Query: ${renderQueryTree(decoded.tree)} _(from ${decoded.querySource})_`);
+
+ if (scope.all) {
+ lines.push(`- Scope: whole corpus`);
+ } else {
+ const names = scope.matched.length ? scope.matched.join(", ") : "(none)";
+ lines.push(`- Scope: ${scope.channels.length} channel(s): ${names}`);
+ if (scope.unknown.length > 0) {
+ lines.push(` - ⚠ channel name(s) not found in this corpus: ${scope.unknown.join(", ")}`);
+ }
+ if (scope.channels.length === 0) {
+ lines.push(
+ ` - ⚠ the link selects 0 channels here — nothing will match. Re-call ` +
+ `with overrides.clear_channels to search the whole corpus.`,
+ );
+ }
+ }
+
+ const filterLines = describeFilters(decoded.filters);
+ if (filterLines.length > 0) {
+ lines.push(`- Filters:`);
+ for (const f of filterLines) lines.push(` - ${f}`);
+ } else {
+ lines.push(`- Filters: none`);
+ }
+
+ for (const w of decoded.warnings) lines.push(`- Note: ${w}`);
+
+ if (!applied) {
+ lines.push(
+ ``,
+ `No search run yet. Call again with apply:true to switch source and ` +
+ `search, or pass overrides to adjust first.`,
+ );
+ }
+ return lines.join("\n");
+}
+
+// A ScopedSnippet line: a linked `[m:ss]` for timed scopes, or a scope-tagged
+// note for the non-timed scopes (metadata / description / tags).
+function renderScopedSnippet(
+ source: ShardSource,
+ hit: SpecHit,
+ s: ScopedSnippet,
+): string {
+ if (s.seconds > 0) {
+ const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `;
+ return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`;
+ }
+ return ` - [${s.scope}] ${s.text}`;
+}
+
+function renderSpecResults(source: ShardSource, hits: SpecHit[]): string {
+ return hits
+ .map((h) => {
+ const head =
+ `### ${h.title}\n` +
+ `- video_id: ${h.videoId} | channel: ${h.channelName}` +
+ (h.siteTitle ? ` | site: ${h.siteTitle}` : "") +
+ ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` +
+ (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "");
+ const snips = h.snippets
+ .map((s) => renderScopedSnippet(source, h, s))
+ .join("\n");
+ return snips ? `${head}\n${snips}` : head;
+ })
+ .join("\n\n");
+}
+
+async function handleOpenLink(
+ controller: SourceControllerLike,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const linkStr = typeof args.link === "string" ? args.link.trim() : "";
+ if (!linkStr) return errorText("link is required");
+
+ let decoded: DecodedLink;
+ try {
+ decoded = decodeShareLink(linkStr);
+ } catch (e) {
+ return errorText(`could not parse link as a URL: ${(e as Error).message}`);
+ }
+ decoded = applyLinkOverrides(decoded, parseOverrides(args.overrides));
+
+ const apply = args.apply === true;
+ const probe = await probeOrigin(decoded.origin);
+
+ if (!apply) {
+ // Preview: resolve channels against a transient source; do not commit.
+ const preview: ShardSource =
+ probe.kind === "hub"
+ ? new HubSource(decoded.origin)
+ : new RemoteSource(decoded.origin);
+ let scope: ChannelScope;
+ try {
+ scope = await resolveLinkChannels(preview, decoded);
+ } catch (e) {
+ return errorText(
+ `could not read the corpus at ${decoded.origin}: ${(e as Error).message}`,
+ );
+ }
+ return text(buildLinkPlan(decoded, probe, scope, false));
+ }
+
+ // Apply: commit the source (reusing the controller), then search.
+ const spec: SourceSpec =
+ probe.kind === "hub"
+ ? { kind: "hub", url: decoded.origin }
+ : { kind: "remote", url: decoded.origin };
+ try {
+ await controller.switchTo(spec);
+ } catch (e) {
+ return errorText(`open_link could not switch source: ${(e as Error).message}`);
+ }
+ const source = controller.current;
+
+ let scope: ChannelScope;
+ try {
+ scope = await resolveLinkChannels(source, decoded);
+ } catch (e) {
+ return errorText(
+ `switched to ${source.label} but could not read its corpus: ${(e as Error).message}`,
+ );
+ }
+ const plan = buildLinkPlan(decoded, probe, scope, true);
+
+ const limit = typeof args.limit === "number" ? args.limit : 20;
+ const offset = typeof args.offset === "number" ? args.offset : 0;
+ const result = await runSearchSpec(
+ source,
+ scope.channels,
+ {
+ tree: decoded.tree,
+ filters: decoded.filters,
+ aliases: await source.loadAliases(),
+ },
+ { limit, offset },
+ );
+
+ const aliasNote = describeFiredAliases(result.firedAliases);
+ const rangeStart = result.total === 0 ? 0 : result.offset + 1;
+ const rangeEnd = result.offset + result.hits.length;
+ const footer =
+ `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` +
+ `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` +
+ `page(s) across ${result.scanned.channels} channel(s)` +
+ (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") +
+ (aliasNote ? `; ${aliasNote}` : "") +
+ ")";
+
+ const body =
+ result.hits.length === 0
+ ? result.total === 0
+ ? "No matches for this link's search."
+ : `No matches in this page (offset ${result.offset} is past the ${result.total} total).`
+ : `${result.total} video(s) matched:\n\n${renderSpecResults(source, result.hits)}`;
+
+ return text(`${plan}\n\n---\n\n${body}${footer}`);
+}
+
// ─── Prompts: the first-class `sweep` entry point ───
// A single slash command that drives Claude Code to run the browser's
// "corpus sweep" on plan usage: enumerate a query's full match set, batch the
@@ -912,7 +1352,21 @@ const PROMPTS = [
"on plan usage, no API key. With no scope arg it lists the channel groups " +
"and asks you to pick a group/channels (or confirm 'all') before sweeping.",
arguments: [
- { name: "query", description: "Term or phrase to sweep for.", required: true },
+ {
+ name: "query",
+ description:
+ "Term or phrase to sweep for. Optional when a `link` is given (the " +
+ "link supplies the query).",
+ required: false,
+ },
+ {
+ name: "link",
+ description:
+ "An archilyzer viewer share URL to seed the sweep from — its origin " +
+ "(source), query tree, and filters are decoded and applied via " +
+ "open_link before enumerating.",
+ required: false,
+ },
{
name: "channel",
description: "Optional channel slug/name to restrict the sweep to.",
@@ -958,7 +1412,10 @@ function argStr(args: Record<string, unknown>, key: string): string | undefined
function buildSweepPrompt(args: Record<string, unknown>) {
const query = argStr(args, "query");
- if (!query) throw new Error("sweep requires a query argument");
+ const link = argStr(args, "link");
+ if (!query && !link) {
+ throw new Error("sweep requires a query argument (or a link)");
+ }
const channel = argStr(args, "channel");
const group = argStr(args, "group");
const channelsRaw = argStr(args, "channels");
@@ -968,10 +1425,11 @@ function buildSweepPrompt(args: Record<string, unknown>) {
const directive = argStr(args, "directive") ?? "key claims & contradictions";
const batchSize = argStr(args, "batch_size") ?? "8";
const reportPath = argStr(args, "report_path") ?? "./sweep-report.md";
+ const subject = query ? `"${query}"` : "the share link's search";
// Human-readable scope clauses + the literal search_transcripts scope args to
- // pass. When none is given, the sweep must pick-first (list + ask) rather than
- // silently scanning the whole corpus.
+ // pass. When none is given (and there's no link), the sweep must pick-first
+ // (list + ask) rather than silently scanning the whole corpus.
const scopeClauses: string[] = [];
if (channel) scopeClauses.push(`channel "${channel}"`);
if (channels.length > 0)
@@ -980,70 +1438,107 @@ function buildSweepPrompt(args: Record<string, unknown>) {
const hasScope = scopeClauses.length > 0;
const scopeArgsText = scopeClauses.join(" and ");
- const introScope = hasScope
- ? ` scoped to ${scopeArgsText}`
- : " over a scope you will confirm with me first (see step 1)";
+ const introScope = link
+ ? ` seeded from a share link (its source, query tree, and filters — see step 1)`
+ : hasScope
+ ? ` scoped to ${scopeArgsText}`
+ : " over a scope you will confirm with me first (see step 1)";
- // Build the numbered steps. With an explicit scope we search directly; with
- // none we insert a pick-first step and the Search step uses the chosen scope.
const steps: string[] = [];
- if (!hasScope) {
+ // ── Discovery / enumeration differs for a link-seeded sweep vs a query one ──
+ if (link) {
+ steps.push(
+ `**Decode & confirm the link.** Call \`open_link\` with ` +
+ `link="${link}" (preview mode — no apply). It returns a plan: the ` +
+ `resolved source (origin, hub or single-site), the query tree, every ` +
+ `active filter, the channel scope validated against the corpus, and any ` +
+ `ignored bits (e.g. the vestigial \`fk\` tracks). **Show me the plan and ` +
+ `confirm it captures what I want.** If I ask for a change ("drop the ` +
+ `availability filter", "only channel X", "search Y instead"), re-call ` +
+ `\`open_link\` with the matching \`overrides\` until the plan is right.`,
+ );
+ steps.push(
+ `**Apply & enumerate.** Call \`open_link\` again with the confirmed ` +
+ `arguments plus apply:true — this switches the active source to the ` +
+ `link's origin and runs the search (the full query tree + filters, at ` +
+ `fidelity). Page it with a rising \`offset\` (offset += limit) until ` +
+ `\`has_more\` is no to collect the whole worklist of video ids; note the ` +
+ `\`total\`. Results already carry linked \`[mm:ss](url)\` timestamps and ` +
+ `the applied plan is echoed at the top — record the source, query, and ` +
+ `filters in the report. If coverage is PARTIAL, say so.`,
+ );
+ } else {
+ if (!hasScope) {
+ steps.push(
+ `**Choose the scope first — do NOT default to the whole corpus.** No ` +
+ `channel/channels/group was supplied. Call \`list_channels\`, present ` +
+ `the channel groups and their channels to me, and ask which group(s) ` +
+ `or channel(s) to sweep — or to confirm **all** for the whole corpus. ` +
+ `Wait for my choice before enumerating anything. Only sweep everything ` +
+ `if I explicitly choose "all". Use my choice as the ` +
+ `\`channel\`/\`channels\`/\`group\` scope in every ` +
+ `\`search_transcripts\` call below.`,
+ );
+ }
steps.push(
- `**Choose the scope first — do NOT default to the whole corpus.** No ` +
- `channel/channels/group was supplied. Call \`list_channels\`, present ` +
- `the channel groups and their channels to me, and ask which group(s) or ` +
- `channel(s) to sweep — or to confirm **all** for the whole corpus. Wait ` +
- `for my choice before enumerating anything. Only sweep everything if I ` +
- `explicitly choose "all". Use my choice as the \`channel\`/\`channels\`/` +
- `\`group\` scope in every \`search_transcripts\` call below.`,
+ `**Search.** Call \`search_transcripts\` with query "${query}"` +
+ (hasScope
+ ? ` and ${scopeArgsText}`
+ : ` and the scope I chose in step 1`) +
+ `. Curated aliases auto-expand the query — the footer reports which ` +
+ `fired (e.g. mis-transcribed spellings) and names the resolved scope ` +
+ `(and warns about any channel/group token that matched nothing — fix a ` +
+ `typo before continuing). Treat the *expanded* match set as your target ` +
+ `and mention the expansion and the scope in the report.`,
+ );
+ steps.push(
+ `**Enumerate the full worklist.** Page the complete set with ` +
+ `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` +
+ `until \`has_more\` is false — this gives you every id/title/channel/date ` +
+ `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` +
+ `(page/video cap), say so in the report — the sweep is then a sample, ` +
+ `not exhaustive.`,
);
}
steps.push(
- `**Search.** Call \`search_transcripts\` with query "${query}"` +
- (hasScope
- ? ` and ${scopeArgsText}`
- : ` and the scope I chose in step 1`) +
- `. Curated aliases auto-expand the query — the footer reports which fired ` +
- `(e.g. mis-transcribed spellings) and names the resolved scope (and warns ` +
- `about any channel/group token that matched nothing — fix a typo before ` +
- `continuing). Treat the *expanded* match set as your target and mention ` +
- `the expansion and the scope in the report.`,
- );
-
- steps.push(
- `**Enumerate the full worklist.** Page the complete set with ` +
- `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` +
- `until \`has_more\` is false — this gives you every id/title/channel/date ` +
- `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` +
- `(page/video cap), say so in the report — the sweep is then a sample, not ` +
- `exhaustive.`,
- );
-
- steps.push(
`**Plan.** With N total matches and a batch size of ${batchSize}, that is ` +
`\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch ` +
`count) before you start.`,
);
+ // ── Feature 2: subagent-per-batch map-reduce so the heavy transcript text
+ // lives only in ephemeral subagents; the orchestrator keeps just the distilled
+ // cited findings. Feature 1: linked citations built from the moment URLs.
steps.push(
- `**Per batch**, for each group of up to ${batchSize} video ids:\n` +
- ` - Call \`get_transcripts\` with those ids **and the query** so each ` +
- `transcript comes back as bounded, timestamped excerpt windows around the ` +
- `matches (alias-correct, high-signal).\n` +
- ` - **Cross-reference** the batch against the report so far. Upsert ` +
- `findings — claims, and contradictions with earlier claims — into ` +
- `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` +
- ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it ` +
- `each batch), then **drop the raw transcript text** once folded — don't ` +
- `carry it forward.`,
+ `**Per batch (map-reduce), for each group of up to ${batchSize} video ids:**\n` +
+ ` - **Spawn a subagent** (the Task tool) for the batch. Give it the ` +
+ `batch's ids, the query, and the directive, and tell it to: call ` +
+ `\`get_transcripts\` with those ids **and the query** (bounded, ` +
+ `timestamped, alias-correct excerpt windows), extract only what serves the ` +
+ `directive, and **return a compact markdown fragment of cited, linked ` +
+ `findings and nothing else** — the raw transcript text stays inside the ` +
+ `subagent and never enters your context. Every citation must be a link: ` +
+ `**\`[title @ mm:ss](<moment url>)\`**, where the moment URL is taken ` +
+ `straight from the \`[mm:ss](url)\` links in that batch's ` +
+ `\`get_transcripts\` output (they seek to the exact second).\n` +
+ ` - **Merge** the returned fragment into \`${reportPath}\`: cross-` +
+ `reference it against the report so far and upsert findings — claims, and ` +
+ `contradictions with earlier claims — into well-titled \`## sections\` ` +
+ `(Write/Edit). Then discard the fragment. Batches are independent, so you ` +
+ `may dispatch several subagents in parallel.\n` +
+ ` - **Fallback:** if no subagent/Task tool is available, do the batch ` +
+ `inline — call \`get_transcripts\` yourself, fold the cited linked ` +
+ `findings into the report, then **drop the raw transcript text** before ` +
+ `moving on (don't carry it forward).`,
);
steps.push(
`**Finish.** Repeat to the end of the worklist, then write a short summary ` +
`section (the scope swept, how many videos covered, headline findings, any ` +
- `partial-coverage caveat) and tell me the report path.`,
+ `partial-coverage caveat) and tell me the report path. Keep every citation ` +
+ `a clickable \`[title @ mm:ss](url)\` link.`,
);
const numbered = steps
@@ -1051,16 +1546,18 @@ function buildSweepPrompt(args: Record<string, unknown>) {
.join("\n\n");
const text =
- `Run a **corpus sweep** for the query **"${query}"**${introScope}, ` +
- `extracting **${directive}**, and maintain a running markdown report at ` +
+ `Run a **corpus sweep** for ${subject}${introScope}, extracting ` +
+ `**${directive}**, and maintain a running markdown report at ` +
`\`${reportPath}\`. You are the sweep engine — work through the whole match ` +
`set methodically, using the transcript MCP tools for evidence and your own ` +
`Write/Edit tools for the report. The MCP is read-only; never try to change ` +
- `the archive.\n\n` +
+ `the archive. **Cite every finding as a clickable ` +
+ `\`[title @ mm:ss](<moment url>)\` link** (the moment URLs come straight ` +
+ `from the tool output).\n\n` +
`Follow these steps:\n\n${numbered}`;
return {
- description: `Corpus sweep for "${query}" → ${reportPath}`,
+ description: `Corpus sweep for ${subject} → ${reportPath}`,
messages: [
{
role: "user" as const,
diff --git a/mcp/src/shareLink.test.ts b/mcp/src/shareLink.test.ts
@@ -0,0 +1,168 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ newGroup,
+ newLeaf,
+ stringifyRoot,
+} from "yt-dlp-transcript-common/lib/searchQuery";
+import {
+ decodeShareLink,
+ applyLinkOverrides,
+ renderQueryTree,
+ describeFilters,
+} from "./shareLink";
+
+const ORIGIN = "https://rekietalyzer.pages.dev";
+
+// A realistic composite qt tree: transcript:"coffee" AND NOT tags:"espresso".
+function qtParam(): string {
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags", negate: true }),
+ ],
+ });
+ return encodeURIComponent(stringifyRoot(tree));
+}
+
+// A full-fidelity share link: qt tree + every filter facet + a vestigial fk.
+function fullLink(): string {
+ const p = new URLSearchParams();
+ p.set("qt", "__QT__");
+ p.set("fv", "1");
+ p.append("fc", "Rekieta Law");
+ p.append("fc", "Rekieta Law (Rumble)");
+ p.append("ft", "v"); // videos only
+ p.append("fa", "a"); // all-ages only
+ p.append("fav", "a"); // available only
+ p.append("fk", "live_chat"); // vestigial
+ p.set("fdf", "20250101");
+ p.set("fdt", "20251231");
+ // qt must not be URL-double-encoded by URLSearchParams — splice it in raw.
+ return `${ORIGIN}/?${p.toString().replace("__QT__", qtParam())}`;
+}
+
+test("decode: origin is extracted from the pasted URL", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=x`);
+ assert.equal(d.origin, ORIGIN);
+});
+
+test("decode: qt tree parses to a composite query", () => {
+ const d = decodeShareLink(fullLink());
+ assert.equal(d.querySource, "qt");
+ const rendered = renderQueryTree(d.tree);
+ assert.match(rendered, /transcript:"coffee"/);
+ assert.match(rendered, /NOT tags:"espresso"/);
+ assert.match(rendered, /AND/);
+});
+
+test("decode: every v1 filter facet is decoded", () => {
+ const d = decodeShareLink(fullLink());
+ assert.ok(d.filters);
+ const f = d.filters!;
+ // ft=v → videos kept, livestreams dropped.
+ assert.equal(f.videos, true);
+ assert.equal(f.livestreams, false);
+ // fa=a → all-ages kept, restricted dropped.
+ assert.equal(f.allAges, true);
+ assert.equal(f.restricted, false);
+ // fav=a → available kept, unlisted/deleted dropped.
+ assert.equal(f.available, true);
+ assert.equal(f.unlisted, false);
+ assert.equal(f.deleted, false);
+ // dates.
+ assert.equal(f.dateFrom, "20250101");
+ assert.equal(f.dateTo, "20251231");
+});
+
+test("decode: fc channel names are captured; fk is decoded but flagged ignored", () => {
+ const d = decodeShareLink(fullLink());
+ assert.deepEqual(d.channelNames.sort(), [
+ "Rekieta Law",
+ "Rekieta Law (Rumble)",
+ ]);
+ assert.equal(d.hasChannelFilter, true);
+ assert.deepEqual(d.ignoredTracks, ["live_chat"]);
+ assert.ok(d.warnings.some((w) => /fk/.test(w) && /ignored/.test(w)));
+});
+
+test("decode: legacy q/re → a single regex transcripts leaf, no filters", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=coffee&re=1`);
+ assert.equal(d.querySource, "legacy");
+ assert.equal(d.filters, null);
+ assert.equal(d.hasChannelFilter, false);
+ const leaf = d.tree.children[0];
+ assert.equal(leaf.kind, "leaf");
+ assert.match(renderQueryTree(d.tree), /transcript:\/coffee\//);
+});
+
+test("decode: legacy m=subs → a live-chat leaf", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=hello&m=subs`);
+ assert.match(renderQueryTree(d.tree), /live-chat:"hello"/);
+});
+
+test("decode: no fv block → unfiltered, whole-corpus scope", () => {
+ const d = decodeShareLink(`${ORIGIN}/?fc=Some%20Channel`);
+ assert.equal(d.filters, null);
+ assert.equal(d.hasChannelFilter, false);
+ assert.deepEqual(d.channelNames, []);
+});
+
+test("decode: malformed qt falls back to legacy/empty with a warning", () => {
+ const d = decodeShareLink(`${ORIGIN}/?qt=not-json&q=fallback`);
+ assert.equal(d.querySource, "legacy");
+ assert.ok(d.warnings.some((w) => /malformed/.test(w)));
+ assert.match(renderQueryTree(d.tree), /transcript:"fallback"/);
+});
+
+// ─── overrides ───
+
+test("override: clearAvailability resets the fav facet to keep-all", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearAvailability: true,
+ });
+ assert.equal(d.filters!.available, true);
+ assert.equal(d.filters!.unlisted, true);
+ assert.equal(d.filters!.deleted, true);
+ assert.ok(d.warnings.some((w) => /availability filter removed/.test(w)));
+});
+
+test("override: clearFilters drops the whole filter block", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearFilters: true,
+ });
+ assert.equal(d.filters, null);
+});
+
+test("override: channels replaces the channel scope", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ channels: ["Only This"],
+ });
+ assert.deepEqual(d.channelNames, ["Only This"]);
+ assert.equal(d.hasChannelFilter, true);
+});
+
+test("override: clearChannels widens to the whole corpus", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearChannels: true,
+ });
+ assert.deepEqual(d.channelNames, []);
+ assert.equal(d.hasChannelFilter, false);
+});
+
+test("override: query replaces the tree with a single leaf", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ query: "different term",
+ });
+ assert.match(renderQueryTree(d.tree), /transcript:"different term"/);
+});
+
+test("describeFilters: names the constrained facets only", () => {
+ const d = decodeShareLink(fullLink());
+ const lines = describeFilters(d.filters);
+ assert.ok(lines.some((l) => /videos only/.test(l)));
+ assert.ok(lines.some((l) => /all-ages only/.test(l)));
+ assert.ok(lines.some((l) => /available.*kept/.test(l)));
+ assert.ok(lines.some((l) => /uploaded/.test(l)));
+});
diff --git a/mcp/src/shareLink.ts b/mcp/src/shareLink.ts
@@ -0,0 +1,316 @@
+// Decode an archilyzer viewer **share link** into a search spec + source target.
+//
+// A share URL is `origin` + a `qt=` composite query tree (or a legacy `q`/`re`)
+// + the v1 filter params (`fc/ft/fa/fav/fk/fdf/fdt`, gated by `fv=1`). This
+// module turns that URL into the pieces the MCP needs to reproduce the search at
+// full fidelity:
+// • the origin (→ probed by the server for hub vs single-site),
+// • the query tree (parsed via the browser's own `parseRoot`),
+// • the decoded filters (via the browser's own `parseShareV1`),
+// • the requested channel NAMES (`fc`, resolved against the live corpus later),
+// • and any ignored/warned pieces (`fk` tracks are vestigial in the composite
+// share model — decoded and reported as ignored, not enforced).
+//
+// Pure: no network. The server does the origin probe and channel resolution.
+
+import {
+ parseRoot,
+ rootFromLegacy,
+ emptyRoot,
+ isLeaf,
+ isGroup,
+ isNodeActive,
+ newGroup,
+ newLeaf,
+ type GroupNode,
+ type QueryNode,
+} from "yt-dlp-transcript-common/lib/searchQuery";
+import {
+ hasShareV1,
+ parseShareV1,
+} from "yt-dlp-transcript-common/components/shareUrl";
+import type { SearchFilters } from "./search";
+
+export type DecodedLink = {
+ // The pasted URL's origin (scheme + host[:port]). The source target.
+ origin: string;
+ // The composite query tree — from `qt=`, a synthesized legacy leaf, or an
+ // empty root (filter-only browse).
+ tree: GroupNode;
+ querySource: "qt" | "legacy" | "empty";
+ // Positive share filters (ft/fa/fav + dates), or null when the link carries no
+ // v1 filter block (`fv=1`) — i.e. an unfiltered search.
+ filters: SearchFilters | null;
+ // Channel NAMES the link selects (`fc`). Resolved against the live corpus by
+ // the caller. Empty with `filters === null` means "whole corpus".
+ channelNames: string[];
+ // Whether the link constrained channels at all (had a v1 filter block). When
+ // false, channel scope is the whole corpus regardless of `channelNames`.
+ hasChannelFilter: boolean;
+ // Vestigial `fk` subtitle-track tokens — decoded but a no-op.
+ ignoredTracks: string[];
+ // Human-facing notes (malformed qt, ignored fk, …).
+ warnings: string[];
+};
+
+// Decode a share link. Never throws for a malformed query tree (falls back to a
+// legacy/empty tree with a warning); throws only if `link` isn't a URL at all.
+export function decodeShareLink(link: string): DecodedLink {
+ const url = new URL(link);
+ const search = url.search;
+ const params = new URLSearchParams(search);
+ const warnings: string[] = [];
+
+ // ── Query: qt= tree preferred, else legacy q/re, else empty (filter-only).
+ let tree: GroupNode = emptyRoot();
+ let querySource: DecodedLink["querySource"] = "empty";
+ const qt = params.get("qt");
+ if (qt) {
+ const parsed = parseRoot(qt);
+ if (parsed) {
+ tree = parsed;
+ querySource = "qt";
+ } else {
+ warnings.push("malformed `qt` query tree — ignored");
+ }
+ }
+ if (querySource === "empty") {
+ const q = params.get("q") ?? "";
+ const mode = params.get("m") === "subs" ? "subs" : "transcripts";
+ const re = params.get("re") === "1";
+ if (q.trim() !== "") {
+ tree = rootFromLegacy(q, mode, re);
+ querySource = "legacy";
+ }
+ }
+
+ // ── Filters: only meaningful when the v1 block is present (`fv=1`).
+ const v1 = hasShareV1(search);
+ // Pass the requested fc names AS the channel universe so none are dropped —
+ // real validation against the live corpus happens at apply time.
+ const requestedChannels = params.getAll("fc");
+ const sel = parseShareV1(search, requestedChannels);
+ const ignoredTracks = [...sel.tracks];
+ if (ignoredTracks.length > 0) {
+ warnings.push(
+ `\`fk\` subtitle-track filter is vestigial in the composite share model — ` +
+ `${ignoredTracks.length} track token(s) decoded but ignored: ${ignoredTracks.join(", ")}`,
+ );
+ }
+
+ const filters: SearchFilters | null = v1
+ ? {
+ videos: sel.videos,
+ livestreams: sel.livestreams,
+ allAges: sel.allAges,
+ restricted: sel.restricted,
+ available: sel.available,
+ unlisted: sel.unlisted,
+ deleted: sel.deleted,
+ ...(sel.dateFrom ? { dateFrom: sel.dateFrom } : {}),
+ ...(sel.dateTo ? { dateTo: sel.dateTo } : {}),
+ }
+ : null;
+
+ return {
+ origin: url.origin,
+ tree,
+ querySource,
+ filters,
+ channelNames: v1 ? [...sel.selectedChannels] : [],
+ hasChannelFilter: v1,
+ ignoredTracks,
+ warnings,
+ };
+}
+
+// Structured adjustments an agent can pass to honor a natural-language edit
+// ("remove the availability filter", "only channel X", "search 'foo' instead").
+// Every field is optional; only the provided facets are changed.
+export type LinkOverrides = {
+ // Drop every decoded filter (ft/fa/fav/dates) → unfiltered.
+ clearFilters?: boolean;
+ // Reset a single filter facet to its keep-everything default.
+ clearAvailability?: boolean; // fav
+ clearType?: boolean; // ft
+ clearAge?: boolean; // fa
+ clearDates?: boolean; // fdf/fdt
+ // Widen the channel scope to the whole corpus (ignore fc).
+ clearChannels?: boolean;
+ // Replace the channel scope with these names (validated against the corpus).
+ channels?: string[];
+ // Replace/clear the upload-date bounds ("YYYYMMDD"; null clears that bound).
+ dateFrom?: string | null;
+ dateTo?: string | null;
+ // Replace the whole query with a single leaf (query + optional regex/scope).
+ query?: string;
+ regex?: boolean;
+ queryScope?: "transcripts" | "chat" | "metadata" | "description" | "tags";
+};
+
+const KEEP_ALL_FILTERS: SearchFilters = {
+ videos: true,
+ livestreams: true,
+ allAges: true,
+ restricted: true,
+ available: true,
+ unlisted: true,
+ deleted: true,
+};
+
+// Apply overrides to a decoded link, returning a new DecodedLink. Records what
+// changed as warnings so the plan can echo it.
+export function applyLinkOverrides(
+ decoded: DecodedLink,
+ overrides: LinkOverrides | undefined,
+): DecodedLink {
+ if (!overrides) return decoded;
+ const next: DecodedLink = {
+ ...decoded,
+ warnings: [...decoded.warnings],
+ channelNames: [...decoded.channelNames],
+ filters: decoded.filters ? { ...decoded.filters } : null,
+ };
+ const note = (s: string): void => {
+ next.warnings.push(`override: ${s}`);
+ };
+
+ // ── Query replacement.
+ if (typeof overrides.query === "string" && overrides.query.trim() !== "") {
+ next.tree = newGroup({
+ children: [
+ newLeaf({
+ query: overrides.query,
+ scope: overrides.queryScope ?? "transcripts",
+ useRegex: overrides.regex === true,
+ }),
+ ],
+ });
+ next.querySource = "qt";
+ note(`query replaced with "${overrides.query}"`);
+ }
+
+ // ── Channel scope.
+ if (overrides.clearChannels) {
+ next.channelNames = [];
+ next.hasChannelFilter = false;
+ note("channel scope widened to the whole corpus");
+ }
+ if (Array.isArray(overrides.channels)) {
+ next.channelNames = overrides.channels.filter((c) => c.trim() !== "");
+ next.hasChannelFilter = true;
+ note(`channel scope set to [${next.channelNames.join(", ")}]`);
+ }
+
+ // ── Filters.
+ if (overrides.clearFilters) {
+ next.filters = null;
+ note("all filters removed");
+ } else if (next.filters) {
+ const f = next.filters;
+ if (overrides.clearAvailability) {
+ f.available = true;
+ f.unlisted = true;
+ f.deleted = true;
+ note("availability filter removed");
+ }
+ if (overrides.clearType) {
+ f.videos = true;
+ f.livestreams = true;
+ note("type filter removed");
+ }
+ if (overrides.clearAge) {
+ f.allAges = true;
+ f.restricted = true;
+ note("age filter removed");
+ }
+ if (overrides.clearDates) {
+ delete f.dateFrom;
+ delete f.dateTo;
+ note("date filter removed");
+ }
+ if (overrides.dateFrom !== undefined) {
+ if (overrides.dateFrom === null) delete f.dateFrom;
+ else f.dateFrom = overrides.dateFrom;
+ note(`dateFrom set to ${overrides.dateFrom ?? "(none)"}`);
+ }
+ if (overrides.dateTo !== undefined) {
+ if (overrides.dateTo === null) delete f.dateTo;
+ else f.dateTo = overrides.dateTo;
+ note(`dateTo set to ${overrides.dateTo ?? "(none)"}`);
+ }
+ } else if (
+ overrides.dateFrom !== undefined ||
+ overrides.dateTo !== undefined
+ ) {
+ // Setting a date on an otherwise-unfiltered link creates a filter block.
+ next.filters = { ...KEEP_ALL_FILTERS };
+ if (typeof overrides.dateFrom === "string") next.filters.dateFrom = overrides.dateFrom;
+ if (typeof overrides.dateTo === "string") next.filters.dateTo = overrides.dateTo;
+ note("date filter added");
+ }
+
+ return next;
+}
+
+// ── Human-readable renderings for the preview plan ──
+
+const SCOPE_LABEL: Record<string, string> = {
+ transcripts: "transcript",
+ chat: "live-chat",
+ metadata: "title/channel",
+ description: "description",
+ tags: "tags",
+};
+
+// Render a query tree readably, e.g.
+// transcript:"coffee" AND tags:"espresso" AND NOT chat:"spam"
+export function renderQueryTree(node: QueryNode): string {
+ if (isLeaf(node)) {
+ const label = SCOPE_LABEL[node.scope] ?? node.scope;
+ const kind = node.useRegex ? "/" : '"';
+ const end = node.useRegex ? "/" : '"';
+ const body = `${label}:${kind}${node.query}${end}`;
+ return node.negate ? `NOT ${body}` : body;
+ }
+ if (!isGroup(node)) return "";
+ const parts = node.children
+ .filter(isNodeActive)
+ .map((c) => {
+ const s = renderQueryTree(c);
+ return isGroup(c) ? `(${s})` : s;
+ })
+ .filter((s) => s !== "");
+ if (parts.length === 0) return "(everything in scope)";
+ const joined = parts.join(` ${node.op} `);
+ return node.negate ? `NOT (${joined})` : joined;
+}
+
+// A compact list of the active filters for the plan, or [] when unfiltered.
+export function describeFilters(f: SearchFilters | null): string[] {
+ if (!f) return [];
+ const out: string[] = [];
+ // Type (ft): only worth noting when it excludes one side.
+ if (f.videos !== f.livestreams) {
+ out.push(`type: ${f.videos ? "videos only" : "livestreams only"}`);
+ } else if (!f.videos && !f.livestreams) {
+ out.push("type: none kept (videos and livestreams both excluded)");
+ }
+ // Audience (fa).
+ if (f.allAges !== f.restricted) {
+ out.push(`audience: ${f.allAges ? "all-ages only" : "age-restricted only"}`);
+ } else if (!f.allAges && !f.restricted) {
+ out.push("audience: none kept");
+ }
+ // Availability (fav).
+ const av: string[] = [];
+ if (f.available) av.push("available");
+ if (f.unlisted) av.push("unlisted");
+ if (f.deleted) av.push("deleted");
+ if (av.length < 3) out.push(`availability: ${av.length ? av.join(" + ") : "none"} kept`);
+ // Dates (fdf/fdt).
+ if (f.dateFrom || f.dateTo) {
+ out.push(`uploaded: ${f.dateFrom ?? "…"} → ${f.dateTo ?? "…"}`);
+ }
+ return out;
+}
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -2,9 +2,17 @@ import { readFile, readdir } from "node:fs/promises";
import path from "node:path";
import {
transcriptPageFileName,
+ subsPageFileName,
+ pageFileName,
type ChannelTranscriptsManifest,
+ type ChannelSubsManifest,
+ type Manifest,
} from "yt-dlp-transcript-common/lib/manifest";
-import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type {
+ TranscriptDetail,
+ DisplaySummary,
+} from "yt-dlp-transcript-common/lib/transcripts";
+import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs";
import {
coerceAliasConfig,
type SearchAlias,
@@ -48,6 +56,48 @@ const EMPTY_GROUPS: ChannelGroups = {
defaultGroupId: DEFAULT_GROUP_FALLBACK_ID,
};
+// Per-video availability, joined in from the summaries shards (a
+// TranscriptDetail record does not carry deleted/unlisted flags). Keyed by the
+// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript
+// page record carries — so the search engine can apply the `fav` filter.
+export type VideoAvailability = { isDeleted: boolean; isUnlisted: boolean };
+
+// Read a site's global summaries shards (summaries/manifest.json +
+// summaries/page-NNNN.json) via `readPage` and fold them into a slug →
+// availability map. Tolerant: an absent/malformed manifest yields an empty map,
+// and a page that fails to read is skipped. `readManifest`/`readPage` throw or
+// return null on absence per the source's transport.
+async function buildAvailabilityMap(
+ readManifest: () => Promise<Manifest | null>,
+ readPage: (page: number) => Promise<DisplaySummary[] | null>,
+): Promise<Map<string, VideoAvailability>> {
+ const map = new Map<string, VideoAvailability>();
+ let manifest: Manifest | null;
+ try {
+ manifest = await readManifest();
+ } catch {
+ return map;
+ }
+ if (!manifest || typeof manifest.pageCount !== "number") return map;
+ for (let page = 0; page < manifest.pageCount; page++) {
+ let records: DisplaySummary[] | null;
+ try {
+ records = await readPage(page);
+ } catch {
+ continue;
+ }
+ if (!records) continue;
+ for (const r of records) {
+ if (typeof r.slug !== "string") continue;
+ map.set(r.slug, {
+ isDeleted: r.isDeleted === true,
+ isUnlisted: r.isUnlisted === true,
+ });
+ }
+ }
+ return map;
+}
+
// Parse a summaries/manifest.json blob into channel-group defs, tolerating any
// missing/malformed shape (→ empty fallback).
function parseGroupsManifest(raw: unknown): ChannelGroups {
@@ -76,6 +126,25 @@ export interface ShardSource {
// empty fallback when absent/malformed, or in hub mode (federated per-site
// groups are a different model — deferred). Cached per source.
loadGroups(): Promise<ChannelGroups>;
+ // The public origin of the viewer that owns this source's videos, or null
+ // when there isn't one (a local dir on disk). Used to build archilyzer viewer
+ // deep links for cited moments (momentUrl). A single-site remote returns its
+ // base URL; a hub returns null because each video's origin is its member
+ // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video.
+ publicOrigin(): string | null;
+ // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null
+ // when the channel ships no subs shards. Same slugToPage/pageCount shape as
+ // the transcripts manifest. Fetched lazily — only the chat search scope needs
+ // it.
+ subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>;
+ // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record
+ // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`).
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>;
+ // A map of every video's availability (deleted/unlisted), keyed by the
+ // member-local video slug (`<channelSlug>/<id>`), built from the summaries
+ // shards. Fetched lazily and cached — only the `fav` availability filter needs
+ // it. Empty when the source ships no summaries.
+ availabilityMap(): Promise<Map<string, VideoAvailability>>;
}
// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose
@@ -104,10 +173,58 @@ export class LocalSource implements ShardSource {
readonly label: string;
private aliases?: SearchAlias[];
private groups?: ChannelGroups;
+ private availability?: Map<string, VideoAvailability>;
constructor(private dir: string) {
this.label = `local:${dir}`;
}
+ // A local dir has no public viewer origin — cited moments fall back to
+ // platform links (momentUrl).
+ publicOrigin(): string | null {
+ return null;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ try {
+ const raw = await readFile(
+ path.join(this.dir, "subs", ch.slug, "manifest.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as ChannelSubsManifest;
+ } catch {
+ return null; // channel ships no subs shards
+ }
+ }
+
+ async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ const raw = await readFile(
+ path.join(this.dir, "subs", ch.slug, subsPageFileName(page)),
+ "utf8",
+ );
+ return JSON.parse(raw) as SubsDetail[];
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ this.availability = await buildAvailabilityMap(
+ async () => {
+ const raw = await readFile(
+ path.join(this.dir, "summaries", "manifest.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as Manifest;
+ },
+ async (page) => {
+ const raw = await readFile(
+ path.join(this.dir, "summaries", pageFileName(page)),
+ "utf8",
+ );
+ return JSON.parse(raw) as DisplaySummary[];
+ },
+ );
+ return this.availability;
+ }
+
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
@@ -195,11 +312,45 @@ export class RemoteSource implements ShardSource {
private base: string;
private aliases?: SearchAlias[];
private groups?: ChannelGroups;
+ private availability?: Map<string, VideoAvailability>;
constructor(baseUrl: string) {
this.base = baseUrl.replace(/\/+$/, "");
this.label = `remote:${this.base}`;
}
+ // The deployed site origin — the archilyzer viewer that owns these videos.
+ publicOrigin(): string | null {
+ return this.base;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ try {
+ const res = await fetch(`${this.base}/subs/${ch.slug}/manifest.json`);
+ return res.ok ? ((await res.json()) as ChannelSubsManifest) : null;
+ } catch {
+ return null;
+ }
+ }
+
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ return this.getJson(`/subs/${ch.slug}/${subsPageFileName(page)}`);
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ this.availability = await buildAvailabilityMap(
+ async () => {
+ const res = await fetch(`${this.base}/summaries/manifest.json`);
+ return res.ok ? ((await res.json()) as Manifest) : null;
+ },
+ async (page) => {
+ const res = await fetch(`${this.base}/summaries/${pageFileName(page)}`);
+ return res.ok ? ((await res.json()) as DisplaySummary[]) : null;
+ },
+ );
+ return this.availability;
+ }
+
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
@@ -263,6 +414,7 @@ export class HubSource implements ShardSource {
readonly hubBase: string;
private members = new Map<string, RemoteSource>(); // siteId -> source
private aliases?: SearchAlias[];
+ private availability?: Map<string, VideoAvailability>;
// Optional subset allowlist of member siteIds. Undefined = federate every
// member; a set restricts listChannels() to those members (site discovery via
// listSites() stays unfiltered so a picker can still see all members).
@@ -359,4 +511,55 @@ export class HubSource implements ShardSource {
if (!ch.siteId) throw new Error("hub channel ref missing siteId");
return this.memberFor(ch.siteId).transcriptPage(ch, page);
}
+
+ // A hub has no single viewer origin — each video's origin is its member
+ // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers
+ // per-video. Return null so we never mint a wrong-origin viewer link.
+ publicOrigin(): string | null {
+ return null;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ if (!ch.siteId) return null;
+ try {
+ return await this.memberFor(ch.siteId).subsManifest(ch);
+ } catch {
+ return null; // member not yet registered / unreachable
+ }
+ }
+
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ if (!ch.siteId) throw new Error("hub channel ref missing siteId");
+ return this.memberFor(ch.siteId).subsPage(ch, page);
+ }
+
+ // Merge each member's availability map. Keys are member-local slugs
+ // (`<channelSlug>/<id>`) — the same slug a member's transcript page records
+ // carry — so a per-record `fav` lookup joins correctly. Built lazily/cached.
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ const merged = new Map<string, VideoAvailability>();
+ let sites: HubSite[];
+ try {
+ sites = (await this.listSites()).filter(
+ (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId),
+ );
+ } catch {
+ this.availability = merged;
+ return merged;
+ }
+ for (const site of sites) {
+ const remote = this.members.get(site.siteId) ?? new RemoteSource(site.url);
+ this.members.set(site.siteId, remote);
+ try {
+ for (const [slug, avail] of await remote.availabilityMap()) {
+ merged.set(slug, avail);
+ }
+ } catch {
+ // skip an unreachable member
+ }
+ }
+ this.availability = merged;
+ return merged;
+ }
}
diff --git a/mcp/src/sourceController.test.ts b/mcp/src/sourceController.test.ts
@@ -8,7 +8,13 @@ import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
-import type { ChannelGroups, ChannelRef, HubSite, ShardSource } from "./source";
+import type {
+ ChannelGroups,
+ ChannelRef,
+ HubSite,
+ ShardSource,
+ VideoAvailability,
+} from "./source";
import type { SourceSpec } from "./sources";
import { SourceController } from "./sourceController";
import { createServer } from "./server";
@@ -56,6 +62,18 @@ class FakeSource implements ShardSource {
async transcriptPage(): Promise<TranscriptDetail[]> {
return [];
}
+ publicOrigin(): string | null {
+ return null;
+ }
+ async subsManifest(): Promise<null> {
+ return null;
+ }
+ async subsPage(): Promise<[]> {
+ return [];
+ }
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map();
+ }
}
function labelFor(spec: SourceSpec): string {