Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fe845f38f099b0791e14ded305221a3fd6da012a
parent cb05a9059cc98e4a3f414ad350e0c5677954bba2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 21 Jul 2026 14:42:20 -0400

Merge feat/mcp-cite-link-open-link: MCP cite-and-link output, subagent-compacted sweeps, and open_link

Diffstat:
Acommon/lib/momentUrl.ts | 110+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/transcriptToMarkdown.ts | 14+++++++++++++-
Mmcp/README.md | 81+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----------
Amcp/src/momentUrl.test.ts | 127+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/search.test.ts | 318+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mmcp/src/search.ts | 452++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/server.ts | 609+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Amcp/src/shareLink.test.ts | 168+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Amcp/src/shareLink.ts | 316+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/source.ts | 205++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/sourceController.test.ts | 20+++++++++++++++++++-
11 files changed, 2339 insertions(+), 81 deletions(-)

diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts @@ -0,0 +1,110 @@ +// Build a timestamped deep link for a cited moment in a video. +// +// Two shapes, in preference order: +// +// 1. **Archilyzer viewer link** — `${siteOrigin}/?v=<slug>&t=<seconds>`. This +// is the exact scheme `common/components/urlState.ts` reads: `v` is the +// video slug (`<channelSlug>/<id>`, the value the modal compares against +// `detail.slug`) and `t` is whole seconds. Opening it lands in the archive's +// transcript modal at the cited moment. Used whenever we know the public +// origin of the viewer that owns the video (a deployed site, or a hub +// member's url). +// +// 2. **Platform link** — the video's own `webpageUrl` plus a per-platform time +// parameter, mirroring the seek patterns the in-app players use +// (`common/components/{Odysee,Rumble,Twitch}Player.tsx`). Used as a fallback +// when there is no viewer origin (e.g. an MCP pointed at a local build). +// +// Pure and dependency-light so it can be reused by the MCP server, the browser +// report-citation UI, and build tools alike. + +import { detectPlatform, type Platform } from "./platform"; + +export type MomentUrlInput = { + // Public origin of the archilyzer viewer that owns this video (a RemoteSource + // base, or a hub member's url). When present and non-empty, we build a viewer + // deep link. Null/undefined ⇒ fall back to a platform link. + siteOrigin?: string | null; + // The video's slug as the viewer's `?v=` param expects it (`channelSlug/id`). + slug?: string | null; + // The cited moment, in seconds. + seconds: number; + // Fallback building blocks when there is no viewer origin. + webpageUrl?: string | null; + platform?: Platform | null; +}; + +// Twitch watch/VOD URLs take an `XhYmZs` time token, not raw seconds. +function twitchTime(totalSeconds: number): string { + const s = Math.max(0, Math.floor(totalSeconds)); + const h = Math.floor(s / 3600); + const m = Math.floor((s % 3600) / 60); + return `${h}h${m}m${s % 60}s`; +} + +// Best-effort deep link into a platform's own watch page at `seconds`. Mirrors +// the per-platform time params the in-app players build. Platforms whose watch +// page has no reliable start param (Rumble, Kick) get the bare `webpageUrl`. +// Returns null only when there is no `webpageUrl` to work from. +export function platformMomentUrl( + webpageUrl: string | null | undefined, + platform: Platform | null | undefined, + seconds: number, +): string | null { + if (!webpageUrl) return null; + const secs = Math.max(0, Math.floor(seconds || 0)); + if (secs <= 0) return webpageUrl; + const plat = platform ?? detectPlatform(webpageUrl); + let u: URL; + try { + u = new URL(webpageUrl); + } catch { + return webpageUrl; + } + switch (plat) { + case "youtube": + u.searchParams.set("t", `${secs}s`); + return u.toString(); + case "odysee": + u.searchParams.set("t", String(secs)); + return u.toString(); + case "twitch": + u.searchParams.set("t", twitchTime(secs)); + return u.toString(); + // Rumble / Kick / unknown: the watch page has no dependable start param — + // return the plain webpage URL rather than an invalid seek. + default: + return webpageUrl; + } +} + +// The archilyzer viewer deep link, or null when `siteOrigin`/`slug` are missing +// or the origin can't be parsed. +export function viewerMomentUrl( + siteOrigin: string | null | undefined, + slug: string | null | undefined, + seconds: number, +): string | null { + if (!siteOrigin || !slug) return null; + const secs = Math.max(0, Math.floor(seconds || 0)); + let u: URL; + try { + u = new URL(siteOrigin); + } catch { + return null; + } + u.pathname = "/"; + u.search = ""; + u.hash = ""; + u.searchParams.set("v", slug); + if (secs > 0) u.searchParams.set("t", String(secs)); + return u.toString(); +} + +// The preferred moment link: viewer deep link when we have an origin+slug, +// else the platform fallback. Null when neither can be built. +export function momentUrl(input: MomentUrlInput): string | null { + const viewer = viewerMomentUrl(input.siteOrigin, input.slug, input.seconds); + if (viewer) return viewer; + return platformMomentUrl(input.webpageUrl, input.platform, input.seconds); +} diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts @@ -31,6 +31,11 @@ export type TranscriptMarkdownOptions = { // Cap the number of cue lines emitted (for fitting a context window). When // truncated, a marker line is appended. Default: no cap. maxCues?: number; + // Optional builder turning a cue's start seconds into a deep link. When it + // returns a URL, the timestamp is rendered as a Markdown link + // (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when + // `timestamps` is false or when it returns null. See `momentUrl`. + linkForCue?: (seconds: number) => string | null; }; // [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so @@ -53,6 +58,7 @@ export function transcriptToMarkdown( includeDescription = true, includeTags = false, maxCues, + linkForCue, } = options; const lines: string[] = []; @@ -97,7 +103,13 @@ export function transcriptToMarkdown( const cue = cues[i]; const text = cue.text.trim(); if (!text) continue; - lines.push(timestamps ? `[${stamp(cue.start)}] ${text}` : text); + if (!timestamps) { + lines.push(text); + continue; + } + const url = linkForCue ? linkForCue(cue.start) : null; + const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start); + lines.push(`[${label}] ${text}`); } if (limit < cues.length) { lines.push(""); diff --git a/mcp/README.md b/mcp/README.md @@ -13,14 +13,22 @@ reads the site's already-published static JSON shards (`corpus.json` + | Tool | What it does | |------|--------------| | `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. | -| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). | -| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. `channel`/`channels` hints speed the lookup. | -| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). | +| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment**. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). | +| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`), or full transcripts without a query. `channel`/`channels` hints speed the lookup. | +| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). | | `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. | +| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity). **Previews** a plan by default; **applies** it (switch source + search, linked results) on `apply:true`. Adjust in natural language via `overrides`. | | `list_sources` | Show the **active** corpus (label + kind + target) and, with a hub context, its member sites (`siteId · title · url`, marking the current subset) so you can pick one to switch to. | | `use_source` | **Switch which corpus is read**, on the fly — a hub member (`site`), a hub subset (`sites`), or an arbitrary `remote` URL / `local` dir / `hub` URL. Persists across reconnects. | | `reset_source` | Return to the source the server was started with and clear the persisted selection. | +**Clickable moment links.** Every timestamp the read tools emit is a Markdown link +to the exact second — an **archilyzer viewer** deep link (`…/?v=<slug>&t=<sec>`, +opening the transcript modal at the moment) when the source has a public origin, +otherwise the video's platform watch page with a per-platform time param +(YouTube `&t=<sec>s`, Odysee/Twitch equivalents). A `--local` source with no +origin falls back to platform links. + ### `search_transcripts` Beyond `query`, `regex`, and `limit`: @@ -66,6 +74,40 @@ video's full transcript comes back as markdown. Missing ids are reported inline. lookup when a batch spans several channels (the ids are already scoped by the search that produced them). +### `open_link` — paste a viewer share link, at full fidelity + +The archilyzer viewer's **Share** button produces a URL that encodes the whole +search: the origin, a `qt=` composite **query tree** (any mix of scopes — +transcripts, live-chat, title/channel, description, tags — combined with +AND/OR/NOT), and the filter block (`fc` channels, `ft` type, `fa` audience, +`fav` availability, `fdf`/`fdt` upload-date range). `open_link` re-runs that exact +search here — no manual source-switching or query reconstruction, nothing lost. + +It is a **preview → adjust → apply** flow: + +- **Preview** (default, `apply:false`) — decode the link and return a *plan*: the + resolved source (origin, hub vs single-site, auto-probed from `corpus.json`), + the query tree rendered readably, every active filter, the channel scope + validated against the live corpus, and anything ignored or warned (the `fk` + subtitle-track token is vestigial in the composite share model — decoded and + reported as ignored). **No source switch, no search yet.** +- **Adjust** — re-call with structured `overrides` to honor a natural-language + edit: `clear_availability` (drop the availability filter), `clear_type`, + `clear_age`, `clear_dates`, `clear_filters`, `clear_channels`, + `channels:[…]` (re-scope), `date_from`/`date_to`, or `query`/`regex`/ + `query_scope` (replace the search). The plan updates in place. +- **Apply** (`apply:true`) — switch the active source to the link's origin (a + single-site remote, or a hub when the origin federates), run the search at full + fidelity (the query tree + filters, honoring the chat and availability scopes), + and return the first page of **linked** results. The applied plan is echoed at + the top for transparency. Pageable via `limit`/`offset`. Still read-only. + +``` +open_link link="https://rekietalyzer.pages.dev/?qt=…&fv=1&fc=Rekieta%20Law&ft=v&fav=a" +open_link link="…" overrides={ "clear_availability": true } # "drop the availability filter" +open_link link="…" apply=true # switch source + search +``` + ## The `sweep` prompt A first-class slash command that turns **Claude Code itself** into the corpus @@ -74,8 +116,9 @@ browser does with a BYO AI key, but driven by your Claude **plan usage** (no API key) and with the report written to a file. Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query` -(required), `channel?`, `channels?` (comma-separated slugs/names), `group?` (a -channel group id or name), `directive?` (default *"key claims & +(required *unless* a `link` is given), `link?` (an archilyzer viewer share URL to +seed the sweep from), `channel?`, `channels?` (comma-separated slugs/names), +`group?` (a channel group id or name), `directive?` (default *"key claims & contradictions"*), `batch_size?` (default 8), `report_path?` (default `./sweep-report.md`). @@ -84,8 +127,15 @@ contradictions"*), `batch_size?` (default 8), `report_path?` (default /mcp__rekietalyzer__sweep query="k cups" group="other" # a whole group /mcp__rekietalyzer__sweep query="k cups" group="Extended Universe" # …by name /mcp__rekietalyzer__sweep query="k cups" # pick-first (see below) +/mcp__rekietalyzer__sweep link="https://…/?qt=…&fv=1&fc=…" # seed from a share link ``` +**Seed from a share link (`link=`).** Pass a viewer share URL instead of a +`query` and the sweep starts by driving `open_link`: it previews the decoded plan +(source, query tree, filters, scope, ignored bits), lets you confirm or adjust in +natural language, then applies it — switching source and enumerating the full +tree+filter match set — before batching. Everything the link encodes is honored. + **Pick-first when no scope is given.** Because a whole-corpus sweep can be a lot of work, invoking `sweep` **without** a `channel`/`channels`/`group` makes Claude FIRST call `list_channels`, present the groups and their channels, and ask you @@ -97,13 +147,24 @@ name (`group="other"` ≡ `group="Extended Universe"`). Once scoped, the prompt instructs Claude Code to: **search** (aliases auto-expand; the footer names the resolved scope) → **enumerate** the full worklist by paging with `include_snippets:false` until `has_more` is false → -**plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts` for -windowed context, cross-reference against the report so far, and upsert findings -(claims, contradictions, sources cited as *title + [mm:ss]*) into well-titled -`##` sections of the report file with its own Write/Edit tools → finish with a -short summary. The MCP stays read-only; only the report file is written, in +**plan** `ceil(N / batch_size)` batches → **per batch, map-reduce** → finish with +a short summary. The MCP stays read-only; only the report file is written, in Claude's working directory. +**Subagent-compacted batches.** Each batch is handed to a **subagent** (Claude +Code's Task tool) that calls `get_transcripts`, extracts findings for the +directive, and returns *only a compact fragment of cited, linked findings* — the +heavy transcript text lives and dies inside the subagent, so the orchestrator's +context keeps just the distilled report. The orchestrator merges each fragment +into the report and discards it; batches are independent, so several can run in +parallel. If no subagent tool is available, the batch is processed inline and the +raw text dropped after folding (the original behaviour). + +**Linked citations.** Every source is cited as a clickable +`[title @ mm:ss](<moment url>)` link — the moment URL comes straight from the +`[mm:ss](url)` links in the `get_transcripts` output, so a click seeks to the +exact second (see *Clickable moment links* above). + ## Data source (pick one) Resolved from flags or env — precedence hub > remote > local: diff --git a/mcp/src/momentUrl.test.ts b/mcp/src/momentUrl.test.ts @@ -0,0 +1,127 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + momentUrl, + viewerMomentUrl, + platformMomentUrl, +} from "yt-dlp-transcript-common/lib/momentUrl"; + +// ─── Archilyzer viewer link (preferred) ─── + +test("viewer link: origin + slug + seconds → ?v=<slug>&t=<sec>", () => { + const url = viewerMomentUrl( + "https://rekietalyzer.pages.dev", + "rekietalaw/0eYLR6CbR78", + 754, + ); + assert.ok(url); + const u = new URL(url!); + assert.equal(u.origin, "https://rekietalyzer.pages.dev"); + assert.equal(u.pathname, "/"); + // The slug round-trips through URL decoding to exactly the modal's ?v= value. + assert.equal(u.searchParams.get("v"), "rekietalaw/0eYLR6CbR78"); + assert.equal(u.searchParams.get("t"), "754"); +}); + +test("viewer link: an origin with its own path/query is normalized to /", () => { + const url = viewerMomentUrl("https://site.example/search?q=x", "ch/vid", 10); + const u = new URL(url!); + assert.equal(u.pathname, "/"); + assert.equal(u.searchParams.get("q"), null); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("viewer link: t is omitted at second 0", () => { + const url = viewerMomentUrl("https://site.example", "ch/vid", 0); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), null); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("viewer link: null without an origin or a slug", () => { + assert.equal(viewerMomentUrl(null, "ch/vid", 10), null); + assert.equal(viewerMomentUrl("https://site.example", null, 10), null); +}); + +// ─── Platform fallback ─── + +test("platform link: YouTube gets &t=<sec>s appended", () => { + const url = platformMomentUrl( + "https://www.youtube.com/watch?v=abc123", + "youtube", + 90, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("v"), "abc123"); + assert.equal(u.searchParams.get("t"), "90s"); +}); + +test("platform link: platform is detected from the URL when not given", () => { + const url = platformMomentUrl("https://www.youtube.com/watch?v=abc123", null, 5); + assert.match(url!, /t=5s/); +}); + +test("platform link: Odysee gets ?t=<sec> (raw seconds)", () => { + const url = platformMomentUrl( + "https://odysee.com/@chan/video", + "odysee", + 42, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), "42"); +}); + +test("platform link: Twitch gets ?t=<XhYmZs>", () => { + const url = platformMomentUrl( + "https://www.twitch.tv/videos/12345", + "twitch", + 3661, + ); + const u = new URL(url!); + assert.equal(u.searchParams.get("t"), "1h1m1s"); +}); + +test("platform link: Rumble has no reliable start param → bare webpage url", () => { + const url = platformMomentUrl("https://rumble.com/v123-title.html", "rumble", 60); + assert.equal(url, "https://rumble.com/v123-title.html"); +}); + +test("platform link: second 0 → bare webpage url; no webpage url → null", () => { + assert.equal( + platformMomentUrl("https://www.youtube.com/watch?v=abc", "youtube", 0), + "https://www.youtube.com/watch?v=abc", + ); + assert.equal(platformMomentUrl(null, "youtube", 30), null); +}); + +// ─── momentUrl: prefers the viewer link, falls back to the platform link ─── + +test("momentUrl: viewer link wins when an origin is present", () => { + const url = momentUrl({ + siteOrigin: "https://site.example", + slug: "ch/vid", + seconds: 12, + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + const u = new URL(url!); + assert.equal(u.origin, "https://site.example"); + assert.equal(u.searchParams.get("v"), "ch/vid"); +}); + +test("momentUrl: falls back to the platform link with no origin (local source)", () => { + const url = momentUrl({ + siteOrigin: null, + slug: "ch/vid", + seconds: 12, + webpageUrl: "https://www.youtube.com/watch?v=vid", + platform: "youtube", + }); + assert.match(url!, /youtube\.com/); + assert.match(url!, /t=12s/); +}); + +test("momentUrl: null when neither a viewer nor a platform link can be built", () => { + const url = momentUrl({ siteOrigin: null, slug: "ch/vid", seconds: 12 }); + assert.equal(url, null); +}); diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts @@ -2,29 +2,60 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; -import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; +import type { + ChannelTranscriptsManifest, + ChannelSubsManifest, +} from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups"; -import type { ChannelGroups, ChannelRef, ShardSource } from "./source"; +import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; +import type { + ChannelGroups, + ChannelRef, + ShardSource, + VideoAvailability, +} from "./source"; import { searchTranscripts, getWindowedTranscript, buildMatcher, resolveSelectedChannels, + runSearchSpec, + type SearchFilters, } from "./search"; import { createServer } from "./server"; +// Every filter facet kept (the "no-op" filter) — override fields per test. +const KEEP_ALL: SearchFilters = { + videos: true, + livestreams: true, + allAges: true, + restricted: true, + available: true, + unlisted: true, + deleted: true, +}; + // ─── A tiny in-memory ShardSource for the tests ─── function cues(...pairs: [number, string][]): Cue[] { return pairs.map(([start, text]) => ({ start, end: start + 3, text })); } -function vid(id: string, title: string, channelSlug: string, cs: Cue[]): TranscriptDetail { +function vid( + id: string, + title: string, + channelSlug: string, + cs: Cue[], + extra: Partial<TranscriptDetail> = {}, +): TranscriptDetail { return { - slug: id, + // Real records carry a channel-prefixed slug (`<channelSlug>/<id>`); mirror + // that so viewer-link + availability joins are exercised. + slug: `${channelSlug}/${id}`, id, channelSlug, title, @@ -38,6 +69,7 @@ function vid(id: string, title: string, channelSlug: string, cs: Cue[]): Transcr platform: "youtube", webpageUrl: `https://example.test/${id}`, cues: cs, + ...extra, }; } @@ -51,16 +83,40 @@ const K_CUPS_ALIAS: SearchAlias = { }; // Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup". +// Extra metadata on a1/a3/a4/a5 exercises the description/tags scopes and the +// type/age/date filters without changing what matches the "coffee" cue search. const CHAN_A: TranscriptDetail[] = [ - vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"])), + vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), { + description: "a pour over brewing guide", + tags: ["espresso", "beans"], + }), vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])), - vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"])), - vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"])), - vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"])), + vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), { + isLivestream: true, + }), + vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), { + ageRestricted: true, + }), + vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"]), { + uploadDate: "20250601", + }), vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])), vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])), ]; +// Live-chat cues, keyed by video slug — served via the subs shard methods so a +// `chat`-scope leaf has something to match. Only b1 has a chat track. +const CHAT: Record<string, Cue[]> = { + "chan-b/b1": cues([12, "@fan: banana bread anyone"], [15, "@mod: stay on topic"]), +}; + +// Availability overrides, keyed by video slug (defaults: available). a1 deleted, +// a2 unlisted — the rest available. +const AVAILABILITY: Record<string, VideoAvailability> = { + "chan-a/a1": { isDeleted: true, isUnlisted: false }, + "chan-a/a2": { isDeleted: false, isUnlisted: true }, +}; + // Channel B: one long transcript for windowing (a single match at 100s). const CHAN_B: TranscriptDetail[] = [ vid( @@ -136,6 +192,52 @@ class StubSource implements ShardSource { async transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> { return this.pages(ch)[page] ?? []; } + + publicOrigin(): string | null { + return null; // a stub has no viewer origin + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + // Only chan-b ships a (single-page) subs shard, with b1 on page 0. + if (ch.slug !== "chan-b") return null; + return { + version: 1, + channelSlug: ch.slug, + pageCount: 1, + maxPageBytes: 0, + generatedAt: "", + tracks: ["live_chat"], + slugToPage: { b1: 0 }, + }; + } + + async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + if (ch.slug !== "chan-b" || page !== 0) return []; + const slug = "chan-b/b1"; + return [ + { + slug, + id: "b1", + channelSlug: "chan-b", + title: "Long one", + uploadDate: "20240101", + date: "2024-01-01", + duration: "5:00", + channel: "Channel B", + isLivestream: false, + ageRestricted: false, + isDeleted: false, + isUnlisted: false, + platform: "youtube", + webpageUrl: "https://example.test/b1", + tracks: { live_chat: CHAT[slug] }, + }, + ]; + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + return new Map(Object.entries(AVAILABILITY)); + } } // ─── (a) Paging: offset / limit / total / hasMore ─── @@ -302,6 +404,173 @@ test("getWindowedTranscript: no matches yields no lines", async () => { assert.equal(lines.length, 0); }); +// ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ─── + +async function allChannels(src: StubSource): Promise<ChannelRef[]> { + return (await resolveSelectedChannels(src, {})).channels; +} + +function leafTree( + query: string, + scope: "transcripts" | "chat" | "metadata" | "description" | "tags", + extra: Record<string, unknown> = {}, +) { + return newGroup({ children: [newLeaf({ query, scope, ...extra })] }); +} + +async function specIds( + src: StubSource, + tree: ReturnType<typeof newGroup>, + spec: { filters?: SearchFilters | null; aliases?: SearchAlias[] } = {}, +): Promise<string[]> { + const res = await runSearchSpec(src, await allChannels(src), { tree, ...spec }, { limit: 50 }); + return res.hits.map((h) => h.videoId).sort(); +} + +test("spec transcripts leaf: matches cue text (same set as the plain scanner)", async () => { + const src = new StubSource(); + assert.deepEqual(await specIds(src, leafTree("coffee", "transcripts")), [ + "a1", "a2", "a3", "a4", "a5", "b1", + ]); +}); + +test("spec metadata leaf: matches title/channel, not cues", async () => { + const src = new StubSource(); + // "kcup" is in a6's title but no cue (cue says "k cups" with a space). + assert.deepEqual(await specIds(src, leafTree("kcup", "metadata")), ["a6"]); +}); + +test("spec description leaf: matches the description only", async () => { + const src = new StubSource(); + // "brewing" is only in a1's description, in no cue. + assert.deepEqual(await specIds(src, leafTree("brewing", "description")), ["a1"]); +}); + +test("spec tags leaf: matches the joined tags", async () => { + const src = new StubSource(); + assert.deepEqual(await specIds(src, leafTree("espresso", "tags")), ["a1"]); +}); + +test("spec chat leaf: matches live_chat cues fetched from subs shards", async () => { + const src = new StubSource(); + // "banana" only appears in b1's live chat, in no transcript cue. + const res = await runSearchSpec(src, await allChannels(src), { + tree: leafTree("banana", "chat"), + }, { limit: 50 }); + assert.deepEqual(res.hits.map((h) => h.videoId), ["b1"]); + assert.equal(res.hits[0].snippets[0].scope, "chat"); + assert.equal(res.hits[0].snippets[0].track, "live_chat"); +}); + +test("spec AND: transcripts AND tags narrows to the intersection", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags" }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a1"]); +}); + +test("spec OR: description OR chat unions the two", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "OR", + children: [ + newLeaf({ query: "brewing", scope: "description" }), + newLeaf({ query: "banana", scope: "chat" }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a1", "b1"]); +}); + +test("spec negate: coffee AND NOT (espresso tag) drops a1", async () => { + const src = new StubSource(); + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags", negate: true }), + ], + }); + assert.deepEqual(await specIds(src, tree), ["a2", "a3", "a4", "a5", "b1"]); +}); + +test("spec filter ft: livestreams:false drops the livestream (a3)", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, livestreams: false }, + }); + assert.deepEqual(ids, ["a1", "a2", "a4", "a5", "b1"]); +}); + +test("spec filter fa: keep only age-restricted → a4", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, allAges: false }, + }); + assert.deepEqual(ids, ["a4"]); +}); + +test("spec filter fav: available-only drops deleted a1 + unlisted a2", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, unlisted: false, deleted: false }, + }); + assert.deepEqual(ids, ["a3", "a4", "a5", "b1"]); +}); + +test("spec filter fav: deleted+unlisted-only keeps a1 + a2", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, available: false }, + }); + assert.deepEqual(ids, ["a1", "a2"]); +}); + +test("spec filter dates: fdf bound keeps only the later upload (a5)", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, dateFrom: "20250101" }, + }); + assert.deepEqual(ids, ["a5"]); +}); + +test("spec paging/total: stable total with an offset/limit slice", async () => { + const src = new StubSource(); + const chans = await allChannels(src); + const tree = leafTree("coffee", "transcripts"); + const page = await runSearchSpec(src, chans, { tree }, { limit: 2, offset: 2 }); + assert.equal(page.total, 6); + assert.deepEqual(page.hits.map((h) => h.videoId), ["a3", "a4"]); + assert.equal(page.hasMore, true); +}); + +test("spec aliases: a transcripts leaf expands via curated aliases", async () => { + const src = new StubSource(); + const res = await runSearchSpec( + src, + await allChannels(src), + { tree: leafTree("k cups", "transcripts"), aliases: await src.loadAliases() }, + { limit: 50 }, + ); + assert.deepEqual(res.hits.map((h) => h.videoId).sort(), ["a6", "a7"]); + assert.equal(res.firedAliases.length, 1); + assert.equal(res.firedAliases[0].id, "k-cups"); +}); + +test("spec hits carry slug + platform for moment links", async () => { + const src = new StubSource(); + const res = await runSearchSpec(src, await allChannels(src), { + tree: leafTree("coffee", "transcripts"), + }, { limit: 1 }); + assert.equal(res.hits[0].slug, "chan-a/a1"); + assert.equal(res.hits[0].platform, "youtube"); + assert.ok(res.hits[0].snippets[0].seconds >= 0); +}); + // ─── End-to-end through the MCP server (tools + prompts + missing ids) ─── async function connectClient(source: ShardSource): Promise<Client> { @@ -410,6 +679,7 @@ test("server: the sweep prompt lists with its arguments and renders the query", "channels", "directive", "group", + "link", "query", "report_path", ]); @@ -449,3 +719,35 @@ test("server: the sweep prompt with no scope instructs a group/channel pick firs assert.match(text, /confirm \*\*all\*\*/); await client.close(); }); + +test("server: the sweep prompt mandates linked citations and subagent batches", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { query: "k cups", channel: "chan-a" }, + }); + const text = (got.messages[0].content as { text: string }).text; + // Feature 1: linked citation form (not the old "title + [mm:ss]"). + assert.match(text, /\[title @ mm:ss\]\(<moment url>\)/); + assert.ok(!/title \+ \[mm:ss\]/.test(text), "old bare citation form is gone"); + // Feature 2: subagent map-reduce + inline fallback. + assert.match(text, /Spawn a subagent/); + assert.match(text, /Task tool/); + assert.match(text, /Fallback/); + await client.close(); +}); + +test("server: the sweep prompt accepts a link= seed and drives open_link", async () => { + const client = await connectClient(new StubSource()); + const got = await client.getPrompt({ + name: "sweep", + arguments: { link: "https://site.example/?q=coffee" }, + }); + const text = (got.messages[0].content as { text: string }).text; + assert.match(text, /open_link/); + assert.match(text, /apply:true/); + assert.match(text, /https:\/\/site\.example\/\?q=coffee/); + // A link-only sweep needs no query argument. + assert.match(text, /share link/); + await client.close(); +}); diff --git a/mcp/src/search.ts b/mcp/src/search.ts @@ -1,5 +1,16 @@ import { formatDuration } from "yt-dlp-transcript-common/lib/format"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; +import type { Platform } from "yt-dlp-transcript-common/lib/platform"; +import { + forEachLeaf, + isLeaf, + isLeafActive, + isNodeActive, + type GroupNode, + type LayerScope, + type QueryNode, +} from "yt-dlp-transcript-common/lib/searchQuery"; import { matchAliases, type SearchAlias, @@ -14,7 +25,7 @@ import { resolveChannelGroupId, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; -import type { ChannelRef, ShardSource } from "./source"; +import type { ChannelRef, ShardSource, VideoAvailability } from "./source"; export type Snippet = { clock: string; seconds: number; text: string }; @@ -126,9 +137,17 @@ export async function resolveSelectedChannels( export type SearchHit = { videoId: string; + // The video's slug (`<channelSlug>/<id>`) — the archilyzer viewer's `?v=` + // value, used to build a moment deep link (momentUrl). + slug: string; channelSlug: string; channelName: string; siteTitle?: string; + // The public origin of the site that owns this video (a member url in hub + // mode, the site base for a single remote). Used with `slug` for a viewer + // link; absent for a local source (→ platform-link fallback). + siteUrl?: string; + platform?: Platform; title: string; uploadDate: string; webpageUrl?: string; @@ -325,9 +344,12 @@ export async function searchTranscripts( if (matches === 0 && !titleHit) continue; all.push({ videoId: rec.id, + slug: rec.slug, channelSlug: ch.slug, channelName: ch.name, ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}), + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(rec.platform ? { platform: rec.platform } : {}), title: rec.title, uploadDate: rec.uploadDate, webpageUrl: rec.webpageUrl, @@ -376,6 +398,9 @@ export function getWindowedTranscript( after?: number; maxCues?: number; timestamps?: boolean; + // Optional builder turning a line's start seconds into a moment deep link; + // when it returns a URL the timestamp is rendered as a Markdown link. + link?: (seconds: number) => string | null; } = {}, ): { lines: string[]; matchCount: number } { const cues = record.cues ?? []; @@ -394,9 +419,12 @@ export function getWindowedTranscript( ); merged = mergeSnippets(merged, win, WINDOW_LINE_CAP); } - const lines = merged.map((s) => - timestamps ? `[${s.clock}] ${s.text}` : s.text, - ); + const lines = merged.map((s) => { + if (!timestamps) return s.text; + const url = opts.link ? opts.link(s.seconds) : null; + const stamp = url ? `[${s.clock}](${url})` : s.clock; + return `[${stamp}] ${s.text}`; + }); return { lines, matchCount }; } @@ -445,3 +473,419 @@ export async function findVideo( } return null; } + +// ─── Full-fidelity spec engine: query tree + filter predicate ─── +// +// `runSearchSpec` is the per-record evaluator that backs the `open_link` / +// `sweep link=` flows: it honors everything an archilyzer share link can carry +// (a composite `qt=` query tree over any mix of scopes, plus the `fc/ft/fa/fav/ +// fdf/fdt` filters), while keeping the same paging / total / truncation +// contract as `searchTranscripts`. The plain `search_transcripts` path stays on +// the simpler `searchTranscripts` scanner above (a trivial single-leaf query), +// so today's callers/tests are unaffected. +// +// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate +// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean +// instead of over slug sets — and the per-scope text extraction of +// searchPipeline.ts. + +// The positive share-filter selection (parseShareV1), as a predicate input. +export type SearchFilters = { + // ft — video / livestream types kept. + videos: boolean; + livestreams: boolean; + // fa — all-ages / age-restricted kept. + allAges: boolean; + restricted: boolean; + // fav — availability three-way (available / unlisted / deleted) kept. + available: boolean; + unlisted: boolean; + deleted: boolean; + // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD". + dateFrom?: string; + dateTo?: string; +}; + +export type SearchSpec = { + // The root of the composite query (a `qt=` tree, or a synthesized single leaf + // for a legacy `q`/`re` link). + tree: GroupNode; + // The decoded share filters, or null for an unfiltered search. + filters?: SearchFilters | null; + // Curated aliases to expand transcripts-scope leaves through (as in the plain + // path). Empty/omitted → no expansion. + aliases?: SearchAlias[]; + // Expand transcripts leaves via aliases (default true; skipped per-leaf for a + // regex leaf). + useAliases?: boolean; +}; + +// A snippet tagged with the scope it came from, carrying the seconds needed to +// build a moment link. `seconds` is 0 for the non-timed scopes (metadata / +// description / tags). +export type ScopedSnippet = { + scope: LayerScope; + track?: string; + clock: string; + seconds: number; + text: string; +}; + +export type SpecHit = { + videoId: string; + slug: string; + channelSlug: string; + channelName: string; + siteTitle?: string; + siteUrl?: string; + platform?: Platform; + title: string; + uploadDate: string; + webpageUrl?: string; + matches: number; + snippets: ScopedSnippet[]; +}; + +export type SpecResult = { + hits: SpecHit[]; + total: number; + offset: number; + limit: number; + hasMore: boolean; + firedAliases: SearchAlias[]; + scanned: { channels: number; pages: number }; + truncated: boolean; +}; + +type LeafMatcher = { scope: LayerScope; test: Matcher }; + +// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless +// they're regex); every other scope matches plain-substring / regex only. Fired +// aliases are unioned for the caller to report. +function buildLeafMatchers( + root: QueryNode, + aliases: SearchAlias[], + useAliases: boolean, +): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } { + const matchers = new Map<string, LeafMatcher>(); + const fired: SearchAlias[] = []; + forEachLeaf(root, (leaf) => { + if (!isLeafActive(leaf)) return; + const aliasAware = + leaf.scope === "transcripts" && !leaf.useRegex && useAliases; + const built = buildMatcher({ + query: leaf.query, + regex: leaf.useRegex, + useAliases: aliasAware, + aliases: aliasAware ? aliases : [], + }); + matchers.set(leaf.id, { scope: leaf.scope, test: built.match }); + for (const a of built.firedAliases) { + if (!fired.some((x) => x.id === a.id)) fired.push(a); + } + }); + return { matchers, fired }; +} + +// Per-record context the tree evaluates against. +type RecordCtx = { + title: string; + channel: string; + description: string; + tags: string; + cues: Cue[]; + chatCues: Cue[]; + snippetsPerVideo: number; + includeSnippets: boolean; +}; + +type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] }; + +function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome { + if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] }; + const hits: ScopedSnippet[] = []; + const push = (s: ScopedSnippet): void => { + if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s); + }; + let count = 0; + switch (m.scope) { + case "transcripts": + for (const cue of ctx.cues) { + if (!m.test(cue.text)) continue; + count++; + push({ + scope: "transcripts", + clock: clock(cue.start), + seconds: cue.start, + text: truncate(cue.text), + }); + } + break; + case "chat": + for (const cue of ctx.chatCues) { + if (!m.test(cue.text)) continue; + count++; + push({ + scope: "chat", + track: "live_chat", + clock: clock(cue.start), + seconds: cue.start, + text: truncate(cue.text), + }); + } + break; + case "metadata": { + const titleHit = m.test(ctx.title); + const channelHit = m.test(ctx.channel); + if (titleHit || channelHit) { + count++; + if (titleHit) { + push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) }); + } else { + push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` }); + } + } + break; + } + case "description": + if (ctx.description && m.test(ctx.description)) { + count++; + push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) }); + } + break; + case "tags": + if (ctx.tags && m.test(ctx.tags)) { + count++; + push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) }); + } + break; + } + return { matched: count > 0, count, hits }; +} + +type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] }; + +// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate, +// per record. A negated node contributes no hits (like the browser's `diff`). +// An inactive (empty) subtree is identity (matches, no hits). +function evalNode( + node: QueryNode, + matchers: Map<string, LeafMatcher>, + ctx: RecordCtx, +): NodeOutcome { + if (!isNodeActive(node)) return { match: true, count: 0, hits: [] }; + if (isLeaf(node)) { + const m = matchers.get(node.id); + if (!m) return { match: true, count: 0, hits: [] }; + const r = evalLeaf(node, m, ctx); + if (node.negate) return { match: !r.matched, count: 0, hits: [] }; + return { + match: r.matched, + count: node.contributeHits ? r.count : 0, + hits: node.contributeHits ? r.hits : [], + }; + } + const active = node.children.filter(isNodeActive); + if (active.length === 0) return { match: true, count: 0, hits: [] }; + if (node.op === "AND") { + let allMatch = true; + let count = 0; + const hits: ScopedSnippet[] = []; + for (const c of active) { + const r = evalNode(c, matchers, ctx); + if (!r.match) { + allMatch = false; + break; + } + count += r.count; + hits.push(...r.hits); + } + const match = node.negate ? !allMatch : allMatch; + return match && !node.negate + ? { match, count, hits } + : { match, count: 0, hits: [] }; + } + // OR + let any = false; + let count = 0; + const hits: ScopedSnippet[] = []; + for (const c of active) { + const r = evalNode(c, matchers, ctx); + if (r.match) { + any = true; + count += r.count; + hits.push(...r.hits); + } + } + const match = node.negate ? !any : any; + return match && !node.negate + ? { match, count, hits } + : { match, count: 0, hits: [] }; +} + +// The share-filter predicate over a transcript record + its availability. +// Mirrors SearchSessionContext.tsx `passesFilter` exactly. +function passesFilters( + rec: TranscriptDetail, + f: SearchFilters, + avail: VideoAvailability | undefined, +): boolean { + // ft — type + if (rec.isLivestream ? !f.livestreams : !f.videos) return false; + // fa — audience + if (rec.ageRestricted ? !f.restricted : !f.allAges) return false; + // fav — availability three-way + if (avail?.isDeleted) { + if (!f.deleted) return false; + } else if (avail?.isUnlisted) { + if (!f.unlisted) return false; + } else if (!f.available) { + return false; + } + // fdf / fdt — upload-date range (lexicographic on YYYYMMDD) + if (f.dateFrom && rec.uploadDate < f.dateFrom) return false; + if (f.dateTo && rec.uploadDate > f.dateTo) return false; + return true; +} + +// True when the fav filter could exclude something (so availability must be +// fetched). If every availability bucket is kept, there's nothing to look up. +function needsAvailability(f: SearchFilters | null | undefined): boolean { + return !!f && (!f.available || !f.unlisted || !f.deleted); +} + +// A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a +// chat-scope leaf only pulls the subs shards it actually touches. +function makeChatFetcher(source: ShardSource) { + const manifests = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsManifest"]>>>>(); + const pages = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsPage"]>>>>(); + return async function chatCues(ch: ChannelRef, rec: TranscriptDetail): Promise<Cue[]> { + let mp = manifests.get(ch.key); + if (!mp) { + mp = source.subsManifest(ch).catch(() => null); + manifests.set(ch.key, mp); + } + const manifest = await mp; + if (!manifest) return []; + const page = manifest.slugToPage[rec.id]; + if (page === undefined) return []; + const pk = `${ch.key}:${page}`; + let pp = pages.get(pk); + if (!pp) { + pp = source.subsPage(ch, page).catch(() => []); + pages.set(pk, pp); + } + const records = await pp; + const found = records.find((r) => r.slug === rec.slug || r.id === rec.id); + const lc = found?.tracks?.live_chat; + return Array.isArray(lc) ? lc : []; + }; +} + +// Run a full-fidelity spec over the given (already-resolved) channel set. Same +// paging/total/truncation semantics as searchTranscripts. +export async function runSearchSpec( + source: ShardSource, + channels: ChannelRef[], + spec: SearchSpec, + opts: { + offset?: number; + limit?: number; + includeSnippets?: boolean; + maxPages?: number; + snippetsPerVideo?: number; + } = {}, +): Promise<SpecResult> { + const limit = opts.limit ?? 20; + const offset = Math.max(0, opts.offset ?? 0); + const maxPages = opts.maxPages ?? MAX_PAGES; + const includeSnippets = opts.includeSnippets !== false; + const snippetsPerVideo = opts.snippetsPerVideo ?? 4; + const filters = spec.filters ?? null; + + const { matchers, fired } = buildLeafMatchers( + spec.tree, + spec.aliases ?? [], + spec.useAliases !== false, + ); + const wantsChat = [...matchers.values()].some((m) => m.scope === "chat"); + const chatCuesFor = wantsChat ? makeChatFetcher(source) : null; + const availability = needsAvailability(filters) + ? await source.availabilityMap() + : null; + + const all: SpecHit[] = []; + let pagesScanned = 0; + let channelsScanned = 0; + let truncated = false; + + outer: for (const ch of channels) { + let manifest; + try { + manifest = await source.transcriptsManifest(ch); + } catch { + continue; + } + channelsScanned++; + for (let page = 0; page < manifest.pageCount; page++) { + if (pagesScanned >= maxPages) { + truncated = true; + break outer; + } + let records: TranscriptDetail[]; + try { + records = await source.transcriptPage(ch, page); + } catch { + continue; + } + pagesScanned++; + for (const rec of records) { + if (filters && !passesFilters(rec, filters, availability?.get(rec.slug))) { + continue; + } + const ctx: RecordCtx = { + title: rec.title ?? "", + channel: rec.channel ?? ch.name, + description: rec.description ?? "", + tags: (rec.tags ?? []).join(", "), + cues: rec.cues ?? [], + chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [], + snippetsPerVideo, + includeSnippets, + }; + const r = evalNode(spec.tree, matchers, ctx); + if (!r.match) continue; + all.push({ + videoId: rec.id, + slug: rec.slug, + channelSlug: ch.slug, + channelName: ch.name, + ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}), + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(rec.platform ? { platform: rec.platform } : {}), + title: rec.title, + uploadDate: rec.uploadDate, + webpageUrl: rec.webpageUrl, + matches: r.count || 1, + snippets: r.hits, + }); + if (all.length >= HARD_VIDEO_CAP) { + truncated = true; + break outer; + } + } + } + } + + const total = all.length; + return { + hits: all.slice(offset, offset + limit), + total, + offset, + limit, + hasMore: offset + limit < total, + firedAliases: fired, + scanned: { channels: channelsScanned, pages: pagesScanned }, + truncated, + }; +} diff --git a/mcp/src/server.ts b/mcp/src/server.ts @@ -7,6 +7,8 @@ import { } from "@modelcontextprotocol/sdk/types.js"; import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown"; import { formatDate } from "yt-dlp-transcript-common/lib/format"; +import { momentUrl } from "yt-dlp-transcript-common/lib/momentUrl"; +import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { sortGroups, @@ -14,7 +16,13 @@ import { FALLBACK_GROUP, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; -import { HubSource, type ChannelRef, type HubSite, type ShardSource } from "./source"; +import { + HubSource, + RemoteSource, + type ChannelRef, + type HubSite, + type ShardSource, +} from "./source"; import type { SourceSpec } from "./sources"; import type { SourceController } from "./sourceController"; import { @@ -22,8 +30,56 @@ import { findVideo, buildMatcher, getWindowedTranscript, + runSearchSpec, type SearchResult, + type SpecHit, + type ScopedSnippet, } from "./search"; +import { + decodeShareLink, + applyLinkOverrides, + renderQueryTree, + describeFilters, + type DecodedLink, + type LinkOverrides, +} from "./shareLink"; + +// The building blocks momentUrl needs from a hit (viewer origin comes from the +// per-video siteUrl, else the source's single public origin). +type LinkableHit = { + slug: string; + siteUrl?: string; + webpageUrl?: string; + platform?: Platform; +}; + +// A moment deep link for one cited second of a hit, or null when neither a +// viewer nor a platform link can be built (e.g. a local source + no webpage url). +function momentLinkFor( + source: ShardSource, + h: LinkableHit, + seconds: number, +): string | null { + return momentUrl({ + siteOrigin: h.siteUrl ?? source.publicOrigin(), + slug: h.slug, + seconds, + webpageUrl: h.webpageUrl, + platform: h.platform, + }); +} + +// A `[m:ss]` timestamp rendered as a Markdown link when we can build one, else +// bare. Used inside a `- [ … ] text` snippet line → `- [[m:ss](url)] text`. +function stampMarkup( + source: ShardSource, + h: LinkableHit, + clock: string, + seconds: number, +): string { + const url = momentLinkFor(source, h, seconds); + return url ? `[${clock}](${url})` : clock; +} type ToolResult = { content: { type: "text"; text: string }[]; @@ -291,6 +347,103 @@ const TOOLS = [ "--local flag or env) and clear the persisted selection.", inputSchema: { type: "object", properties: {}, additionalProperties: false }, }, + { + name: "open_link", + description: + "Paste an archilyzer viewer **share link** (origin + a `qt=` query tree + " + + "`fc/ft/fa/fav/fdf/fdt` filters) to re-run that exact search here — at full " + + "fidelity, with clickable timestamped result links. Two-phase: by default " + + "(apply:false) it PREVIEWS — decodes the link and returns a plan (resolved " + + "source, the query tree rendered readably, every active filter, the channel " + + "scope validated against the live corpus, and anything ignored, e.g. the " + + "vestigial `fk` tracks) WITHOUT switching source or searching. Adjust in " + + "natural language by re-calling with `overrides` (e.g. clear_availability " + + "to drop the availability filter, channels to re-scope, query to replace " + + "the search). When the plan looks right, call again with apply:true: it " + + "switches the active source to the link's origin (hub or single-site, " + + "auto-detected) and returns the first page of linked results. Read-only.", + inputSchema: { + type: "object", + properties: { + link: { + type: "string", + description: "The archilyzer viewer share URL to decode and run.", + }, + apply: { + type: "boolean", + description: + "false (default) previews the plan only; true switches source and " + + "runs the search.", + }, + overrides: { + type: "object", + description: + "Structured edits to the decoded search, so a natural-language " + + "adjustment can be honored by re-calling with the changed facet.", + properties: { + clear_filters: { type: "boolean", description: "Drop all filters." }, + clear_availability: { + type: "boolean", + description: "Remove the availability (deleted/unlisted) filter.", + }, + clear_type: { + type: "boolean", + description: "Remove the video/livestream type filter.", + }, + clear_age: { + type: "boolean", + description: "Remove the all-ages/age-restricted filter.", + }, + clear_dates: { + type: "boolean", + description: "Remove the upload-date range filter.", + }, + clear_channels: { + type: "boolean", + description: "Widen the channel scope to the whole corpus.", + }, + channels: { + type: "array", + items: { type: "string" }, + description: "Replace the channel scope with these channel names.", + }, + date_from: { + type: "string", + description: "Set the inclusive lower upload-date bound (YYYYMMDD).", + }, + date_to: { + type: "string", + description: "Set the inclusive upper upload-date bound (YYYYMMDD).", + }, + query: { + type: "string", + description: "Replace the whole query with this single term.", + }, + regex: { + type: "boolean", + description: "Treat the override query as a regex.", + }, + query_scope: { + type: "string", + enum: ["transcripts", "chat", "metadata", "description", "tags"], + description: "Scope for the override query (default transcripts).", + }, + }, + additionalProperties: false, + }, + limit: { + type: "number", + description: "Results per page when apply:true (default 20).", + }, + offset: { + type: "number", + description: "Skip this many matches when apply:true (default 0).", + }, + }, + required: ["link"], + additionalProperties: false, + }, + }, ]; // The slice of SourceController the server needs. A bare ShardSource is wrapped @@ -378,6 +531,8 @@ export function createServer(sourceOrController: ShardSource | SourceController) return await handleUseSource(controller, args); case "reset_source": return await handleResetSource(controller); + case "open_link": + return await handleOpenLink(controller, args); default: return errorText(`unknown tool: ${name}`); } @@ -508,7 +663,9 @@ async function handleSearch( (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); - const snips = h.snippets.map((s) => ` - [${s.clock}] ${s.text}`).join("\n"); + const snips = h.snippets + .map((s) => ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`) + .join("\n"); return snips ? `${head}\n${snips}` : head; }); return text( @@ -604,6 +761,14 @@ async function handleGetTranscripts( continue; } const { record, ch } = found; + const link: LinkableHit = { + slug: record.slug, + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), + ...(record.platform ? { platform: record.platform } : {}), + }; + const linkForSeconds = (seconds: number): string | null => + momentLinkFor(source, link, seconds); const head = `## ${record.title || id}\n` + `- video_id: ${id} | channel: ${ch.name}` + @@ -616,6 +781,7 @@ async function handleGetTranscripts( before, after, timestamps, + link: linkForSeconds, }); const body = matchCount === 0 @@ -626,6 +792,7 @@ async function handleGetTranscripts( const md = transcriptToMarkdown(record, { timestamps, includeTags: true, + linkForCue: linkForSeconds, }); blocks.push(md.trim()); } @@ -658,9 +825,17 @@ async function handleGetTranscript( typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`video not found: ${videoId}`); - const md = transcriptToMarkdown(found.record, { + const { record, ch } = found; + const link: LinkableHit = { + slug: record.slug, + ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), + ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), + ...(record.platform ? { platform: record.platform } : {}), + }; + const md = transcriptToMarkdown(record, { timestamps: args.timestamps !== false, includeTags: true, + linkForCue: (seconds) => momentLinkFor(source, link, seconds), }); return text(md); } @@ -896,6 +1071,271 @@ async function handleResetSource( ); } +// ─── open_link: paste a share link → preview, adjust, apply ─── + +// Map the snake_cased tool `overrides` object onto the LinkOverrides shape. +function parseOverrides(raw: unknown): LinkOverrides | undefined { + if (!raw || typeof raw !== "object") return undefined; + const o = raw as Record<string, unknown>; + const bool = (k: string): boolean | undefined => + o[k] === true ? true : undefined; + const str = (k: string): string | undefined => + typeof o[k] === "string" ? (o[k] as string) : undefined; + const scope = str("query_scope"); + return { + clearFilters: bool("clear_filters"), + clearAvailability: bool("clear_availability"), + clearType: bool("clear_type"), + clearAge: bool("clear_age"), + clearDates: bool("clear_dates"), + clearChannels: bool("clear_channels"), + channels: strArray(o.channels), + dateFrom: str("date_from"), + dateTo: str("date_to"), + query: str("query"), + regex: o.regex === true ? true : undefined, + queryScope: + scope === "transcripts" || + scope === "chat" || + scope === "metadata" || + scope === "description" || + scope === "tags" + ? scope + : undefined, + }; +} + +type OriginProbe = { kind: "hub" | "remote"; siteCount?: number; channelCount?: number }; + +// Probe an origin's corpus.json to classify it as a federated hub (has a +// `sites[]` array) or a single site (has `channels[]`). Transient — no source is +// committed. Defaults to single-site when the probe can't be read. +async function probeOrigin(origin: string): Promise<OriginProbe> { + try { + const res = await fetch(`${origin}/corpus.json`); + if (res.ok) { + const j = (await res.json()) as { + sites?: unknown[]; + channels?: unknown[]; + }; + if (Array.isArray(j.sites)) return { kind: "hub", siteCount: j.sites.length }; + if (Array.isArray(j.channels)) { + return { kind: "remote", channelCount: j.channels.length }; + } + } + } catch { + // unreachable — fall through to the single-site default + } + return { kind: "remote" }; +} + +type ChannelScope = { + channels: ChannelRef[]; + all: boolean; + matched: string[]; + unknown: string[]; +}; + +// Resolve a decoded link's `fc` channel NAMES against a live source's channels +// (by display name or slug, case-insensitive). No channel filter → whole corpus. +async function resolveLinkChannels( + source: ShardSource, + decoded: DecodedLink, +): Promise<ChannelScope> { + const all = await source.listChannels(); + if (!decoded.hasChannelFilter) { + return { channels: all, all: true, matched: [], unknown: [] }; + } + const want = decoded.channelNames.map((n) => n.toLowerCase()); + const matches = (c: ChannelRef): boolean => + want.includes(c.name.toLowerCase()) || want.includes(c.slug.toLowerCase()); + const channels = all.filter(matches); + const matched = channels.map((c) => c.name); + const unknown = decoded.channelNames.filter( + (n) => + !all.some( + (c) => + c.name.toLowerCase() === n.toLowerCase() || + c.slug.toLowerCase() === n.toLowerCase(), + ), + ); + return { channels, all: false, matched, unknown }; +} + +// Render the human-readable preview/echo plan for a decoded link. +function buildLinkPlan( + decoded: DecodedLink, + probe: OriginProbe, + scope: ChannelScope, + applied: boolean, +): string { + const lines: string[] = []; + lines.push(applied ? "## Applied share link" : "## Share-link preview"); + const kindNote = + probe.kind === "hub" + ? `hub (${probe.siteCount ?? "?"} member site(s))` + : `single site${probe.channelCount != null ? ` (${probe.channelCount} channel(s))` : ""}`; + lines.push(`- Source: ${decoded.origin} — ${kindNote}`); + lines.push(`- Query: ${renderQueryTree(decoded.tree)} _(from ${decoded.querySource})_`); + + if (scope.all) { + lines.push(`- Scope: whole corpus`); + } else { + const names = scope.matched.length ? scope.matched.join(", ") : "(none)"; + lines.push(`- Scope: ${scope.channels.length} channel(s): ${names}`); + if (scope.unknown.length > 0) { + lines.push(` - ⚠ channel name(s) not found in this corpus: ${scope.unknown.join(", ")}`); + } + if (scope.channels.length === 0) { + lines.push( + ` - ⚠ the link selects 0 channels here — nothing will match. Re-call ` + + `with overrides.clear_channels to search the whole corpus.`, + ); + } + } + + const filterLines = describeFilters(decoded.filters); + if (filterLines.length > 0) { + lines.push(`- Filters:`); + for (const f of filterLines) lines.push(` - ${f}`); + } else { + lines.push(`- Filters: none`); + } + + for (const w of decoded.warnings) lines.push(`- Note: ${w}`); + + if (!applied) { + lines.push( + ``, + `No search run yet. Call again with apply:true to switch source and ` + + `search, or pass overrides to adjust first.`, + ); + } + return lines.join("\n"); +} + +// A ScopedSnippet line: a linked `[m:ss]` for timed scopes, or a scope-tagged +// note for the non-timed scopes (metadata / description / tags). +function renderScopedSnippet( + source: ShardSource, + hit: SpecHit, + s: ScopedSnippet, +): string { + if (s.seconds > 0) { + const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `; + return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`; + } + return ` - [${s.scope}] ${s.text}`; +} + +function renderSpecResults(source: ShardSource, hits: SpecHit[]): string { + return hits + .map((h) => { + const head = + `### ${h.title}\n` + + `- video_id: ${h.videoId} | channel: ${h.channelName}` + + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + + (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); + const snips = h.snippets + .map((s) => renderScopedSnippet(source, h, s)) + .join("\n"); + return snips ? `${head}\n${snips}` : head; + }) + .join("\n\n"); +} + +async function handleOpenLink( + controller: SourceControllerLike, + args: Record<string, unknown>, +): Promise<ToolResult> { + const linkStr = typeof args.link === "string" ? args.link.trim() : ""; + if (!linkStr) return errorText("link is required"); + + let decoded: DecodedLink; + try { + decoded = decodeShareLink(linkStr); + } catch (e) { + return errorText(`could not parse link as a URL: ${(e as Error).message}`); + } + decoded = applyLinkOverrides(decoded, parseOverrides(args.overrides)); + + const apply = args.apply === true; + const probe = await probeOrigin(decoded.origin); + + if (!apply) { + // Preview: resolve channels against a transient source; do not commit. + const preview: ShardSource = + probe.kind === "hub" + ? new HubSource(decoded.origin) + : new RemoteSource(decoded.origin); + let scope: ChannelScope; + try { + scope = await resolveLinkChannels(preview, decoded); + } catch (e) { + return errorText( + `could not read the corpus at ${decoded.origin}: ${(e as Error).message}`, + ); + } + return text(buildLinkPlan(decoded, probe, scope, false)); + } + + // Apply: commit the source (reusing the controller), then search. + const spec: SourceSpec = + probe.kind === "hub" + ? { kind: "hub", url: decoded.origin } + : { kind: "remote", url: decoded.origin }; + try { + await controller.switchTo(spec); + } catch (e) { + return errorText(`open_link could not switch source: ${(e as Error).message}`); + } + const source = controller.current; + + let scope: ChannelScope; + try { + scope = await resolveLinkChannels(source, decoded); + } catch (e) { + return errorText( + `switched to ${source.label} but could not read its corpus: ${(e as Error).message}`, + ); + } + const plan = buildLinkPlan(decoded, probe, scope, true); + + const limit = typeof args.limit === "number" ? args.limit : 20; + const offset = typeof args.offset === "number" ? args.offset : 0; + const result = await runSearchSpec( + source, + scope.channels, + { + tree: decoded.tree, + filters: decoded.filters, + aliases: await source.loadAliases(), + }, + { limit, offset }, + ); + + const aliasNote = describeFiredAliases(result.firedAliases); + const rangeStart = result.total === 0 ? 0 : result.offset + 1; + const rangeEnd = result.offset + result.hits.length; + const footer = + `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` + + `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` + + `page(s) across ${result.scanned.channels} channel(s)` + + (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") + + (aliasNote ? `; ${aliasNote}` : "") + + ")"; + + const body = + result.hits.length === 0 + ? result.total === 0 + ? "No matches for this link's search." + : `No matches in this page (offset ${result.offset} is past the ${result.total} total).` + : `${result.total} video(s) matched:\n\n${renderSpecResults(source, result.hits)}`; + + return text(`${plan}\n\n---\n\n${body}${footer}`); +} + // ─── Prompts: the first-class `sweep` entry point ─── // A single slash command that drives Claude Code to run the browser's // "corpus sweep" on plan usage: enumerate a query's full match set, batch the @@ -912,7 +1352,21 @@ const PROMPTS = [ "on plan usage, no API key. With no scope arg it lists the channel groups " + "and asks you to pick a group/channels (or confirm 'all') before sweeping.", arguments: [ - { name: "query", description: "Term or phrase to sweep for.", required: true }, + { + name: "query", + description: + "Term or phrase to sweep for. Optional when a `link` is given (the " + + "link supplies the query).", + required: false, + }, + { + name: "link", + description: + "An archilyzer viewer share URL to seed the sweep from — its origin " + + "(source), query tree, and filters are decoded and applied via " + + "open_link before enumerating.", + required: false, + }, { name: "channel", description: "Optional channel slug/name to restrict the sweep to.", @@ -958,7 +1412,10 @@ function argStr(args: Record<string, unknown>, key: string): string | undefined function buildSweepPrompt(args: Record<string, unknown>) { const query = argStr(args, "query"); - if (!query) throw new Error("sweep requires a query argument"); + const link = argStr(args, "link"); + if (!query && !link) { + throw new Error("sweep requires a query argument (or a link)"); + } const channel = argStr(args, "channel"); const group = argStr(args, "group"); const channelsRaw = argStr(args, "channels"); @@ -968,10 +1425,11 @@ function buildSweepPrompt(args: Record<string, unknown>) { const directive = argStr(args, "directive") ?? "key claims & contradictions"; const batchSize = argStr(args, "batch_size") ?? "8"; const reportPath = argStr(args, "report_path") ?? "./sweep-report.md"; + const subject = query ? `"${query}"` : "the share link's search"; // Human-readable scope clauses + the literal search_transcripts scope args to - // pass. When none is given, the sweep must pick-first (list + ask) rather than - // silently scanning the whole corpus. + // pass. When none is given (and there's no link), the sweep must pick-first + // (list + ask) rather than silently scanning the whole corpus. const scopeClauses: string[] = []; if (channel) scopeClauses.push(`channel "${channel}"`); if (channels.length > 0) @@ -980,70 +1438,107 @@ function buildSweepPrompt(args: Record<string, unknown>) { const hasScope = scopeClauses.length > 0; const scopeArgsText = scopeClauses.join(" and "); - const introScope = hasScope - ? ` scoped to ${scopeArgsText}` - : " over a scope you will confirm with me first (see step 1)"; + const introScope = link + ? ` seeded from a share link (its source, query tree, and filters — see step 1)` + : hasScope + ? ` scoped to ${scopeArgsText}` + : " over a scope you will confirm with me first (see step 1)"; - // Build the numbered steps. With an explicit scope we search directly; with - // none we insert a pick-first step and the Search step uses the chosen scope. const steps: string[] = []; - if (!hasScope) { + // ── Discovery / enumeration differs for a link-seeded sweep vs a query one ── + if (link) { + steps.push( + `**Decode & confirm the link.** Call \`open_link\` with ` + + `link="${link}" (preview mode — no apply). It returns a plan: the ` + + `resolved source (origin, hub or single-site), the query tree, every ` + + `active filter, the channel scope validated against the corpus, and any ` + + `ignored bits (e.g. the vestigial \`fk\` tracks). **Show me the plan and ` + + `confirm it captures what I want.** If I ask for a change ("drop the ` + + `availability filter", "only channel X", "search Y instead"), re-call ` + + `\`open_link\` with the matching \`overrides\` until the plan is right.`, + ); + steps.push( + `**Apply & enumerate.** Call \`open_link\` again with the confirmed ` + + `arguments plus apply:true — this switches the active source to the ` + + `link's origin and runs the search (the full query tree + filters, at ` + + `fidelity). Page it with a rising \`offset\` (offset += limit) until ` + + `\`has_more\` is no to collect the whole worklist of video ids; note the ` + + `\`total\`. Results already carry linked \`[mm:ss](url)\` timestamps and ` + + `the applied plan is echoed at the top — record the source, query, and ` + + `filters in the report. If coverage is PARTIAL, say so.`, + ); + } else { + if (!hasScope) { + steps.push( + `**Choose the scope first — do NOT default to the whole corpus.** No ` + + `channel/channels/group was supplied. Call \`list_channels\`, present ` + + `the channel groups and their channels to me, and ask which group(s) ` + + `or channel(s) to sweep — or to confirm **all** for the whole corpus. ` + + `Wait for my choice before enumerating anything. Only sweep everything ` + + `if I explicitly choose "all". Use my choice as the ` + + `\`channel\`/\`channels\`/\`group\` scope in every ` + + `\`search_transcripts\` call below.`, + ); + } steps.push( - `**Choose the scope first — do NOT default to the whole corpus.** No ` + - `channel/channels/group was supplied. Call \`list_channels\`, present ` + - `the channel groups and their channels to me, and ask which group(s) or ` + - `channel(s) to sweep — or to confirm **all** for the whole corpus. Wait ` + - `for my choice before enumerating anything. Only sweep everything if I ` + - `explicitly choose "all". Use my choice as the \`channel\`/\`channels\`/` + - `\`group\` scope in every \`search_transcripts\` call below.`, + `**Search.** Call \`search_transcripts\` with query "${query}"` + + (hasScope + ? ` and ${scopeArgsText}` + : ` and the scope I chose in step 1`) + + `. Curated aliases auto-expand the query — the footer reports which ` + + `fired (e.g. mis-transcribed spellings) and names the resolved scope ` + + `(and warns about any channel/group token that matched nothing — fix a ` + + `typo before continuing). Treat the *expanded* match set as your target ` + + `and mention the expansion and the scope in the report.`, + ); + steps.push( + `**Enumerate the full worklist.** Page the complete set with ` + + `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` + + `until \`has_more\` is false — this gives you every id/title/channel/date ` + + `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` + + `(page/video cap), say so in the report — the sweep is then a sample, ` + + `not exhaustive.`, ); } steps.push( - `**Search.** Call \`search_transcripts\` with query "${query}"` + - (hasScope - ? ` and ${scopeArgsText}` - : ` and the scope I chose in step 1`) + - `. Curated aliases auto-expand the query — the footer reports which fired ` + - `(e.g. mis-transcribed spellings) and names the resolved scope (and warns ` + - `about any channel/group token that matched nothing — fix a typo before ` + - `continuing). Treat the *expanded* match set as your target and mention ` + - `the expansion and the scope in the report.`, - ); - - steps.push( - `**Enumerate the full worklist.** Page the complete set with ` + - `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` + - `until \`has_more\` is false — this gives you every id/title/channel/date ` + - `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` + - `(page/video cap), say so in the report — the sweep is then a sample, not ` + - `exhaustive.`, - ); - - steps.push( `**Plan.** With N total matches and a batch size of ${batchSize}, that is ` + `\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch ` + `count) before you start.`, ); + // ── Feature 2: subagent-per-batch map-reduce so the heavy transcript text + // lives only in ephemeral subagents; the orchestrator keeps just the distilled + // cited findings. Feature 1: linked citations built from the moment URLs. steps.push( - `**Per batch**, for each group of up to ${batchSize} video ids:\n` + - ` - Call \`get_transcripts\` with those ids **and the query** so each ` + - `transcript comes back as bounded, timestamped excerpt windows around the ` + - `matches (alias-correct, high-signal).\n` + - ` - **Cross-reference** the batch against the report so far. Upsert ` + - `findings — claims, and contradictions with earlier claims — into ` + - `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` + - ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it ` + - `each batch), then **drop the raw transcript text** once folded — don't ` + - `carry it forward.`, + `**Per batch (map-reduce), for each group of up to ${batchSize} video ids:**\n` + + ` - **Spawn a subagent** (the Task tool) for the batch. Give it the ` + + `batch's ids, the query, and the directive, and tell it to: call ` + + `\`get_transcripts\` with those ids **and the query** (bounded, ` + + `timestamped, alias-correct excerpt windows), extract only what serves the ` + + `directive, and **return a compact markdown fragment of cited, linked ` + + `findings and nothing else** — the raw transcript text stays inside the ` + + `subagent and never enters your context. Every citation must be a link: ` + + `**\`[title @ mm:ss](<moment url>)\`**, where the moment URL is taken ` + + `straight from the \`[mm:ss](url)\` links in that batch's ` + + `\`get_transcripts\` output (they seek to the exact second).\n` + + ` - **Merge** the returned fragment into \`${reportPath}\`: cross-` + + `reference it against the report so far and upsert findings — claims, and ` + + `contradictions with earlier claims — into well-titled \`## sections\` ` + + `(Write/Edit). Then discard the fragment. Batches are independent, so you ` + + `may dispatch several subagents in parallel.\n` + + ` - **Fallback:** if no subagent/Task tool is available, do the batch ` + + `inline — call \`get_transcripts\` yourself, fold the cited linked ` + + `findings into the report, then **drop the raw transcript text** before ` + + `moving on (don't carry it forward).`, ); steps.push( `**Finish.** Repeat to the end of the worklist, then write a short summary ` + `section (the scope swept, how many videos covered, headline findings, any ` + - `partial-coverage caveat) and tell me the report path.`, + `partial-coverage caveat) and tell me the report path. Keep every citation ` + + `a clickable \`[title @ mm:ss](url)\` link.`, ); const numbered = steps @@ -1051,16 +1546,18 @@ function buildSweepPrompt(args: Record<string, unknown>) { .join("\n\n"); const text = - `Run a **corpus sweep** for the query **"${query}"**${introScope}, ` + - `extracting **${directive}**, and maintain a running markdown report at ` + + `Run a **corpus sweep** for ${subject}${introScope}, extracting ` + + `**${directive}**, and maintain a running markdown report at ` + `\`${reportPath}\`. You are the sweep engine — work through the whole match ` + `set methodically, using the transcript MCP tools for evidence and your own ` + `Write/Edit tools for the report. The MCP is read-only; never try to change ` + - `the archive.\n\n` + + `the archive. **Cite every finding as a clickable ` + + `\`[title @ mm:ss](<moment url>)\` link** (the moment URLs come straight ` + + `from the tool output).\n\n` + `Follow these steps:\n\n${numbered}`; return { - description: `Corpus sweep for "${query}" → ${reportPath}`, + description: `Corpus sweep for ${subject} → ${reportPath}`, messages: [ { role: "user" as const, diff --git a/mcp/src/shareLink.test.ts b/mcp/src/shareLink.test.ts @@ -0,0 +1,168 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + newGroup, + newLeaf, + stringifyRoot, +} from "yt-dlp-transcript-common/lib/searchQuery"; +import { + decodeShareLink, + applyLinkOverrides, + renderQueryTree, + describeFilters, +} from "./shareLink"; + +const ORIGIN = "https://rekietalyzer.pages.dev"; + +// A realistic composite qt tree: transcript:"coffee" AND NOT tags:"espresso". +function qtParam(): string { + const tree = newGroup({ + op: "AND", + children: [ + newLeaf({ query: "coffee", scope: "transcripts" }), + newLeaf({ query: "espresso", scope: "tags", negate: true }), + ], + }); + return encodeURIComponent(stringifyRoot(tree)); +} + +// A full-fidelity share link: qt tree + every filter facet + a vestigial fk. +function fullLink(): string { + const p = new URLSearchParams(); + p.set("qt", "__QT__"); + p.set("fv", "1"); + p.append("fc", "Rekieta Law"); + p.append("fc", "Rekieta Law (Rumble)"); + p.append("ft", "v"); // videos only + p.append("fa", "a"); // all-ages only + p.append("fav", "a"); // available only + p.append("fk", "live_chat"); // vestigial + p.set("fdf", "20250101"); + p.set("fdt", "20251231"); + // qt must not be URL-double-encoded by URLSearchParams — splice it in raw. + return `${ORIGIN}/?${p.toString().replace("__QT__", qtParam())}`; +} + +test("decode: origin is extracted from the pasted URL", () => { + const d = decodeShareLink(`${ORIGIN}/?q=x`); + assert.equal(d.origin, ORIGIN); +}); + +test("decode: qt tree parses to a composite query", () => { + const d = decodeShareLink(fullLink()); + assert.equal(d.querySource, "qt"); + const rendered = renderQueryTree(d.tree); + assert.match(rendered, /transcript:"coffee"/); + assert.match(rendered, /NOT tags:"espresso"/); + assert.match(rendered, /AND/); +}); + +test("decode: every v1 filter facet is decoded", () => { + const d = decodeShareLink(fullLink()); + assert.ok(d.filters); + const f = d.filters!; + // ft=v → videos kept, livestreams dropped. + assert.equal(f.videos, true); + assert.equal(f.livestreams, false); + // fa=a → all-ages kept, restricted dropped. + assert.equal(f.allAges, true); + assert.equal(f.restricted, false); + // fav=a → available kept, unlisted/deleted dropped. + assert.equal(f.available, true); + assert.equal(f.unlisted, false); + assert.equal(f.deleted, false); + // dates. + assert.equal(f.dateFrom, "20250101"); + assert.equal(f.dateTo, "20251231"); +}); + +test("decode: fc channel names are captured; fk is decoded but flagged ignored", () => { + const d = decodeShareLink(fullLink()); + assert.deepEqual(d.channelNames.sort(), [ + "Rekieta Law", + "Rekieta Law (Rumble)", + ]); + assert.equal(d.hasChannelFilter, true); + assert.deepEqual(d.ignoredTracks, ["live_chat"]); + assert.ok(d.warnings.some((w) => /fk/.test(w) && /ignored/.test(w))); +}); + +test("decode: legacy q/re → a single regex transcripts leaf, no filters", () => { + const d = decodeShareLink(`${ORIGIN}/?q=coffee&re=1`); + assert.equal(d.querySource, "legacy"); + assert.equal(d.filters, null); + assert.equal(d.hasChannelFilter, false); + const leaf = d.tree.children[0]; + assert.equal(leaf.kind, "leaf"); + assert.match(renderQueryTree(d.tree), /transcript:\/coffee\//); +}); + +test("decode: legacy m=subs → a live-chat leaf", () => { + const d = decodeShareLink(`${ORIGIN}/?q=hello&m=subs`); + assert.match(renderQueryTree(d.tree), /live-chat:"hello"/); +}); + +test("decode: no fv block → unfiltered, whole-corpus scope", () => { + const d = decodeShareLink(`${ORIGIN}/?fc=Some%20Channel`); + assert.equal(d.filters, null); + assert.equal(d.hasChannelFilter, false); + assert.deepEqual(d.channelNames, []); +}); + +test("decode: malformed qt falls back to legacy/empty with a warning", () => { + const d = decodeShareLink(`${ORIGIN}/?qt=not-json&q=fallback`); + assert.equal(d.querySource, "legacy"); + assert.ok(d.warnings.some((w) => /malformed/.test(w))); + assert.match(renderQueryTree(d.tree), /transcript:"fallback"/); +}); + +// ─── overrides ─── + +test("override: clearAvailability resets the fav facet to keep-all", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearAvailability: true, + }); + assert.equal(d.filters!.available, true); + assert.equal(d.filters!.unlisted, true); + assert.equal(d.filters!.deleted, true); + assert.ok(d.warnings.some((w) => /availability filter removed/.test(w))); +}); + +test("override: clearFilters drops the whole filter block", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearFilters: true, + }); + assert.equal(d.filters, null); +}); + +test("override: channels replaces the channel scope", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + channels: ["Only This"], + }); + assert.deepEqual(d.channelNames, ["Only This"]); + assert.equal(d.hasChannelFilter, true); +}); + +test("override: clearChannels widens to the whole corpus", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + clearChannels: true, + }); + assert.deepEqual(d.channelNames, []); + assert.equal(d.hasChannelFilter, false); +}); + +test("override: query replaces the tree with a single leaf", () => { + const d = applyLinkOverrides(decodeShareLink(fullLink()), { + query: "different term", + }); + assert.match(renderQueryTree(d.tree), /transcript:"different term"/); +}); + +test("describeFilters: names the constrained facets only", () => { + const d = decodeShareLink(fullLink()); + const lines = describeFilters(d.filters); + assert.ok(lines.some((l) => /videos only/.test(l))); + assert.ok(lines.some((l) => /all-ages only/.test(l))); + assert.ok(lines.some((l) => /available.*kept/.test(l))); + assert.ok(lines.some((l) => /uploaded/.test(l))); +}); diff --git a/mcp/src/shareLink.ts b/mcp/src/shareLink.ts @@ -0,0 +1,316 @@ +// Decode an archilyzer viewer **share link** into a search spec + source target. +// +// A share URL is `origin` + a `qt=` composite query tree (or a legacy `q`/`re`) +// + the v1 filter params (`fc/ft/fa/fav/fk/fdf/fdt`, gated by `fv=1`). This +// module turns that URL into the pieces the MCP needs to reproduce the search at +// full fidelity: +// • the origin (→ probed by the server for hub vs single-site), +// • the query tree (parsed via the browser's own `parseRoot`), +// • the decoded filters (via the browser's own `parseShareV1`), +// • the requested channel NAMES (`fc`, resolved against the live corpus later), +// • and any ignored/warned pieces (`fk` tracks are vestigial in the composite +// share model — decoded and reported as ignored, not enforced). +// +// Pure: no network. The server does the origin probe and channel resolution. + +import { + parseRoot, + rootFromLegacy, + emptyRoot, + isLeaf, + isGroup, + isNodeActive, + newGroup, + newLeaf, + type GroupNode, + type QueryNode, +} from "yt-dlp-transcript-common/lib/searchQuery"; +import { + hasShareV1, + parseShareV1, +} from "yt-dlp-transcript-common/components/shareUrl"; +import type { SearchFilters } from "./search"; + +export type DecodedLink = { + // The pasted URL's origin (scheme + host[:port]). The source target. + origin: string; + // The composite query tree — from `qt=`, a synthesized legacy leaf, or an + // empty root (filter-only browse). + tree: GroupNode; + querySource: "qt" | "legacy" | "empty"; + // Positive share filters (ft/fa/fav + dates), or null when the link carries no + // v1 filter block (`fv=1`) — i.e. an unfiltered search. + filters: SearchFilters | null; + // Channel NAMES the link selects (`fc`). Resolved against the live corpus by + // the caller. Empty with `filters === null` means "whole corpus". + channelNames: string[]; + // Whether the link constrained channels at all (had a v1 filter block). When + // false, channel scope is the whole corpus regardless of `channelNames`. + hasChannelFilter: boolean; + // Vestigial `fk` subtitle-track tokens — decoded but a no-op. + ignoredTracks: string[]; + // Human-facing notes (malformed qt, ignored fk, …). + warnings: string[]; +}; + +// Decode a share link. Never throws for a malformed query tree (falls back to a +// legacy/empty tree with a warning); throws only if `link` isn't a URL at all. +export function decodeShareLink(link: string): DecodedLink { + const url = new URL(link); + const search = url.search; + const params = new URLSearchParams(search); + const warnings: string[] = []; + + // ── Query: qt= tree preferred, else legacy q/re, else empty (filter-only). + let tree: GroupNode = emptyRoot(); + let querySource: DecodedLink["querySource"] = "empty"; + const qt = params.get("qt"); + if (qt) { + const parsed = parseRoot(qt); + if (parsed) { + tree = parsed; + querySource = "qt"; + } else { + warnings.push("malformed `qt` query tree — ignored"); + } + } + if (querySource === "empty") { + const q = params.get("q") ?? ""; + const mode = params.get("m") === "subs" ? "subs" : "transcripts"; + const re = params.get("re") === "1"; + if (q.trim() !== "") { + tree = rootFromLegacy(q, mode, re); + querySource = "legacy"; + } + } + + // ── Filters: only meaningful when the v1 block is present (`fv=1`). + const v1 = hasShareV1(search); + // Pass the requested fc names AS the channel universe so none are dropped — + // real validation against the live corpus happens at apply time. + const requestedChannels = params.getAll("fc"); + const sel = parseShareV1(search, requestedChannels); + const ignoredTracks = [...sel.tracks]; + if (ignoredTracks.length > 0) { + warnings.push( + `\`fk\` subtitle-track filter is vestigial in the composite share model — ` + + `${ignoredTracks.length} track token(s) decoded but ignored: ${ignoredTracks.join(", ")}`, + ); + } + + const filters: SearchFilters | null = v1 + ? { + videos: sel.videos, + livestreams: sel.livestreams, + allAges: sel.allAges, + restricted: sel.restricted, + available: sel.available, + unlisted: sel.unlisted, + deleted: sel.deleted, + ...(sel.dateFrom ? { dateFrom: sel.dateFrom } : {}), + ...(sel.dateTo ? { dateTo: sel.dateTo } : {}), + } + : null; + + return { + origin: url.origin, + tree, + querySource, + filters, + channelNames: v1 ? [...sel.selectedChannels] : [], + hasChannelFilter: v1, + ignoredTracks, + warnings, + }; +} + +// Structured adjustments an agent can pass to honor a natural-language edit +// ("remove the availability filter", "only channel X", "search 'foo' instead"). +// Every field is optional; only the provided facets are changed. +export type LinkOverrides = { + // Drop every decoded filter (ft/fa/fav/dates) → unfiltered. + clearFilters?: boolean; + // Reset a single filter facet to its keep-everything default. + clearAvailability?: boolean; // fav + clearType?: boolean; // ft + clearAge?: boolean; // fa + clearDates?: boolean; // fdf/fdt + // Widen the channel scope to the whole corpus (ignore fc). + clearChannels?: boolean; + // Replace the channel scope with these names (validated against the corpus). + channels?: string[]; + // Replace/clear the upload-date bounds ("YYYYMMDD"; null clears that bound). + dateFrom?: string | null; + dateTo?: string | null; + // Replace the whole query with a single leaf (query + optional regex/scope). + query?: string; + regex?: boolean; + queryScope?: "transcripts" | "chat" | "metadata" | "description" | "tags"; +}; + +const KEEP_ALL_FILTERS: SearchFilters = { + videos: true, + livestreams: true, + allAges: true, + restricted: true, + available: true, + unlisted: true, + deleted: true, +}; + +// Apply overrides to a decoded link, returning a new DecodedLink. Records what +// changed as warnings so the plan can echo it. +export function applyLinkOverrides( + decoded: DecodedLink, + overrides: LinkOverrides | undefined, +): DecodedLink { + if (!overrides) return decoded; + const next: DecodedLink = { + ...decoded, + warnings: [...decoded.warnings], + channelNames: [...decoded.channelNames], + filters: decoded.filters ? { ...decoded.filters } : null, + }; + const note = (s: string): void => { + next.warnings.push(`override: ${s}`); + }; + + // ── Query replacement. + if (typeof overrides.query === "string" && overrides.query.trim() !== "") { + next.tree = newGroup({ + children: [ + newLeaf({ + query: overrides.query, + scope: overrides.queryScope ?? "transcripts", + useRegex: overrides.regex === true, + }), + ], + }); + next.querySource = "qt"; + note(`query replaced with "${overrides.query}"`); + } + + // ── Channel scope. + if (overrides.clearChannels) { + next.channelNames = []; + next.hasChannelFilter = false; + note("channel scope widened to the whole corpus"); + } + if (Array.isArray(overrides.channels)) { + next.channelNames = overrides.channels.filter((c) => c.trim() !== ""); + next.hasChannelFilter = true; + note(`channel scope set to [${next.channelNames.join(", ")}]`); + } + + // ── Filters. + if (overrides.clearFilters) { + next.filters = null; + note("all filters removed"); + } else if (next.filters) { + const f = next.filters; + if (overrides.clearAvailability) { + f.available = true; + f.unlisted = true; + f.deleted = true; + note("availability filter removed"); + } + if (overrides.clearType) { + f.videos = true; + f.livestreams = true; + note("type filter removed"); + } + if (overrides.clearAge) { + f.allAges = true; + f.restricted = true; + note("age filter removed"); + } + if (overrides.clearDates) { + delete f.dateFrom; + delete f.dateTo; + note("date filter removed"); + } + if (overrides.dateFrom !== undefined) { + if (overrides.dateFrom === null) delete f.dateFrom; + else f.dateFrom = overrides.dateFrom; + note(`dateFrom set to ${overrides.dateFrom ?? "(none)"}`); + } + if (overrides.dateTo !== undefined) { + if (overrides.dateTo === null) delete f.dateTo; + else f.dateTo = overrides.dateTo; + note(`dateTo set to ${overrides.dateTo ?? "(none)"}`); + } + } else if ( + overrides.dateFrom !== undefined || + overrides.dateTo !== undefined + ) { + // Setting a date on an otherwise-unfiltered link creates a filter block. + next.filters = { ...KEEP_ALL_FILTERS }; + if (typeof overrides.dateFrom === "string") next.filters.dateFrom = overrides.dateFrom; + if (typeof overrides.dateTo === "string") next.filters.dateTo = overrides.dateTo; + note("date filter added"); + } + + return next; +} + +// ── Human-readable renderings for the preview plan ── + +const SCOPE_LABEL: Record<string, string> = { + transcripts: "transcript", + chat: "live-chat", + metadata: "title/channel", + description: "description", + tags: "tags", +}; + +// Render a query tree readably, e.g. +// transcript:"coffee" AND tags:"espresso" AND NOT chat:"spam" +export function renderQueryTree(node: QueryNode): string { + if (isLeaf(node)) { + const label = SCOPE_LABEL[node.scope] ?? node.scope; + const kind = node.useRegex ? "/" : '"'; + const end = node.useRegex ? "/" : '"'; + const body = `${label}:${kind}${node.query}${end}`; + return node.negate ? `NOT ${body}` : body; + } + if (!isGroup(node)) return ""; + const parts = node.children + .filter(isNodeActive) + .map((c) => { + const s = renderQueryTree(c); + return isGroup(c) ? `(${s})` : s; + }) + .filter((s) => s !== ""); + if (parts.length === 0) return "(everything in scope)"; + const joined = parts.join(` ${node.op} `); + return node.negate ? `NOT (${joined})` : joined; +} + +// A compact list of the active filters for the plan, or [] when unfiltered. +export function describeFilters(f: SearchFilters | null): string[] { + if (!f) return []; + const out: string[] = []; + // Type (ft): only worth noting when it excludes one side. + if (f.videos !== f.livestreams) { + out.push(`type: ${f.videos ? "videos only" : "livestreams only"}`); + } else if (!f.videos && !f.livestreams) { + out.push("type: none kept (videos and livestreams both excluded)"); + } + // Audience (fa). + if (f.allAges !== f.restricted) { + out.push(`audience: ${f.allAges ? "all-ages only" : "age-restricted only"}`); + } else if (!f.allAges && !f.restricted) { + out.push("audience: none kept"); + } + // Availability (fav). + const av: string[] = []; + if (f.available) av.push("available"); + if (f.unlisted) av.push("unlisted"); + if (f.deleted) av.push("deleted"); + if (av.length < 3) out.push(`availability: ${av.length ? av.join(" + ") : "none"} kept`); + // Dates (fdf/fdt). + if (f.dateFrom || f.dateTo) { + out.push(`uploaded: ${f.dateFrom ?? "…"} → ${f.dateTo ?? "…"}`); + } + return out; +} diff --git a/mcp/src/source.ts b/mcp/src/source.ts @@ -2,9 +2,17 @@ import { readFile, readdir } from "node:fs/promises"; import path from "node:path"; import { transcriptPageFileName, + subsPageFileName, + pageFileName, type ChannelTranscriptsManifest, + type ChannelSubsManifest, + type Manifest, } from "yt-dlp-transcript-common/lib/manifest"; -import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; +import type { + TranscriptDetail, + DisplaySummary, +} from "yt-dlp-transcript-common/lib/transcripts"; +import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import { coerceAliasConfig, type SearchAlias, @@ -48,6 +56,48 @@ const EMPTY_GROUPS: ChannelGroups = { defaultGroupId: DEFAULT_GROUP_FALLBACK_ID, }; +// Per-video availability, joined in from the summaries shards (a +// TranscriptDetail record does not carry deleted/unlisted flags). Keyed by the +// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript +// page record carries — so the search engine can apply the `fav` filter. +export type VideoAvailability = { isDeleted: boolean; isUnlisted: boolean }; + +// Read a site's global summaries shards (summaries/manifest.json + +// summaries/page-NNNN.json) via `readPage` and fold them into a slug → +// availability map. Tolerant: an absent/malformed manifest yields an empty map, +// and a page that fails to read is skipped. `readManifest`/`readPage` throw or +// return null on absence per the source's transport. +async function buildAvailabilityMap( + readManifest: () => Promise<Manifest | null>, + readPage: (page: number) => Promise<DisplaySummary[] | null>, +): Promise<Map<string, VideoAvailability>> { + const map = new Map<string, VideoAvailability>(); + let manifest: Manifest | null; + try { + manifest = await readManifest(); + } catch { + return map; + } + if (!manifest || typeof manifest.pageCount !== "number") return map; + for (let page = 0; page < manifest.pageCount; page++) { + let records: DisplaySummary[] | null; + try { + records = await readPage(page); + } catch { + continue; + } + if (!records) continue; + for (const r of records) { + if (typeof r.slug !== "string") continue; + map.set(r.slug, { + isDeleted: r.isDeleted === true, + isUnlisted: r.isUnlisted === true, + }); + } + } + return map; +} + // Parse a summaries/manifest.json blob into channel-group defs, tolerating any // missing/malformed shape (→ empty fallback). function parseGroupsManifest(raw: unknown): ChannelGroups { @@ -76,6 +126,25 @@ export interface ShardSource { // empty fallback when absent/malformed, or in hub mode (federated per-site // groups are a different model — deferred). Cached per source. loadGroups(): Promise<ChannelGroups>; + // The public origin of the viewer that owns this source's videos, or null + // when there isn't one (a local dir on disk). Used to build archilyzer viewer + // deep links for cited moments (momentUrl). A single-site remote returns its + // base URL; a hub returns null because each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video. + publicOrigin(): string | null; + // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null + // when the channel ships no subs shards. Same slugToPage/pageCount shape as + // the transcripts manifest. Fetched lazily — only the chat search scope needs + // it. + subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>; + // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record + // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`). + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>; + // A map of every video's availability (deleted/unlisted), keyed by the + // member-local video slug (`<channelSlug>/<id>`), built from the summaries + // shards. Fetched lazily and cached — only the `fav` availability filter needs + // it. Empty when the source ships no summaries. + availabilityMap(): Promise<Map<string, VideoAvailability>>; } // Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose @@ -104,10 +173,58 @@ export class LocalSource implements ShardSource { readonly label: string; private aliases?: SearchAlias[]; private groups?: ChannelGroups; + private availability?: Map<string, VideoAvailability>; constructor(private dir: string) { this.label = `local:${dir}`; } + // A local dir has no public viewer origin — cited moments fall back to + // platform links (momentUrl). + publicOrigin(): string | null { + return null; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + try { + const raw = await readFile( + path.join(this.dir, "subs", ch.slug, "manifest.json"), + "utf8", + ); + return JSON.parse(raw) as ChannelSubsManifest; + } catch { + return null; // channel ships no subs shards + } + } + + async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + const raw = await readFile( + path.join(this.dir, "subs", ch.slug, subsPageFileName(page)), + "utf8", + ); + return JSON.parse(raw) as SubsDetail[]; + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + this.availability = await buildAvailabilityMap( + async () => { + const raw = await readFile( + path.join(this.dir, "summaries", "manifest.json"), + "utf8", + ); + return JSON.parse(raw) as Manifest; + }, + async (page) => { + const raw = await readFile( + path.join(this.dir, "summaries", pageFileName(page)), + "utf8", + ); + return JSON.parse(raw) as DisplaySummary[]; + }, + ); + return this.availability; + } + async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { @@ -195,11 +312,45 @@ export class RemoteSource implements ShardSource { private base: string; private aliases?: SearchAlias[]; private groups?: ChannelGroups; + private availability?: Map<string, VideoAvailability>; constructor(baseUrl: string) { this.base = baseUrl.replace(/\/+$/, ""); this.label = `remote:${this.base}`; } + // The deployed site origin — the archilyzer viewer that owns these videos. + publicOrigin(): string | null { + return this.base; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + try { + const res = await fetch(`${this.base}/subs/${ch.slug}/manifest.json`); + return res.ok ? ((await res.json()) as ChannelSubsManifest) : null; + } catch { + return null; + } + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + return this.getJson(`/subs/${ch.slug}/${subsPageFileName(page)}`); + } + + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + this.availability = await buildAvailabilityMap( + async () => { + const res = await fetch(`${this.base}/summaries/manifest.json`); + return res.ok ? ((await res.json()) as Manifest) : null; + }, + async (page) => { + const res = await fetch(`${this.base}/summaries/${pageFileName(page)}`); + return res.ok ? ((await res.json()) as DisplaySummary[]) : null; + }, + ); + return this.availability; + } + async loadAliases(): Promise<SearchAlias[]> { if (this.aliases) return this.aliases; try { @@ -263,6 +414,7 @@ export class HubSource implements ShardSource { readonly hubBase: string; private members = new Map<string, RemoteSource>(); // siteId -> source private aliases?: SearchAlias[]; + private availability?: Map<string, VideoAvailability>; // Optional subset allowlist of member siteIds. Undefined = federate every // member; a set restricts listChannels() to those members (site discovery via // listSites() stays unfiltered so a picker can still see all members). @@ -359,4 +511,55 @@ export class HubSource implements ShardSource { if (!ch.siteId) throw new Error("hub channel ref missing siteId"); return this.memberFor(ch.siteId).transcriptPage(ch, page); } + + // A hub has no single viewer origin — each video's origin is its member + // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers + // per-video. Return null so we never mint a wrong-origin viewer link. + publicOrigin(): string | null { + return null; + } + + async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> { + if (!ch.siteId) return null; + try { + return await this.memberFor(ch.siteId).subsManifest(ch); + } catch { + return null; // member not yet registered / unreachable + } + } + + subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> { + if (!ch.siteId) throw new Error("hub channel ref missing siteId"); + return this.memberFor(ch.siteId).subsPage(ch, page); + } + + // Merge each member's availability map. Keys are member-local slugs + // (`<channelSlug>/<id>`) — the same slug a member's transcript page records + // carry — so a per-record `fav` lookup joins correctly. Built lazily/cached. + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + if (this.availability) return this.availability; + const merged = new Map<string, VideoAvailability>(); + let sites: HubSite[]; + try { + sites = (await this.listSites()).filter( + (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId), + ); + } catch { + this.availability = merged; + return merged; + } + for (const site of sites) { + const remote = this.members.get(site.siteId) ?? new RemoteSource(site.url); + this.members.set(site.siteId, remote); + try { + for (const [slug, avail] of await remote.availabilityMap()) { + merged.set(slug, avail); + } + } catch { + // skip an unreachable member + } + } + this.availability = merged; + return merged; + } } diff --git a/mcp/src/sourceController.test.ts b/mcp/src/sourceController.test.ts @@ -8,7 +8,13 @@ import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js"; import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; -import type { ChannelGroups, ChannelRef, HubSite, ShardSource } from "./source"; +import type { + ChannelGroups, + ChannelRef, + HubSite, + ShardSource, + VideoAvailability, +} from "./source"; import type { SourceSpec } from "./sources"; import { SourceController } from "./sourceController"; import { createServer } from "./server"; @@ -56,6 +62,18 @@ class FakeSource implements ShardSource { async transcriptPage(): Promise<TranscriptDetail[]> { return []; } + publicOrigin(): string | null { + return null; + } + async subsManifest(): Promise<null> { + return null; + } + async subsPage(): Promise<[]> { + return []; + } + async availabilityMap(): Promise<Map<string, VideoAvailability>> { + return new Map(); + } } function labelFor(spec: SourceSpec): string {