import { normalizeDateArg } from "./dateArg"; import { Server, type Prompt, type ServerOptions, type Tool, } from "@modelcontextprotocol/server"; import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown"; import { formatDate, formatDuration } from "yt-dlp-transcript-common/lib/format"; import { manifestHasDigest } from "yt-dlp-transcript-common/lib/digests"; import type { VideoStat } from "yt-dlp-transcript-common/lib/stats"; import { momentUrl, momentBaseUrl, viewerPostUrl } from "yt-dlp-transcript-common/lib/momentUrl"; import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags"; import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags"; import { cuesOfTrack, hitsAcrossTracks, inTrackLabel, recordTracks, trackLabel, } from "yt-dlp-transcript-common/lib/captionTracks"; import { windowedTranscript } from "yt-dlp-transcript-common/lib/search/window"; import { VIDEO_STATES, isVideoState, type VideoState, } from "yt-dlp-transcript-common/lib/availability"; import type { LayerScope } from "yt-dlp-transcript-common/lib/searchQuery"; import { sortGroups, resolveChannelGroupId, FALLBACK_GROUP, type ChannelGroup, } from "yt-dlp-transcript-common/lib/channelGroups"; import { ioStatsEnabled, ioStatsSnapshot, type ChannelRef, type HubSite, type ShardSource, } from "./source"; import type { SourceSpec } from "./sources"; import { SourceRegistry, type ResolvedSource } from "./sourceRegistry"; import type { Post } from "yt-dlp-transcript-common/lib/posts"; import { searchTranscripts, findPost, getThread, parseContentTypes, findVideo, buildMatcher, getWindowedTranscript, MCP_POLICY, runSearchSpec, type SearchFilters, type SearchResult, type SpecHit, type ScopedSnippet, } from "./search"; import { parsePromptRequest, requestFromArguments, validateSweepArguments, DEFAULT_REPORT_PATH, type PromptRequest, } from "./promptRequest"; import { buildSweepInstructions, buildAskInstructions, type PlanContext, } from "./instructions"; import { renderReportIndex, renderReportPage, reportsLine, sourceReports, } from "./reports"; import { fetchClip, renderFetchClip, validateFetchClipArgs, editorFromEnv, notFoundNote, isVideoId, NO_EDITOR_TEXT, type FetchClipDeps, type PollProgress, } from "./fetchClip"; import { editorTranscript, renderEditorTranscript, type EditorDeps } from "./editorOps"; import { ARCHIVAL_TOOLS, channelCoverageTool, enqueue, getJob, notesTool, type ToolAnswer, } from "./archivalTools"; import { extractVideoId } from "yt-dlp-transcript-common/lib/videoId"; import { decodeShareLink, applyLinkOverrides, renderQueryTree, describeFilters, type DecodedLink, type LinkOverrides, } from "./shareLink"; // The building blocks momentUrl needs from a hit (viewer origin comes from the // per-video siteUrl, else the source's single public origin). type LinkableHit = { slug: string; siteUrl?: string; webpageUrl?: string; platform?: Platform; }; // A moment deep link for one cited second of a hit, or null when neither a // viewer nor a platform link can be built (e.g. a local source + no webpage url). function momentLinkFor( source: ShardSource, h: LinkableHit, seconds: number, ): string | null { return momentUrl({ siteOrigin: h.siteUrl ?? source.publicOrigin(), slug: h.slug, seconds, webpageUrl: h.webpageUrl, platform: h.platform, }); } // A `[m:ss]` timestamp rendered as a Markdown link when we can build one, else // bare. Used inside a `- [ … ] text` snippet line → `- [[m:ss](url)] text`. function stampMarkup( source: ShardSource, h: LinkableHit, clock: string, seconds: number, ): string { const url = momentLinkFor(source, h, seconds); return url ? `[${clock}](${url})` : clock; } // The requested timestamp-link form: "base" is the compact agent-pipeline form // (one `moment_base` per video + `[m:ss|seconds]` lines); anything else — the // default — is the full inline Markdown links. function linkStyleOf(args: Record): "inline" | "base" { return args.link_style === "base" ? "base" : "inline"; } // The compact stamp contents for link_style:"base": `m:ss|156`. The integer is // floored exactly like momentUrl floors its seconds, so appending it to the // video's moment_base cites the identical second the inline link would. function baseStamp(clock: string, seconds: number): string { return `${clock}|${Math.floor(seconds)}`; } // The appendable moment base for a hit (mirrors momentLinkFor), or null when // none can be built — no viewer origin AND the platform's time param doesn't // take raw seconds (Twitch) or doesn't exist (Rumble/Kick/no URL). function momentBaseFor(source: ShardSource, h: LinkableHit): string | null { return momentBaseUrl({ siteOrigin: h.siteUrl ?? source.publicOrigin(), slug: h.slug, webpageUrl: h.webpageUrl, platform: h.platform, }); } // Footer note stating the expansion rule for link_style:"base" output. const BASE_EXPANSION_NOTE = "link_style base: full moment link = `` — append the " + "integer after the `|` to that video's moment_base, e.g. `[title @ 2:36]" + "(156)`; a video with no moment_base line → cite its source " + "url plain"; type ToolResult = { content: { type: "text"; text: string }[]; isError?: boolean; }; function text(s: string): ToolResult { return { content: [{ type: "text", text: s }] }; } function errorText(s: string): ToolResult { return { content: [{ type: "text", text: s }], isError: true }; } // Cache hints for the 2026-07-28 revision's cacheable results (`ttlMs` / // `cacheScope`). Our three cacheable operations — the tool list, the prompt // list, and the `server/discover` descriptor the SDK builds from them — are // literal constants in this file: identical for every caller, unchanged for the // life of the process, and carrying no corpus data (the corpus is chosen // per-call now, not per-connection). So they are safely `public` and worth an // hour. Everything else — every read tool's result — stays uncached by default. // Invalid values throw a RangeError in the Server constructor, so a typo here // fails at startup rather than on the wire. const CACHE_HINTS: NonNullable = { "tools/list": { ttlMs: 3_600_000, cacheScope: "public" }, "prompts/list": { ttlMs: 3_600_000, cacheScope: "public" }, "server/discover": { ttlMs: 3_600_000, cacheScope: "public" }, }; // The `source` argument every read tool accepts. One shared constant spliced // into each literal schema, so adding a tool can't accidentally omit it and // `additionalProperties: false` still holds everywhere. const SOURCE_ARG = { source: { type: "string", description: "Which corpus to read, as a handle: 'default' (this server's startup " + "corpus — the same as omitting it), 'local:', 'remote:', " + "'hub:', or 'hub:#' for a hub subset. A " + "bare site URL, or a hub member's siteId/title, also works and is " + "normalised. There is no active source and nothing persists between " + "calls: every call reads exactly what it asks for, and every result " + "ends with the canonical handle it actually read, as '(corpus: …)'.", }, } as const; // The filter/scope arguments shared by `search_transcripts` and // `enumerate_matches`. ONE constant spliced into both, for the same reason // SOURCE_ARG is: the two tools answer the same question at different // verbosities, so a filter reachable from one and not the other would make them // disagree about coverage — which is precisely the failure the stateless // rebuild set out to make impossible. // // Flat and orthogonal on purpose. Arbitrary boolean trees stay `open_link`'s // job: a share link already encodes one losslessly, and asking a model to // author a `qt=` tree in a tool call trades a real capability for a new class // of malformed input. const SEARCH_FILTER_ARGS = { states: { type: "array", items: { type: "string", enum: [ "available", "maybe_missing", "deleted", "private", "members_only", "unlisted", ], }, description: "Keep only videos in these presence states on the source platform. " + "This is how you ask what the DELETED videos said: " + "states:['deleted','private','members_only','unlisted','maybe_missing'] " + "searches everything that has since left the platform. Omitted = every " + "state. ('maybe_missing' = fell out of the channel listing but was never " + "individually confirmed.)", }, tags: { type: "array", items: { type: "string" }, description: "Keep only videos carrying one of these CURATED TAGS — the operator's " + "cross-channel vocabulary (e.g. tags:['eva-collab'] is 'every stream " + "where she is on mic', on any channel, including channels that are not " + "hers). ORed: naming two tags keeps a video with EITHER. Call " + "`list_tags` for the ids this source publishes and how many videos each " + "one has — do not guess an id. NOT the yt-dlp keywords searched by " + "scopes:['tags']; those are metadata from the uploader, these are " + "curation. A source that publishes no tags says so in the footer rather " + "than returning a quiet zero.", }, date_from: { type: "string", description: "Keep only videos uploaded on or after this date (YYYYMMDD or YYYY-MM-DD; anything else is an error).", }, date_to: { type: "string", description: "Keep only videos uploaded on or before this date (YYYYMMDD or YYYY-MM-DD; anything else is an error).", }, media_type: { type: "string", enum: ["video", "livestream"], description: "Keep only regular videos, or only livestream VODs. Omitted = both.", }, age: { type: "string", enum: ["all_ages", "restricted"], description: "Keep only all-ages or only age-restricted videos. Omitted = both.", }, exclude: { type: "array", items: { type: "string" }, description: "Drop any video that also contains one of these terms — video-level " + "NOT. This is how you get 'cup' but not 'world cup'. Matched the same " + "way the query is (regex too, when regex is set) but never expanded " + "through aliases, so an exclusion stays exactly as narrow as you wrote it.", }, collapse_duplicates: { type: "boolean", description: "Count a recording that exists on more than one platform ONCE (default " + "true). The corpus mirrors some videos across channels/platforms, so " + "without this a sweep counts the same recording twice. The collapsed " + "copies are still NAMED on the row they fold into, so nothing is hidden " + "— set false to get one row per upload instead.", }, scopes: { type: "array", items: { type: "string", enum: [ "transcripts", "chat", "description", "tags", "metadata", "posts", ], }, description: "Which layers to search. Omitted = spoken captions plus the title (the " + "default, and what you almost always want). Naming scopes REPLACES that " + "default: scopes:['description'] searches descriptions and not captions. " + "'metadata' is title + channel name. NOTE 'chat' reads the live-chat " + "shards, which are a separate ~1.2 GB corpus fetched lazily per video — " + "it is much slower than a caption search, so scope it to a channel or a " + "date range.", }, }; // The advertised tool list. Deliberately a hand-written literal array in a // fixed order — never generated from a Map or Object.keys — so `tools/list` is // byte-stable across processes and safely cacheable (see CACHE_HINTS). export const TOOLS: Tool[] = [ { name: "list_channels", description: "List the channels in this transcript archive, organized under their " + "channel groups (in hub mode, across every federated member site). Returns " + "each channel's display name, slug, video count, and owning site, plus a " + "compact list of the groups (id · name · channel count) and whether each " + "is selected by default — so you can pick a group or channels to scope a " + "search or sweep to.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, refresh: { type: "boolean", description: "Re-read the channel list instead of using this process's cached " + "copy (default false). Only needed if the corpus was rebuilt while " + "the server was running.", }, }, additionalProperties: false, }, }, { name: "list_tags", description: "List the CURATED TAGS this archive publishes — the operator's " + "cross-channel vocabulary for marking individual videos (e.g. " + "'eva-collab' = she is on mic), which is what `tags` on " + "search_transcripts / enumerate_matches filters by. Returns each tag's " + "id, label, group, how many videos carry it ON THIS SOURCE, and the " + "per-channel breakdown — so a tag can be scoped, counted and cited " + "without guessing an id. These are NOT the yt-dlp keywords searched by " + "scopes:['tags']: those come from the uploader's metadata, these are " + "curation applied after the fact and cut across channels. A source that " + "publishes none (nothing curated yet, or a site built before corpus " + "spec 4) says so plainly rather than returning an empty list that reads " + "like an answer.", inputSchema: { type: "object", properties: { ...SOURCE_ARG }, additionalProperties: false, }, }, { name: "list_reports", description: "List the cited reports a SITE publishes (corpus spec 5): each report's " + "id, title, kind, claim and citation counts, a fact-check's verdict " + "tally, and its page. A cited-only site publishes nothing else — no " + "channels or transcripts to search. A hub has none of its own; pass a " + "member's remote: handle.", inputSchema: { type: "object", properties: { ...SOURCE_ARG }, additionalProperties: false, }, }, { name: "get_report", description: "Read one published report: title, kind, verdict tally, its timeline " + "(dated entries, newest first) when it keeps one, then each " + "section's claims (verdict, findings) with every citation's verbatim " + "quote, original URL (the platform at the cited second, the post, the " + "document) and the site's moment page. Ids from list_reports.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, report: { type: "string", description: "The report id." }, section: { type: "string", description: "Optional: one section's (or timeline entry's) id, to read just that part.", }, }, required: ["report"], additionalProperties: false, }, }, { name: "search_transcripts", description: "Search the archive for a term or phrase. The corpus holds video " + "transcripts AND social posts (X/Twitter, Bluesky, forum threads) from the same " + "commentators; by default BOTH are searched and returned in one result " + "set — use content_types to narrow. A post hit carries no timestamps " + "(cite it as a bare source, never with @ mm:ss). Returns matching " + "videos with timestamped snippets. Substring match by default; set regex " + "to true for a case-insensitive regular expression. Scope is optional and " + "additive: restrict to one or more channels (channel / channels, by slug " + "or name) and/or one or more channel groups (group / groups, by group id " + "or name) — the search runs over the UNION, and with no scope it covers " + "the whole corpus. Alias-aware: a plain query with a curated search alias " + "auto-expands to the alias regex (e.g. 'k cups' also matches 'cake cup') " + "and the footer reports which aliases fired. Pageable: returns the total " + "match count and whether more pages exist, so a caller can enumerate a " + "query's full match set with offset. Set include_snippets to false for a " + "cheap worklist (no cue text). The footer names the resolved scope and " + "flags any channel/group token that matched nothing. Optional filters — " + "states, date_from/date_to, media_type, age, exclude — and scopes beyond " + "captions; with none given, behaviour is exactly as before, and a " + "selective filter also makes the search dramatically faster by reading " + "only the shard pages that can hold a match.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, ...SEARCH_FILTER_ARGS, query: { type: "string", description: "Term, phrase, or regex to find." }, channels: { type: "array", items: { type: "string" }, description: "Optional list of channel slugs/names to restrict the search to " + "(union with groups). A single channel is just a one-element list.", }, groups: { type: "array", items: { type: "string" }, description: "Optional list of channel groups (ids or names) to expand to their " + "member channels (union with channels). A single group is just a " + "one-element list.", }, regex: { type: "boolean", description: "Treat query as a case-insensitive regex (default false). " + "Disables alias expansion — the pattern is used verbatim.", }, limit: { type: "number", description: "Max matching videos to return in this page (default 20).", }, offset: { type: "number", description: "Skip this many matches before the page (default 0). Page with " + "offset += limit until has_more is false to walk the full set.", }, include_snippets: { type: "boolean", description: "Include timestamped snippet lines per video (default true). Set " + "false for a compact worklist (id/title/channel/date/match count).", }, use_aliases: { type: "boolean", description: "Expand the query via the site's curated search aliases (default " + "true; ignored when regex is true).", }, max_pages: { type: "number", description: "Max shard pages to scan before stopping (default 400). Reaching " + "it marks coverage partial.", }, content_types: { type: "array", items: { type: "string", enum: ["video", "post"] }, description: "Which corpora to search: 'video' (transcripts) and/or 'post' " + "(archived social posts). Defaults to BOTH. Posts have no " + "timeline, so their hits carry no timestamps.", }, link_style: { type: "string", enum: ["inline", "base"], description: "Timestamp link form (default 'inline': every [m:ss] is a full " + "Markdown moment link). 'base' is a compact agent-pipeline form: " + "each video gets one '- moment_base:' header line (a URL ending " + "in 't=') and snippet stamps become [m:ss|]; expand to a " + "full link by appending the integer seconds to the moment_base " + "([title @ m:ss]()). A video with no " + "buildable base omits the line — cite its source url plain.", }, }, required: ["query"], additionalProperties: false, }, }, { name: "enumerate_matches", description: "Get a query's COMPLETE match set as a worklist, in one call — every " + "matching id/title/channel/date, no snippets, plus the batch count for a " + "sweep. Use this, not paged search_transcripts, whenever you need to " + "cover or count everything: the engine materialises the whole match set " + "before slicing, so enumerating is ONE scan where paging the same query " + "would be one full scan per page. If a cap is hit the FIRST line says " + "coverage is partial and the list is a sample — a result without that " + "line is the complete set, and is the only basis on which you may state " + "a total or claim full coverage. Same scoping, filters, exclusions, " + "scopes and alias expansion as search_transcripts — the two always cover " + "exactly the same set, so a count from here matches what that returns.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, ...SEARCH_FILTER_ARGS, query: { type: "string", description: "Term, phrase, or regex to find." }, channels: { type: "array", items: { type: "string" }, description: "Optional channel slugs/names to restrict the scan to.", }, groups: { type: "array", items: { type: "string" }, description: "Optional channel groups (ids or names) to restrict to.", }, regex: { type: "boolean", description: "Treat query as a case-insensitive regex (default false). " + "Disables alias expansion.", }, use_aliases: { type: "boolean", description: "Expand the query via curated aliases (default true; ignored with " + "regex).", }, content_types: { type: "array", items: { type: "string", enum: ["video", "post"] }, description: "Which corpora to enumerate: 'video' and/or 'post'. Defaults to " + "BOTH.", }, batch_size: { type: "number", description: "Videos per batch, used only to report the batch count (default " + "8).", }, max_pages: { type: "number", description: "Max shard pages to scan before stopping (default 400). Reaching " + "it makes coverage partial, which is reported loudly.", }, }, required: ["query"], additionalProperties: false, }, }, { name: "get_transcript", description: "Fetch one video's full transcript as clean markdown (metadata header + " + "timestamped captions). Provide the video id; optionally the channel to " + "skip the cross-channel lookup. A video not yet in the archive (imported " + "or transcribed since the last build) is read off the local editor's " + "disk when one is configured (ARCHILYZER_EDITOR_URL + WORKER_TOKEN), " + "marked as such, with no moment links.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, video_id: { type: "string", description: "The video id." }, channel: { type: "string", description: "Optional owning channel slug/name." }, timestamps: { type: "boolean", description: "Prefix each caption line with a timestamp (default true).", }, track: { type: "string", description: "Optional: read one of the video's other English tracks instead of its " + "primary transcript (e.g. \"en\" for the uploaded captions beside the " + "original-audio \"en-orig\"). The header lists the tracks a video has; " + "only tracks whose words differ from the primary are kept. Omit for the primary.", }, }, required: ["video_id"], additionalProperties: false, }, }, { name: "get_post", description: "Fetch one archived social post by id: its full text, author, timestamp, " + "outbound links and permalink. Posts have no timeline — cite them as a " + "bare source with no @ mm:ss.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, post_id: { type: "string", description: "The post id (tweet id / atproto rkey / forum post id)." }, channel: { type: "string", description: "Optional owning channel slug/name to skip the lookup.", }, }, required: ["post_id"], additionalProperties: false, }, }, { name: "get_thread", description: "Fetch the whole thread a post belongs to (its root and every archived " + "reply), oldest first. For a forum post it is the post's conversation " + "instead: the posts it quotes and the posts quoting it, in thread order. " + "This is the post-corpus analogue of reading the transcript around a " + "cited moment.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, post_id: { type: "string", description: "Any post id in the thread (root or a reply).", }, channel: { type: "string", description: "Optional owning channel slug/name to skip the lookup.", }, }, required: ["post_id"], additionalProperties: false, }, }, { name: "get_transcripts", description: "Batch-read up to 20 videos' transcripts in one call. With a query, each " + "transcript is reduced to bounded, timestamped excerpt windows around the " + "matching lines (alias-aware, like search_transcripts) — high-signal " + "context for folding a batch into a report. Without a query, each video's " + "full transcript is returned as markdown. Missing ids are reported inline. " + "This is the batching workhorse for a sweep.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, video_ids: { type: "array", items: { type: "string" }, description: "The video ids to read (max 20; extras are dropped).", }, query: { type: "string", description: "Optional term/phrase/regex to window around. When given, only " + "excerpts around matches are returned instead of full transcripts.", }, queries: { type: "array", items: { type: "string" }, description: "Several terms to window around in ONE pass (max 8; unioned with " + "query). Windows for all of them merge per video, and each " + "video's header reports the per-query match count — so a query " + "that matched nothing is visible. Use this instead of re-reading " + "the same videos once per term.", }, regex: { type: "boolean", description: "Treat query as a case-insensitive regex (default false). Disables " + "alias expansion.", }, use_aliases: { type: "boolean", description: "Expand query via curated aliases (default true; ignored with regex).", }, channel: { type: "string", description: "Optional owning channel slug/name to speed the lookup.", }, channels: { type: "array", items: { type: "string" }, description: "Optional owning channel slugs/names to speed the lookup when the " + "batch spans several channels (ids are already scoped by search).", }, timestamps: { type: "boolean", description: "Prefix each line with a timestamp (default true).", }, before: { type: "number", description: "Seconds of context before each match (default 30).", }, after: { type: "number", description: "Seconds of context after each match (default 30).", }, content_types: { type: "array", items: { type: "string", enum: ["video", "post"] }, description: "Which corpora to search: 'video' (transcripts) and/or 'post' " + "(archived social posts). Defaults to BOTH. Posts have no " + "timeline, so their hits carry no timestamps.", }, link_style: { type: "string", enum: ["inline", "base"], description: "Timestamp link form (default 'inline': every [m:ss] is a full " + "Markdown moment link). 'base' is a compact agent-pipeline form: " + "each video gets one '- moment_base:' header line (a URL ending " + "in 't=') and line stamps become [m:ss|]; expand to a " + "full link by appending the integer seconds to the moment_base " + "([title @ m:ss]()). A video with no " + "buildable base omits the line — cite its source url plain.", }, max_lines: { type: "number", description: "Cap on merged excerpt lines per video when a query is given " + "(default 200; the earliest lines are kept). Ignored without a " + "query.", }, }, required: ["video_ids"], additionalProperties: false, }, }, { name: "get_video_metadata", description: "Everything the archive knows about one video, without the transcript " + "body: title, channel, upload date, duration, description, tags and " + "source URL, plus — where the site ships them — view/like/comment " + "counts, cue count, platform state, and TRANSCRIPT COVERAGE (a warning " + "when the transcript covers only part of the runtime, which means " + "quotes from the tail are missing and absence of evidence is not " + "evidence of absence). Also lists any other archived copies of the same " + "recording, and says explicitly whether their timings are aligned — if " + "they are not, never map a timestamp between copies. Shows AI chapters " + "and topic tags when a digest exists, flagging one borrowed from a " + "duplicate as describing the other upload.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, video_id: { type: "string", description: "The video id." }, channel: { type: "string", description: "Optional owning channel slug/name." }, }, required: ["video_id"], additionalProperties: false, }, }, { name: "fetch_clip", description: "Get the media behind a cited moment — by asking the local Archilyzer " + "editor, which fetches it through its own paced, cookie-aware, " + "provenanced job (the per-platform sleeps, the rate-limit cooldown, a " + "note beside the file saying who asked and why). NEVER run yt-dlp " + "yourself instead. Give the citation's channel slug, video id, start " + "and end, and a one-line reason; the window is padded (pad, default 3 " + "s) and may be at most 15 min. full: true fetches the whole recording " + "instead, into the editor's saved-video store. maxHeight caps the " + "source height (a window defaults to 720; a whole recording at or under " + "720 is saved as 720p H.264, above it at the original quality); a " + "cached file is served as it is, and the answer gives its height. The " + "answer names the " + "file on disk: a read-only corpus artifact to play or copy, never to " + "move, edit or delete. The call waits up to wait_seconds; if the fetch " + "is still running it returns the job id — call again with job to keep " + "waiting. It sends a progress notification per poll when the client asks " + "for progress. A client with a 60 s default request timeout must raise " + "it or pass wait_seconds ≤ 50 — the fetch continues on the editor either " + "way; resume it with job, and once it has finished the same request " + "finds it cached. The editor must already archive " + "the cited channel. Needs ARCHILYZER_EDITOR_URL and WORKER_TOKEN in this server's " + "environment; without them it says so and fetches nothing.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, channel: { type: "string", description: "The channel slug, as the citation names it.", }, video: { type: "string", description: "The archive's video id, as cited. Pass the corpus the citation " + "came from as `source`, so a Rumble embed id resolves to the " + "editor's directory.", }, start: { oneOf: [{ type: "number" }, { type: "string" }], description: "Start of the cited span: seconds, or mm:ss / h:mm:ss.", }, end: { oneOf: [{ type: "number" }, { type: "string" }], description: "End of the cited span: seconds, or mm:ss / h:mm:ss.", }, pad: { type: "number", description: "Seconds added before start and after end (default 3). The padded " + "window is what is fetched, and it may be at most 900 s.", }, full: { type: "boolean", description: "The whole recording instead of a window (default false) — for a " + "video that must be re-cut freely or watched end to end. It lands " + "in the editor's saved-video store, is much larger than a window, " + "and needs a video the editor already knows (its metadata or " + "playlist entry). start, end and pad are ignored.", }, maxHeight: { type: "integer", minimum: 144, maximum: 2160, description: "Optional: the tallest source video to fetch, in pixels (144–2160). " + "A window is fetched at or under it (default 720, H.264 preferred). " + "With full: true, 720 or less saves the 720p H.264 preset (480p, " + "then whatever the source has, when it has no 720p H.264) and more " + "than 720 saves the original; omitted, the channel's own " + "source-video quality applies. Nothing is re-fetched for it: a " + "file already on disk is returned as it is, and the answer gives " + "its height, so a taller one can be seen.", }, reason: { type: "string", description: "Required: one line saying why these seconds are needed. Stored " + "beside the file (at most 400 characters are kept).", }, report: { type: "string", description: "Optional: the report or manifest this clip is for, recorded with " + "the file.", }, wait_seconds: { type: "number", description: "How long to wait for the fetch before returning its job id " + "(default 90, max 300). Nothing is lost when it runs out.", }, job: { type: "string", description: "Resume: the job id an earlier call returned. When set, every " + "other argument is ignored and the call just waits on that job.", }, }, required: [], additionalProperties: false, }, }, // Archival writes through the editor (release 19 A9): archivalTools.ts. ...(ARCHIVAL_TOOLS as unknown as Tool[]), { name: "list_sources", description: "Show the corpora this server can read: its DEFAULT corpus (the one used " + "when a call omits `source`) and, when that default is a hub, the hub's " + "member sites (siteId · title · url) as ready-to-paste handles. There is " + "no 'active' source to change — every read tool takes its own `source` " + "handle, and a call that omits it reads the default. Strictly read-only.", inputSchema: { type: "object", properties: { ...SOURCE_ARG }, additionalProperties: false, }, }, { name: "resolve_source", description: "Turn a corpus reference into the canonical handle to pass as `source`, " + "and check it can actually be read. Accepts a handle, a bare site or hub " + "URL (auto-detected), or a hub member's siteId/title. Returns the " + "canonical handle plus a channel/group count. It changes NOTHING: there " + "is no active source, so this only tells you what to pass — every read " + "tool still needs the handle in its own `source` argument.", inputSchema: { type: "object", properties: { ...SOURCE_ARG }, required: ["source"], additionalProperties: false, }, }, { name: "open_link", description: "Paste an archilyzer viewer **share link** (origin + a `qt=` query tree + " + "`fc/ft/fa/fav/fdf/fdt` filters) to re-run that exact search here — at " + "full fidelity, with clickable timestamped result links. ONE call does " + "the whole job: it decodes the link, resolves its origin to a source " + "handle (hub or single-site, auto-detected), validates the channel scope " + "against that live corpus, and returns the plan, the first page of " + "results, and the handle — pass that handle as `source` on the follow-up " + "calls. Adjust in natural language by re-calling with `overrides` (e.g. " + "clear_availability to drop the availability filter, channels to " + "re-scope, query to replace the search). Set dry_run:true to see the plan " + "without searching. Read-only: nothing about this server changes.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, link: { type: "string", description: "The archilyzer viewer share URL to decode and run.", }, dry_run: { type: "boolean", description: "true returns the decoded plan WITHOUT running the search — for " + "confirming a link's scope and filters first. Default false: " + "decode and search in one call.", }, overrides: { type: "object", description: "Structured edits to the decoded search, so a natural-language " + "adjustment can be honored by re-calling with the changed facet.", properties: { clear_filters: { type: "boolean", description: "Drop all filters." }, clear_availability: { type: "boolean", description: "Remove the availability filter, so videos in every state " + "are kept (available plus the missing ones: unconfirmed, " + "deleted, private, members-only, unlisted).", }, clear_type: { type: "boolean", description: "Remove the video/livestream type filter.", }, clear_age: { type: "boolean", description: "Remove the all-ages/age-restricted filter.", }, clear_dates: { type: "boolean", description: "Remove the upload-date range filter.", }, clear_channels: { type: "boolean", description: "Widen the channel scope to the whole corpus.", }, channels: { type: "array", items: { type: "string" }, description: "Replace the channel scope with these channel names.", }, date_from: { type: "string", description: "Set the inclusive lower upload-date bound (YYYYMMDD).", }, date_to: { type: "string", description: "Set the inclusive upper upload-date bound (YYYYMMDD).", }, query: { type: "string", description: "Replace the whole query with this single term.", }, regex: { type: "boolean", description: "Treat the override query as a regex.", }, query_scope: { type: "string", enum: [ "transcripts", "chat", "posts", "metadata", "description", "tags", ], description: "Scope for the override query (default transcripts). 'posts' " + "targets the social-post corpus.", }, }, additionalProperties: false, }, limit: { type: "number", description: "Results per page (default 20).", }, offset: { type: "number", description: "Skip this many matches (default 0).", }, }, required: ["link"], additionalProperties: false, }, }, { name: "sweep_plan", description: "Turn a request written in plain English — optionally with a pasted " + "share link — into the step-by-step plan for a corpus sweep, with the " + "scope, corpus handle and batch count already resolved. Call this when " + "the user asks to sweep, survey, or systematically go through the " + "archive. It only RETURNS instructions: it reads nothing, writes " + "nothing, and starts nothing. Following the plan it returns is a long " + "job that writes a report file with your own Write/Edit tools, so run " + "it when that is what was asked for, and show the plan's ⚠ warnings.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, request: { type: "string", description: "The whole request, verbatim and unsplit: a share link and/or " + "what to sweep for, in the user's own words, punctuation intact. " + "Recognised settings may be appended as key=value — channels=, " + "groups=, batch_size=, parse_model=, report=, directive=, " + "source=, content_types=, regex= (quote a multi-word value). " + "Anything else stays part of the question.", }, }, required: ["request"], additionalProperties: false, }, }, { name: "ask_plan", description: "Turn a plain-English question about the archive — optionally with a " + "pasted share link — into a step-by-step evidence plan: which searches " + "to run, against which corpus handle, and how to cite the answer. Use " + "it for a question to be answered in conversation (sweep_plan is for a " + "systematic pass that writes a report). It only RETURNS instructions: " + "it reads nothing and writes nothing.", inputSchema: { type: "object", properties: { ...SOURCE_ARG, request: { type: "string", description: "The whole question, verbatim and unsplit, punctuation intact, " + "plus any share link. The same key=value settings as sweep_plan " + "are recognised; anything else stays part of the question.", }, }, required: ["request"], additionalProperties: false, }, }, ]; // Append the corpus actually read to a result, so no answer can be silently // about the wrong archive. Deliberately `(corpus: …)` and not `source:` — // `- source:` already means "this video's URL" throughout the output. // // This lives in the dispatch wrapper, applied to every result including // errors: a per-handler echo is one that a new tool forgets. function withCorpus(result: ToolResult, resolved: ResolvedSource): ToolResult { const note = [`corpus: ${resolved.handle}`, ...resolved.notes].join("; "); const content = [...result.content]; const last = content[content.length - 1]; if (last && last.type === "text") { content[content.length - 1] = { ...last, text: `${last.text}\n\n(${note})` }; } else { content.push({ type: "text", text: `(${note})` }); } return { ...result, content }; } // The real world for fetch_clip. `env` is process.env itself, so a token set // after startup is still seen. const DEFAULT_FETCH_CLIP_DEPS: FetchClipDeps = { env: process.env, fetch: (url, init) => globalThis.fetch(url, init), sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)), now: () => Date.now(), }; // fetch_clip: validate, map the cited id to the editor's directory id, ask the // editor, and render its answer. The one tool that causes a write — and the // editor makes it, not this process. // // THE RUMBLE TRAP. A published record's `id` is yt-dlp's native id, which on // Rumble is the EMBED id (`vxe1ae`); the editor's directory is the canonical id // from the URL slug (`v1007ay`). Passing the embed id would make the editor // create a NEW data/vxe1ae/ and fetch rumble.com/vxe1ae. The record's // `webpageUrl` is the slug URL, so the canonical id comes from it — the same // `extractVideoId` the editor names its directories with. On YouTube the two // are equal. A video the corpus does not hold goes through as cited, with a // note saying what that risks. async function handleFetchClip( source: ShardSource, resolved: ResolvedSource, args: Record, deps: FetchClipDeps, onPoll?: (p: PollProgress) => Promise, ): Promise { // No editor, no fetch — said first, before a corpus read that could only be // wasted. if (!editorFromEnv(deps.env)) return errorText(NO_EDITOR_TEXT); const v = validateFetchClipArgs(args); if (!v.ok) return errorText(v.error); const request = v.request; let note = ""; let ctx: { channel?: string; video?: string } = {}; if ("target" in request) { const target = request.target; const found = await findVideo(source, target.video, target.channel); if (found) { const webpageUrl = found.record.webpageUrl; const mapped = webpageUrl ? extractVideoId(webpageUrl) : null; target.video = mapped && isVideoId(mapped) ? mapped : target.video; // The channel the record lives in, which is the editor's directory even // when the citation spelled it by name or in another case. target.channel = found.ch.slug; if (target.kind === "window" && webpageUrl) target.webpageUrl = webpageUrl; } else { note = `${notFoundNote(target.video, resolved.handle)}\n\n`; } ctx = { channel: target.channel, video: target.video }; } const outcome = await fetchClip(request, deps, { onPoll }); const rendered = renderFetchClip(outcome, ctx); const body = `${note}${rendered.text}`; return rendered.isError ? errorText(body) : text(body); } // An archival tool's answer (archivalTools.ts) as a tool result. function fromAnswer(a: ToolAnswer): ToolResult { return a.isError ? errorText(a.text) : text(a.text); } // One MCP progress notification per poll while fetch_clip waits, when the // client asked for progress (a `progressToken` in the request's _meta). A // client whose request timeout resets on progress then keeps waiting for as // long as the editor keeps answering. `progress` is the poll count, so it only // ever increases; there is no total, because a queue's length is not knowable // from here. No token, no notifications. function progressNotifier( progressToken: unknown, notify: (n: { method: "notifications/progress"; params: { progressToken: string | number; progress: number; message: string }; }) => Promise, ): ((p: PollProgress) => Promise) | undefined { if (typeof progressToken !== "string" && typeof progressToken !== "number") { return undefined; } return (p) => notify({ method: "notifications/progress", params: { progressToken, progress: p.polls, message: `editor job ${p.jobId}: ${p.status}, ${p.waited}s waited`, }, }); } // Build a configured MCP server over a data source or a SourceRegistry. The // core read tools work for local / remote / hub sources — only the ShardSource // differs — and which one a call reads is decided per call by its `source` // argument, resolved through the registry. A bare ShardSource is wrapped in a // one-source registry, so `createServer(someSource)` still works. export function createServer( sourceOrRegistry: ShardSource | SourceRegistry, opts: { fetchClipDeps?: Partial } = {}, ): Server { // fetch_clip's world — env, HTTP, the clock — injected so a test drives it // with no network and no timers. Read per call, never cached: the env is the // live process.env unless a test says otherwise. const fetchClipDeps: FetchClipDeps = { ...DEFAULT_FETCH_CLIP_DEPS, ...opts.fetchClipDeps, }; const registry = sourceOrRegistry instanceof SourceRegistry ? sourceOrRegistry : SourceRegistry.forSource(sourceOrRegistry); const server = new Server( { name: "yt-dlp-transcript-mcp", version: "0.1.0" }, { capabilities: { tools: {}, prompts: {} }, cacheHints: CACHE_HINTS }, ); server.setRequestHandler("tools/list", async () => ({ tools: TOOLS })); server.setRequestHandler("prompts/list", async () => ({ prompts: PROMPTS, })); server.setRequestHandler("prompts/get", async (req) => { if (req.params.name !== "sweep") { throw new Error(`unknown prompt: ${req.params.name}`); } return buildSweepPrompt((req.params.arguments ?? {}) as Record); }); server.setRequestHandler("tools/call", async (req, ctx) => { const name = req.params.name; const args = (req.params.arguments ?? {}) as Record; // Resolving the handle is the FIRST thing every call does, so a typo'd or // unreachable corpus fails as itself rather than as an empty result set. let resolved: ResolvedSource; try { resolved = await registry.resolve(args.source); } catch (e) { return errorText(`${name}: ${(e as Error).message}`); } const source = resolved.source; // Opt-in per-call I/O accounting for mcp/bench (MCP_IO_STATS=1). Written to // stderr as one JSON line — stdout is the JSON-RPC channel and must stay // clean. Off by default; this is measurement scaffolding, not telemetry. const ioBefore = ioStatsEnabled() ? ioStatsSnapshot() : null; const startedAt = ioBefore ? performance.now() : 0; try { const result = await (async (): Promise => { switch (name) { case "list_channels": return handleListChannels(source, args); case "list_tags": return handleListTags(source); case "list_reports": return handleListReports(source); case "get_report": return handleGetReport(source, args); case "search_transcripts": return handleSearch(source, args); case "enumerate_matches": return handleEnumerateMatches(source, args); case "get_transcript": return handleGetTranscript(source, args, fetchClipDeps); case "get_transcripts": return handleGetTranscripts(source, args); case "get_post": return handleGetPost(source, args); case "get_thread": return handleGetThread(source, args); case "get_video_metadata": return handleGetMetadata(source, args); case "fetch_clip": return handleFetchClip( source, resolved, args, fetchClipDeps, progressNotifier(ctx.mcpReq._meta?.progressToken, ctx.mcpReq.notify), ); case "get_job": return fromAnswer(await getJob(args, fetchClipDeps)); case "enqueue": return fromAnswer(await enqueue(args, fetchClipDeps)); case "channel_coverage": return fromAnswer(await channelCoverageTool(args, fetchClipDeps)); case "notes": return fromAnswer(await notesTool(args, fetchClipDeps)); case "list_sources": return handleListSources(registry, resolved); case "resolve_source": return handleResolveSource(resolved); // Unadvertised one-release alias. It no longer switches anything — // it resolves, and says to pass the handle per call instead. case "use_source": return handleLegacyUseSource(registry, args); case "open_link": return handleOpenLink(registry, resolved, args); case "sweep_plan": return handlePlan("sweep", registry, resolved, args); case "ask_plan": return handlePlan("ask", registry, resolved, args); default: return errorText(`unknown tool: ${name}`); } })(); reportIo(name, ioBefore, startedAt); return withCorpus(result, resolved); } catch (e) { reportIo(name, ioBefore, startedAt); return withCorpus( errorText(`${name} failed: ${(e as Error).message}`), resolved, ); } }); return server; } // Emit this call's shard-read deltas as one stderr JSON line, so the benchmark // can report the STRUCTURAL cost of a query — pages read, bytes parsed — beside // its wall time. That separation is the point: wall time on a shared box is // only meaningful when the box is idle, while read and byte counts are // properties of the query plan and hold under any load. function reportIo( tool: string, before: Record | null, startedAt: number, ): void { if (!before) return; const after = ioStatsSnapshot(); const delta: Record = {}; let reads = 0; let bytes = 0; for (const [kind, now] of Object.entries(after)) { const was = before[kind] ?? { reads: 0, bytes: 0 }; const d = { reads: now.reads - was.reads, bytes: now.bytes - was.bytes }; if (d.reads === 0 && d.bytes === 0) continue; delta[kind] = d; reads += d.reads; bytes += d.bytes; } console.error( `[io] ${JSON.stringify({ tool, ms: Math.round(performance.now() - startedAt), reads, bytes, byKind: delta, })}`, ); } function channelLine(c: ChannelRef): string { const parts = [`slug: ${c.slug}`]; if (c.videoCount != null) parts.push(`${c.videoCount} videos`); if (c.siteTitle) parts.push(`site: ${c.siteTitle}`); return ` - ${c.name} (${parts.join(", ")})`; } // List channels organized under their resolved group, pick-friendly: a header // per group (name || id, member count, selectedByDefault), the channels beneath // it, then a compact "Groups" line (id · name · N channels) so a human can name // a group selector for a scoped search/sweep. When the source has no groups // (absent manifest, or hub mode) every channel folds under the fallback group // and the Groups line is omitted. async function handleListChannels( source: ShardSource, args: Record, ): Promise { const channels = await source.listChannels({ refresh: args.refresh === true }); if (channels.length === 0) { // A cited-only site has no channels by design; say what it does publish. const line = reportsLine(await sourceReports(source)); return text( line ? `${source.label} — ${line}` : `No channels found in ${source.label}.`, ); } const { groups, defaultGroupId } = await source.loadGroups(); // Bucket channels by their resolved group id (browser semantics). const byGroup = new Map(); for (const c of channels) { const gid = groups.length > 0 ? resolveChannelGroupId(c.groupId, groups, defaultGroupId) : FALLBACK_GROUP.id; const bucket = byGroup.get(gid) ?? []; bucket.push(c); byGroup.set(gid, bucket); } // Render in group order, only including groups that actually have members. const ordered = groups.length > 0 ? sortGroups(groups) : [FALLBACK_GROUP]; const knownIds = new Set(ordered.map((g) => g.id)); const rendered: ChannelGroup[] = ordered.filter((g) => (byGroup.get(g.id)?.length ?? 0) > 0); // Any bucket whose id isn't a known group (shouldn't happen after resolve, but // be defensive) is appended under a synthetic header so no channel is dropped. const strayIds = [...byGroup.keys()].filter((id) => !knownIds.has(id)); const sections: string[] = []; for (const g of rendered) { const members = byGroup.get(g.id) ?? []; const header = `### ${g.name || g.id} (${members.length} channel(s)` + (g.selectedByDefault ? "" : "; not selected by default") + `)` + (g.description ? ` — ${g.description}` : ""); sections.push(`${header}\n${members.map(channelLine).join("\n")}`); } for (const id of strayIds) { const members = byGroup.get(id) ?? []; sections.push(`### ${id} (${members.length} channel(s))\n${members.map(channelLine).join("\n")}`); } // Compact selector cheat-sheet — only meaningful when real groups exist. const groupsLine = groups.length > 0 ? "\n\nGroups (select by id or name):\n" + sortGroups(groups) .map( (g) => `- ${g.id} · ${g.name || "(unnamed)"} · ${byGroup.get(g.id)?.length ?? 0} channel(s)` + (g.selectedByDefault ? "" : " · not selected by default"), ) .join("\n") : ""; return text( `${channels.length} channel(s) in ${source.label}:\n\n${sections.join("\n\n")}${groupsLine}`, ); } // The curated-tag vocabulary a source publishes, with its own counts. // // The empty case is answered in WORDS, not with an empty list. An agent that // gets `[]` back reads it as "no tags here" and moves on; what it actually // needs to know is that a tag filter against this source will match nothing and // why — the site publishes none, or it was built before the corpus spec that // carries them. That is the same failure the footer warning below exists to // stop, said once up front. async function handleListTags(source: ShardSource): Promise { const tags = await loadSourceTags(source); if (tags.length === 0) { return text( `${source.label} publishes no curated tags.\n\n` + `Either nothing has been tagged for this site yet, or it was built ` + `before curated tags existed (corpus spec 4) — /tags.json is absent. ` + `A \`tags\` filter against this source would match nothing, so do not ` + `use one here; scope with channel/group/date instead.`, ); } const groups = groupPublishedTags(tags); const sections = groups.map((g) => { const head = `### ${g.label || "(ungrouped)"}`; const lines = g.tags.map((t) => { const per = Object.entries(t.channels) .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])) .map(([slug, n]) => `${slug} ${n}`) .join(", "); return ( `- ${t.id} · ${t.label} · ${t.count} video(s)` + (per ? ` — ${per}` : "") ); }); return `${head}\n${lines.join("\n")}`; }); return text( `${tags.length} curated tag(s) in ${source.label}:\n\n` + `${sections.join("\n\n")}\n\n` + `(filter with tags:["${tags[0].id}"] on search_transcripts or ` + `enumerate_matches; several tags are ORed. A video may carry more than ` + `one tag, so these counts do not sum to a video count.)`, ); } // The reports a site publishes (spec 5). Read through sourceReports, so a hub // or a stub (no reports of its own) and a site with none each get a sentence. async function handleListReports(source: ShardSource): Promise { const reports = await sourceReports(source); return text(renderReportIndex(source.label, reports, source.publicOrigin())); } // A local source learns its site's origin from corpus.json with the channel // list, so read that first: a moment link is then absolute, as it is remotely. async function reportOrigin(source: ShardSource): Promise { await source.listChannels().catch(() => []); return source.publicOrigin(); } async function handleGetReport( source: ShardSource, args: Record, ): Promise { const id = typeof args.report === "string" ? args.report.trim() : ""; if (!id) return errorText("get_report: `report` (a report id) is required — list_reports names them."); if (typeof source.reportPage !== "function") { return errorText(renderReportIndex(source.label, await sourceReports(source), null)); } const view = await source.reportPage(id); if (!view) { const known = (await sourceReports(source)).reports.map((r) => r.id); return errorText( `report not found: ${id}` + (known.length > 0 ? ` — this source publishes: ${known.join(", ")}` : " — this source publishes no reports"), ); } const section = typeof args.section === "string" && args.section.trim() ? args.section.trim() : undefined; if (section && !view.sections.some((s) => s.id === section) && !(view.entries ?? []).some((e) => e.id === section)) { const entries = (view.entries ?? []).map((e) => e.id); return errorText( `report ${id} has no section ${section} — its sections: ${view.sections.map((s) => s.id).join(", ")}` + (entries.length > 0 ? `; its timeline entries: ${entries.join(", ")}` : ""), ); } return text(renderReportPage(view, await reportOrigin(source), section)); } // Every tag read goes through here so "this source cannot report tags at all" // (an in-memory stub, which has no such concept) and "this source publishes // none" land in the same place, as the same empty list. async function loadSourceTags(source: ShardSource): Promise { if (typeof source.loadTags !== "function") return []; try { return await source.loadTags(); } catch { return []; } } // A tag filter against a source that publishes no /tags.json matches nothing. // That is CORRECT — a record with no curated tags carries none — but a bare // zero is indistinguishable from "searched, found nothing", so it is named. // // The scan itself is unaffected and stays honest: the page planner only ever // PRUNES, and a video its index has never heard of gets its page read anyway // (search.ts's load-bearing invariant), so this is a reporting fix, never a // coverage one. async function describeTagCoverage( source: ShardSource, parsed: ParsedSearchArgs, ): Promise { const asked = parsed.filters?.curatedTags ?? []; if (asked.length === 0) return ""; const published = await loadSourceTags(source); if (published.length === 0) { return ( `⚠ this source publishes no /tags.json (nothing curated, or built ` + `before corpus spec 4) — a tag filter matches NOTHING here, so this ` + `result is not evidence of absence` ); } const known = new Set(published.map((t) => t.id)); const unknown = asked.filter((t) => !known.has(t)); if (unknown.length === 0) return ""; return ( `⚠ tag(s) this source does not publish: ${unknown.join(", ")} — they ` + `match nothing here; call list_tags for the ids it has` ); } async function handleSearch( source: ShardSource, args: Record, ): Promise { const query = String(args.query ?? "").trim(); if (!query) return errorText("query is required"); const includeSnippets = args.include_snippets !== false; const base = linkStyleOf(args) === "base"; const parsed = parseSearchArgs(args); if (parsed.error) return errorText(parsed.error); const result = await searchTranscripts(source, { query, filters: parsed.filters, exclude: parsed.exclude, scopes: parsed.scopes, collapseDuplicates: parsed.collapseDuplicates, channel: typeof args.channel === "string" ? args.channel : undefined, channels: strArray(args.channels), group: typeof args.group === "string" ? args.group : undefined, groups: strArray(args.groups), regex: args.regex === true, limit: typeof args.limit === "number" ? args.limit : undefined, offset: typeof args.offset === "number" ? args.offset : undefined, includeSnippets, useAliases: args.use_aliases !== false, maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined, contentTypes: parseContentTypes(args.content_types), }); const aliasNote = describeFiredAliases(result.firedAliases); const scopeNote = describeScope(result.selection); const postsNote = describePostsPass(result); const filterNote = describeSearchFilters(parsed); const tagNote = await describeTagCoverage(source, parsed); const prunedNote = describeCoverage(result); const dupNote = describeDuplicates(result); const rangeStart = result.total === 0 ? 0 : result.offset + 1; const rangeEnd = result.offset + result.hits.length; const footer = `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` + `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` + `page(s) across ${result.scanned.channels} channel(s)` + (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") + (prunedNote ? `; ${prunedNote}` : "") + (dupNote ? `; ${dupNote}` : "") + (scopeNote ? `; ${scopeNote}` : "") + (filterNote ? `; ${filterNote}` : "") + (postsNote ? `; ${postsNote}` : "") + (aliasNote ? `; ${aliasNote}` : "") + (tagNote ? `; ${tagNote}` : "") + parsed.warnings.map((w) => `; ⚠ ${w}`).join("") + ")"; // The warning goes ABOVE the hits, not only in the footer. The footer said // "has_more: yes" on the run that motivated this and it was read straight // past — a report then claimed 319 videos swept from a 200-hit page. const banner = incompletePageBanner(result, "search_transcripts"); if (result.hits.length === 0) { const head = result.total === 0 ? `No matches for "${query}".` : `No matches in this page (offset ${result.offset} is past the ${result.total} total).`; return text(`${banner}${head}${footer}`); } const blocks = result.hits.map((h) => { // A POST has no timeline: no moment link, no [m:ss] stamps. Render it as a // dated, authored block so the model can cite it as a bare source. if (h.contentType === "post") { const head = `### post by ${h.author ?? h.channelName}\n` + `- post_id: ${h.videoId} | channel: ${h.channelName}` + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + ` | posted: ${formatDate(h.uploadDate)} | platform: ${h.platform ?? "post"}` + (includeSnippets && h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); const body = h.snippets.map((sn) => ` - ${sn.text}`).join("\n"); return body ? `${head}\n${body}` : head; } // Worklist mode (include_snippets:false) trims the source line too — the // caller only wants ids/titles/counts, so no per-video URLs at all. const baseUrl = base && includeSnippets ? momentBaseFor(source, h) : null; const head = `### ${h.title}\n` + `- video_id: ${h.videoId} | channel: ${h.channelName}` + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + (includeSnippets && h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "") + (baseUrl ? `\n- moment_base: ${baseUrl}` : "") + (h.mirrors && h.mirrors.length > 0 ? `\n- ⧉ same recording also archived as: ` + h.mirrors.map((m) => `${m.videoId} (${m.channelName})`).join(", ") + ` — counted once` : ""); const snips = h.snippets .map((s) => { // A hit in a non-timed layer (description / tags / channel name) is // tagged with that layer and carries NO timestamp link — citing it as // "@ 0:00" would assert someone said it at the start of the video. if (s.scope && s.scope !== "transcripts" && s.seconds === 0) { return ` - [${s.scope}] ${s.text}`; } const stamp = base ? baseStamp(s.clock, s.seconds) : stampMarkup(source, h, s.clock, s.seconds); const tag = s.scope && s.scope !== "transcripts" ? `${s.scope} ` : s.track ? `${inTrackLabel(s.track)} ` : ""; return ` - [${tag}${stamp}] ${s.text}`; }) .join("\n"); return snips ? `${head}\n${snips}` : head; }); // Worklist mode has no stamps (or bases) to expand — skip the note too. const baseNote = base && includeSnippets ? `\n\n(${BASE_EXPANSION_NOTE})` : ""; const postCount = result.hits.filter((h) => h.contentType === "post").length; const label = postCount === 0 ? "video(s)" : postCount === result.hits.length ? "post(s)" : "result(s) (videos + posts)"; return text( `${banner}${result.total} ${label} matching "${query}":\n\n` + `${blocks.join("\n\n")}${footer}${baseNote}`, ); } // The unmissable header for a page that is not the whole match set. Rendered // before the hits so it cannot be scrolled past, and phrased as an instruction // because the failure mode it exists to stop is reporting a count from one // page. `enumerate_matches` is named because it is the cheap fix: the engine // already materialises the full set, so enumerating is ONE scan where paging // this query would be ceil(total/limit) full scans. function incompletePageBanner( result: { total: number; offset: number; limit: number; hasMore: boolean }, tool: string, ): string { if (!result.hasMore) return ""; const shown = `${result.offset + 1}–${Math.min(result.offset + result.limit, result.total)}`; return ( `⚠ INCOMPLETE PAGE — ${result.total} total, showing ${shown}. Do NOT ` + `report a count or "all of them" from this page. Call ` + `\`enumerate_matches\` for the full worklist in one scan` + (tool === "search_transcripts" ? "" : ` (this ${tool} page is a slice)`) + `.\n\n` ); } // A note distinguishing "the posts corpus was searched and matched nothing" // from "there is no posts corpus here at all". The second used to render as // `scanned 0 page(s) across 0 channel(s)`, which reads like nothing ran. function describePostsPass(result: SearchResult): string { const p = result.postsScanned; // Said out loud, because the alternative is a caller concluding from silence // that the posts were searched and matched nothing. if (p.skippedForTagFilter) { return ( "posts: skipped — a tag filter was given and posts carry no curated " + "tags (the export UI does the same)" ); } if (!p.requested) return ""; if (p.channels === 0) { return ( "posts: no channel in scope ships a posts index — the post corpus is " + "EMPTY here, not merely unmatched" ); } return `posts: scanned ${p.pages} page(s) across ${p.channels} posting channel(s)`; } // The full match set as a worklist, in a single scan. This exists because the // paging instruction was in the sweep prompt all along and got ignored — prose // cannot be the enforcement mechanism, so the cheap, correct thing is also the // one-call thing. A cap being hit is stated on the FIRST line, never only in a // footer. async function handleEnumerateMatches( source: ShardSource, args: Record, ): Promise { const query = String(args.query ?? "").trim(); if (!query) return errorText("query is required"); const batchSizeArg = typeof args.batch_size === "number" ? Math.floor(args.batch_size) : 8; const batchSize = batchSizeArg >= 1 ? batchSizeArg : 8; const parsed = parseSearchArgs(args); if (parsed.error) return errorText(parsed.error); const result = await searchTranscripts(source, { query, filters: parsed.filters, exclude: parsed.exclude, scopes: parsed.scopes, collapseDuplicates: parsed.collapseDuplicates, // The singulars stay parsed even though they are no longer advertised. channel: typeof args.channel === "string" ? args.channel : undefined, channels: strArray(args.channels), group: typeof args.group === "string" ? args.group : undefined, groups: strArray(args.groups), regex: args.regex === true, useAliases: args.use_aliases !== false, maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined, contentTypes: parseContentTypes(args.content_types), includeSnippets: false, offset: 0, // The whole set: searchTranscripts collects up to HARD_VIDEO_CAP and then // slices, so asking for every collected match costs nothing extra. limit: Number.MAX_SAFE_INTEGER, }); const partial = result.truncated; const capNote = describeCapCoverage(result); const head = partial ? `⚠ COVERAGE PARTIAL — the scan hit its page/video cap before the corpus ` + `was exhausted. The ${result.total} match(es) below are a SAMPLE, not ` + `the full set: the true total is higher. Narrow the scope (channels/` + `groups) or raise max_pages to enumerate exhaustively, and say so in any ` + `report built on this.\n` + // The cap cuts in channel order, so the sample is channel-biased, not // random. Say which channels went unread — that is what makes it fixable. (capNote ? `The sample is biased by CHANNEL ORDER, not random — ${capNote}.\n` : "") + `\n` : ""; const rows = result.hits.map((h) => { const kind = h.contentType === "post" ? "post" : "video"; const mirrors = h.mirrors && h.mirrors.length > 0 ? ` | ⧉ +${h.mirrors.length} mirror(s): ${h.mirrors .map((m) => m.videoId) .join(", ")}` : ""; return ( `- ${h.videoId} | ${kind} | ${h.channelName} | ` + `${formatDate(h.uploadDate)} | ${h.matches} match(es) | ${h.title}${mirrors}` ); }); const batches = Math.ceil(result.total / batchSize); const scopeNote = describeScope(result.selection); const aliasNote = describeFiredAliases(result.firedAliases); const postsNote = describePostsPass(result); const filterNote = describeSearchFilters(parsed); const tagNote = await describeTagCoverage(source, parsed); const prunedNote = describeCoverage(result); const dupNote = describeDuplicates(result); const footer = `\n\n(complete set: ${partial ? "NO — capped" : "yes"}; ` + `${result.total} match(es); ${batches} batch(es) of ${batchSize}; ` + `scanned ${result.scanned.pages} page(s) across ` + `${result.scanned.channels} channel(s)` + (prunedNote ? `; ${prunedNote}` : "") + (dupNote ? `; ${dupNote}` : "") + (scopeNote ? `; ${scopeNote}` : "") + (filterNote ? `; ${filterNote}` : "") + (postsNote ? `; ${postsNote}` : "") + (aliasNote ? `; ${aliasNote}` : "") + (tagNote ? `; ${tagNote}` : "") + parsed.warnings.map((w) => `; ⚠ ${w}`).join("") + ")"; if (result.total === 0) { return text(`${head}No matches for "${query}".${footer}`); } return text( `${head}${result.total} match(es) for "${query}" — ` + `${partial ? "a PARTIAL worklist" : "the complete worklist"}:\n\n` + `${rows.join("\n")}${footer}`, ); } // A short human note describing the curated aliases a search expanded through, // e.g. "expanded via alias K-Cups → `(k|cake)[ -]?cup`". Empty when none fired. function describeFiredAliases(fired: SearchAlias[]): string { if (fired.length === 0) return ""; const parts = fired.map((a) => `${a.label} → \`${a.suggestion}\``); return `expanded via alias ${parts.join(", ")}`; } // ─── the shared filter/scope parsing for both search tools ─── // The parsed form of SEARCH_FILTER_ARGS, plus any tokens that were not // understood. Unknown tokens are REPORTED rather than dropped silently: a // typo'd state ("removed" for "deleted") would otherwise widen the search back // to the whole corpus and look like a legitimate empty result. type ParsedSearchArgs = { filters: SearchFilters | null; exclude?: string[]; scopes?: LayerScope[]; collapseDuplicates: boolean; warnings: string[]; // Set when an argument must stop the call rather than be dropped (a date // bound that is not a date). Callers return it as an error and search nothing. error?: string; }; const SCOPE_TOKENS: ReadonlyArray = [ "transcripts", "chat", "description", "tags", "metadata", "posts", ]; function parseSearchArgs(args: Record): ParsedSearchArgs { const warnings: string[] = []; const rawStates = strArray(args.states); let states: VideoState[] | undefined; if (rawStates) { states = rawStates.filter((s): s is VideoState => isVideoState(s)); const bad = rawStates.filter((s) => !isVideoState(s)); if (bad.length > 0) { warnings.push( `unknown state(s) ignored: ${bad.join(", ")} (valid: ${VIDEO_STATES.join(", ")})`, ); } if (states.length === 0) { states = undefined; warnings.push("no valid states given — the state filter was NOT applied"); } } const mediaType = args.media_type === "video" || args.media_type === "livestream" ? args.media_type : undefined; if (args.media_type !== undefined && mediaType === undefined) { warnings.push(`unknown media_type ignored: ${String(args.media_type)}`); } const age = args.age === "all_ages" || args.age === "restricted" ? args.age : undefined; if (args.age !== undefined && age === undefined) { warnings.push(`unknown age ignored: ${String(args.age)}`); } // A date bound the caller cannot have meant is an ERROR, not a warning: an // ignored bound widens the search to the whole corpus and the count reads as // the bounded one (see dateArg.ts). The common spellings are normalised. const from = normalizeDateArg(args.date_from); const to = normalizeDateArg(args.date_to); const badDates = [ ["date_from", args.date_from, from], ["date_to", args.date_to, to], ] .filter(([, , v]) => v === null) .map(([name, raw]) => `${name}=${JSON.stringify(raw)}`); const error = badDates.length ? `not a date: ${badDates.join(", ")} — use YYYYMMDD or YYYY-MM-DD. Nothing was searched.` : undefined; // Curated tags. A malformed id is REPORTED, never silently dropped: dropping // the only tag in the list would widen the search back to the whole corpus // and read as a legitimate result, which is the same failure a typo'd state // would cause. const rawTags = strArray(args.tags); let tags: string[] | undefined; if (rawTags) { const normalized = rawTags.map((t) => t.trim().toLowerCase()); tags = normalized.filter((t) => isTagId(t)); const bad = normalized.filter((t) => !isTagId(t)); if (bad.length > 0) { warnings.push( `not a tag id, ignored: ${bad.join(", ")} (ids are lowercase, ` + `[a-z0-9._-]; call list_tags)`, ); } if (tags.length === 0) { tags = undefined; warnings.push("no valid tags given — the tag filter was NOT applied"); } } const anyFilter = states !== undefined || mediaType !== undefined || age !== undefined || Boolean(from) || Boolean(to) || tags !== undefined; const filters: SearchFilters | null = anyFilter ? { videos: mediaType === undefined || mediaType === "video", livestreams: mediaType === undefined || mediaType === "livestream", allAges: age === undefined || age === "all_ages", restricted: age === undefined || age === "restricted", states: new Set(states ?? VIDEO_STATES), ...(from ? { dateFrom: from } : {}), ...(to ? { dateTo: to } : {}), ...(tags ? { curatedTags: tags } : {}), } : null; const rawScopes = strArray(args.scopes); let scopes: LayerScope[] | undefined; if (rawScopes) { scopes = rawScopes.filter((s): s is LayerScope => (SCOPE_TOKENS as readonly string[]).includes(s), ); const bad = rawScopes.filter( (s) => !(SCOPE_TOKENS as readonly string[]).includes(s), ); if (bad.length > 0) { warnings.push(`unknown scope(s) ignored: ${bad.join(", ")}`); } if (scopes.length === 0) { scopes = undefined; warnings.push("no valid scopes given — searching captions + title"); } } return { filters, exclude: strArray(args.exclude), scopes, collapseDuplicates: args.collapse_duplicates !== false, warnings, ...(error ? { error } : {}), }; } // A human note for the footer describing the filters actually applied, so a // result can never be read as unfiltered when it wasn't. function describeSearchFilters(p: ParsedSearchArgs): string { const parts: string[] = []; const f = p.filters; if (f) { if (!VIDEO_STATES.every((s) => f.states.has(s))) { parts.push(`states: ${[...f.states].join(", ")}`); } if (!f.videos || !f.livestreams) { parts.push(`type: ${f.videos ? "videos" : "livestreams"} only`); } if (!f.allAges || !f.restricted) { parts.push(`audience: ${f.allAges ? "all-ages" : "age-restricted"} only`); } if (f.dateFrom || f.dateTo) { parts.push(`uploaded ${f.dateFrom ?? "…"}–${f.dateTo ?? "…"}`); } if (f.curatedTags && f.curatedTags.length > 0) { parts.push(`tags: ${f.curatedTags.join(" OR ")}`); } } if (p.exclude && p.exclude.length > 0) { parts.push(`excluding ${p.exclude.map((e) => `"${e}"`).join(", ")}`); } if (p.scopes) parts.push(`scopes: ${p.scopes.join(", ")}`); return parts.length > 0 ? `filters — ${parts.join("; ")}` : ""; } // Which channels a capped scan actually covered. // // A cap truncates in CHANNEL ITERATION ORDER, never at random, so "partial" on // its own hides the shape of the bias: the sample is whatever sorts first. The // measured case that motivated this returned exactly 2000 for a common term — // a number that reads like a count. Naming the channels that were finished, the // one it stopped inside, and the ones it never opened turns that into something // a caller can act on: re-run scoped to the remainder. function describeCapCoverage(result: SearchResult): string { const c = result.coverage; if (!result.truncated) return ""; const parts: string[] = []; if (c.channelsCompleted.length > 0) { parts.push( `fully scanned (${c.channelsCompleted.length}): ${c.channelsCompleted.join(", ")}`, ); } if (c.channelStopped) { parts.push( `stopped inside ${c.channelStopped.channel} at page ` + `${c.channelStopped.page}/${c.channelStopped.pages}`, ); } if (c.channelsNotReached.length > 0) { parts.push( `NEVER REACHED (${c.channelsNotReached.length}): ${c.channelsNotReached.join(", ")}`, ); } return parts.length > 0 ? parts.join("; ") : ""; } // What duplicate collapsing did, in the words a report should use. Stated // whenever it changed the count: a total that silently differs from the one a // previous run reported is worse than a total that explains itself. function describeDuplicates(result: SearchResult): string { const d = result.duplicates; if (!d.available || d.collapsed === 0) return ""; return ( `${d.collapsed} mirror(s) collapsed across ${d.clusters} cluster(s) — ` + `the ${result.total} above are distinct RECORDINGS, not uploads` ); } // How much of the corpus the scan actually had to open. Filter-first planning // means a filtered query reads a fraction of the pages, and that has to be // distinguishable from a scan that stopped early — "scanned 8 page(s)" reads // like partial coverage unless it says why it was only 8. function describeCoverage(result: SearchResult): string { const c = result.coverage; if (!c.pruned || c.pagesTotal === 0) return ""; if (c.pagesPlanned >= c.pagesTotal) return ""; return ( `filter-pruned: planned ${c.pagesPlanned} of ${c.pagesTotal} page(s)` + (c.unknownVideos > 0 ? ` (+${c.unknownVideos} video(s) not in the summaries index, whose pages were read anyway)` : "") ); } // Coerce a tool arg to a trimmed non-empty string[] (drops non-strings/blanks). function strArray(v: unknown): string[] | undefined { if (!Array.isArray(v)) return undefined; const out = v .filter((x): x is string => typeof x === "string") .map((x) => x.trim()) .filter((x) => x !== ""); return out.length > 0 ? out : undefined; } // A human scope note for the search footer: names the resolved scope (whole // corpus, a named group + channel count, or an N-channel set) and warns about // any channel/group token that matched nothing — so a typo is surfaced, not // silently a full-corpus scan. function describeScope(sel: SearchResult["selection"]): string { const parts: string[] = []; if (sel.all) { parts.push("scope: whole corpus"); } else if (sel.matchedGroups.length > 0) { const names = sel.matchedGroups.map((g) => g.name || g.id).join(", "); const label = sel.matchedGroups.length === 1 ? "group" : "groups"; parts.push(`scope: ${label} ${names} (${sel.channelCount} channel(s))`); } else { parts.push(`scope: ${sel.channelCount} channel(s)`); } if (sel.unknownChannels.length > 0) { parts.push(`unknown channel(s): ${sel.unknownChannels.join(", ")}`); } if (sel.unknownGroups.length > 0) { parts.push(`unknown group(s): ${sel.unknownGroups.join(", ")}`); } return parts.join("; "); } async function handleGetTranscripts( source: ShardSource, args: Record, ): Promise { const rawIds = Array.isArray(args.video_ids) ? args.video_ids : []; const ids = rawIds .filter((v): v is string => typeof v === "string" && v.trim() !== "") .map((v) => v.trim()); if (ids.length === 0) return errorText("video_ids is required (non-empty)"); const CAP = 20; const dropped = ids.length > CAP ? ids.length - CAP : 0; const batch = ids.slice(0, CAP); // One query or several. `queries` kills the refetch-per-quote pattern: the // windows for every query merge in ONE pass per video (mergeSnippets), and // each header reports a per-query count — so a query that matched nothing // becomes visible instead of vanishing into a merged excerpt. const QUERY_CAP = 8; const queryList = [ ...(typeof args.query === "string" && args.query.trim() !== "" ? [args.query.trim()] : []), ...(strArray(args.queries) ?? []), ]; const droppedQueries = queryList.length > QUERY_CAP ? queryList.length - QUERY_CAP : 0; const queries = queryList.slice(0, QUERY_CAP); const channelHint = [ ...(typeof args.channel === "string" ? [args.channel] : []), ...(strArray(args.channels) ?? []), ]; const timestamps = args.timestamps !== false; const types = parseContentTypes(args.content_types); const wantVideos = types.includes("video"); const wantPosts = types.includes("post"); // Build the (alias-aware) matchers once for the whole batch when windowing: // one per query for the counts, plus their OR for the window pass. const firedAliases: SearchAlias[] = []; let perQuery: { query: string; match: (t: string) => boolean }[] = []; let matcher: { match: (t: string) => boolean } | null = null; if (queries.length > 0) { const useAliases = args.use_aliases !== false && args.regex !== true; const aliases = useAliases ? await source.loadAliases() : []; perQuery = queries.map((q) => { const built = buildMatcher({ query: q, regex: args.regex === true, useAliases, aliases, }); for (const a of built.firedAliases) { if (!firedAliases.some((x) => x.id === a.id)) firedAliases.push(a); } return { query: q, match: built.match }; }); matcher = { match: (t: string) => perQuery.some((m) => m.match(t)) }; } const before = typeof args.before === "number" ? args.before : 30; const after = typeof args.after === "number" ? args.after : 30; const base = linkStyleOf(args) === "base"; const maxLinesArg = typeof args.max_lines === "number" ? Math.floor(args.max_lines) : undefined; const maxLines = maxLinesArg !== undefined && maxLinesArg >= 1 ? maxLinesArg : undefined; const blocks: string[] = []; const missing: string[] = []; for (const id of batch) { const found = wantVideos ? await findVideo(source, id, channelHint) : null; if (!found) { // content_types was declared on this tool and never read, so a batch of // post ids silently came back "not found". Fall through to the post // corpus for anything the transcript lookup didn't resolve. const post = wantPosts ? await findPost(source, id, channelHint) : null; if (post) { blocks.push(postToMarkdown(post.post)); continue; } missing.push(id); continue; } const { record, ch } = found; const link: LinkableHit = { slug: record.slug, ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), ...(record.platform ? { platform: record.platform } : {}), }; const baseUrl = base ? momentBaseFor(source, link) : null; // The bracketed stamp's contents: the full inline Markdown link (byte- // identical to the pre-link_style output), or the compact base form. const stamp = base ? baseStamp : (clock: string, seconds: number): string => { const url = momentLinkFor(source, link, seconds); return url ? `[${clock}](${url})` : clock; }; const head = `## ${record.title || id}\n` + `- video_id: ${id} | channel: ${ch.name}` + (ch.siteTitle ? ` | site: ${ch.siteTitle}` : "") + ` | uploaded: ${formatDate(record.uploadDate)}` + (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "") + (baseUrl ? `\n- moment_base: ${baseUrl}` : ""); if (matcher) { const { lines, matchCount } = getWindowedTranscript(record, matcher.match, { before, after, timestamps, stamp, maxLines, }); // Per-query counts, so a query contributing nothing to this video is // visible rather than absorbed into the merged window — that visibility // is the real diagnostic value of batching several queries. const cues = record.cues ?? []; const counts = perQuery.map((m) => ({ query: m.query, n: cues.reduce((acc, c) => acc + (m.match(c.text) ? 1 : 0), 0), })); const countNote = counts.length > 1 ? counts.map((c) => `"${c.query}": ${c.n}`).join(", ") : `${matchCount} matching line(s)`; // The record's alternate tracks (lib/captionTracks.ts): a match only // an alternate holds is windowed from that track, under its name. const altBlocks: string[] = []; let altMatches = 0; const across = hitsAcrossTracks(record, (list) => list.filter((c) => matcher.match(c.text)), ); for (const alt of record.altTracks ?? []) { const own = new Set(across.filter((h) => h.track === alt.track).map((h) => h.text)); if (own.size === 0) continue; const w = windowedTranscript(alt.cues, (t) => own.has(t), { before, after, timestamps, stamp, maxLines: maxLines ?? MCP_POLICY.windowLineCap, }); altMatches += w.matchCount; altBlocks.push( `_(${w.matchCount} matching line(s) only ${inTrackLabel(alt.track)} (track ${alt.track}), windowed)_\n${w.lines.join("\n")}`, ); } const body = matchCount === 0 && altMatches === 0 ? `_(no lines matched ${ counts.length > 1 ? "any query" : "the query" } in this transcript${counts.length > 1 ? ` — ${countNote}` : ""})_` : [ ...(matchCount > 0 ? [`_(${countNote}, windowed)_\n${lines.join("\n")}`] : []), ...altBlocks, ].join("\n\n"); blocks.push(`${head}\n\n${body}`); } else { const md = transcriptToMarkdown( record, base ? { timestamps, includeTags: true, stampForCue: baseStamp, extraMeta: baseUrl ? [`moment_base: ${baseUrl}`] : undefined, } : { timestamps, includeTags: true, linkForCue: (seconds) => momentLinkFor(source, link, seconds), }, ); blocks.push(md.trim()); } } const notes: string[] = []; if (queries.length > 1) notes.push(`queries: ${queries.map((q) => `"${q}"`).join(", ")}`); const aliasNote = describeFiredAliases(firedAliases); if (aliasNote) notes.push(aliasNote); if (missing.length > 0) notes.push(`not found: ${missing.join(", ")}`); if (dropped > 0) notes.push(`${dropped} extra id(s) beyond the 20-cap dropped`); if (droppedQueries > 0) { notes.push(`${droppedQueries} quer(ies) beyond the ${QUERY_CAP}-cap dropped`); } if (base) notes.push(BASE_EXPANSION_NOTE); const footer = notes.length > 0 ? `\n\n(${notes.join("; ")})` : ""; if (blocks.length === 0) { return text(`No transcripts read.${footer}`); } return text(`${blocks.join("\n\n---\n\n")}${footer}`); } // Render one post as markdown. No timestamps anywhere — a post has no timeline, // and emitting a 0:00 would invite a bogus `@ mm:ss` citation. function postToMarkdown(post: Post, heading = true, archive: string | null = null): string { const lines: string[] = []; if (heading) lines.push(`# Post by ${post.authorName || post.author}`); lines.push( post.platform === "xenforo" ? `- author: ${post.author}${post.forum?.authorId ? ` (member ${post.forum.authorId})` : ""}` : `- author: ${post.authorName ? `${post.authorName} (@${post.author})` : `@${post.author}`}`, ); lines.push(`- posted: ${post.createdAt}`); lines.push(`- platform: ${post.platform}`); lines.push(`- post_id: ${post.id}`); lines.push(`- source: ${post.url}`); // The post's page on the archive, when the source has a viewer: a citation // that survives the post, or the platform, going away. if (archive) lines.push(`- archive: ${archive}`); if (post.isRepost) lines.push(`- repost: yes`); if (post.isReply) lines.push(`- reply: yes`); if (post.threadId && post.threadId !== post.id) { lines.push(`- thread_id: ${post.threadId}`); } // A forum post: which thread, where in it, edits, and whom it quotes. Quoted // text in the body is marked "> " — it is the quoted author's, not this // post's author's. if (post.forum) { const f = post.forum; lines.push( `- forum thread: ${f.threadTitle ? `${f.threadTitle} ` : ""}(${f.host}, thread ${f.threadId}` + `${f.page ? `, page ${f.page}` : ""}${f.position ? `, post #${f.position}` : ""})`, ); if (f.editedAt) lines.push(`- last edited: ${f.editedAt}`); if (f.quotes?.length) { lines.push( `- quotes: ${f.quotes .map((q) => `${q.author ?? "unnamed"}${q.postId ? ` (post_id ${q.postId})` : ""}`) .join("; ")} — quoted text is marked "> " in the body`, ); } } if (post.media?.length) { lines.push(`- media: ${post.media.length} item(s), linked not archived:`); for (const m of post.media.slice(0, 20)) { lines.push(` - ${m.kind}${m.provider ? ` (${m.provider})` : ""}: ${m.url}`); } } else if (post.mediaCount) { lines.push(`- media: ${post.mediaCount} attachment(s) (not archived)`); } if (post.engagement) { const e = post.engagement; const parts = [ e.likes != null ? `${e.likes} likes` : "", e.reposts != null ? `${e.reposts} reposts` : "", e.replies != null ? `${e.replies} replies` : "", e.quotes != null ? `${e.quotes} quotes` : "", ].filter(Boolean); if (parts.length) lines.push(`- engagement: ${parts.join(", ")}`); } lines.push(""); lines.push(post.text); if (post.links.length > 0) { lines.push(""); lines.push("Links:"); for (const l of post.links) lines.push(`- ${l}`); } return lines.join("\n"); } async function handleGetPost( source: ShardSource, args: Record, ): Promise { const postId = String(args.post_id ?? "").trim(); if (!postId) return errorText("post_id is required"); const found = await findPost( source, postId, typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`post not found: ${postId}`); const archive = viewerPostUrl(found.ch.siteUrl ?? source.publicOrigin(), found.post.slug); return text(postToMarkdown(found.post, true, archive)); } async function handleGetThread( source: ShardSource, args: Record, ): Promise { const postId = String(args.post_id ?? "").trim(); if (!postId) return errorText("post_id is required"); const found = await findPost( source, postId, typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`post not found: ${postId}`); const thread = await getThread(source, found.ch, found.post); const head = `# Thread (${thread.length} post${thread.length === 1 ? "" : "s"}) — ` + `${found.post.authorName || found.post.author}\n`; const body = thread .map((p, i) => `## ${i + 1}. ${p.authorName || p.author}\n${postToMarkdown(p, false)}`) .join("\n\n"); return text(`${head}\n${body}`); } async function handleGetTranscript( source: ShardSource, args: Record, editorDeps?: EditorDeps, ): Promise { const videoId = String(args.video_id ?? "").trim(); if (!videoId) return errorText("video_id is required"); const channelArg = typeof args.channel === "string" ? args.channel : undefined; const found = await findVideo(source, videoId, channelArg); if (!found) { // NOT IN THE ARCHIVE: a video imported or transcribed since the last // build. With a local editor configured, its cues are read off the // editor's disk (release 19 A7) — the primary transcript only. const askedTrack = typeof args.track === "string" && args.track.trim() !== ""; if (editorDeps && !askedTrack) { const fromEditor = await editorTranscript(editorDeps, videoId, channelArg); if (fromEditor.ok) { return text(renderEditorTranscript(fromEditor.t, { timestamps: args.timestamps !== false })); } if (fromEditor.error) return errorText(`video not found in the archive: ${videoId}\n\n${fromEditor.error}`); } return errorText(`video not found: ${videoId}`); } const { record, ch } = found; const link: LinkableHit = { slug: record.slug, ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}), ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}), ...(record.platform ? { platform: record.platform } : {}), }; // The record's tracks (lib/captionTracks.ts): the primary unless `track` // names an alternate it holds. const tracks = recordTracks(record); const asked = typeof args.track === "string" ? args.track.trim() : ""; const cues = cuesOfTrack(record, asked); if (asked && cues === undefined) { return errorText( tracks.length > 1 ? `video ${videoId} has no track "${asked}"; its tracks are: ${tracks.map((t) => `${t} (${trackLabel(t)})`).join(", ")}` : `video ${videoId} has no track "${asked}": it has only its primary transcript`, ); } const shown = asked || record.track; const extraMeta = tracks.length > 1 && shown ? [ `track: ${shown} (${trackLabel(shown)})${shown === record.track ? " — the primary" : ""}`, `other tracks: ${tracks .filter((t) => t !== shown) .map((t) => `${t} (${trackLabel(t)}${t === record.track ? ", the primary" : ""})`) .join(", ")} — pass track to read one`, ] : undefined; const md = transcriptToMarkdown( { ...record, cues }, { timestamps: args.timestamps !== false, includeTags: true, linkForCue: (seconds) => momentLinkFor(source, link, seconds), ...(extraMeta ? { extraMeta } : {}), }, ); return text(md); } // Everything the archive knows about one video, joined from the four layers // that ship alongside the transcript: the transcript record itself, `stats/` // (engagement + transcript coverage), `duplicates.json` (other copies of the // same recording), and `digests/` (AI chapters and topic tags). // // Each layer is optional and degrades to a stated absence rather than silence: // "no digest" and "there is no digest layer here" are different facts, and a // caller deciding whether to trust a quote needs to be able to tell them apart. async function handleGetMetadata( source: ShardSource, args: Record, ): Promise { const videoId = String(args.video_id ?? "").trim(); if (!videoId) return errorText("video_id is required"); const found = await findVideo( source, videoId, typeof args.channel === "string" ? args.channel : undefined, ); if (!found) return errorText(`video not found: ${videoId}`); const { cues, altTracks, ...meta } = found.record; const slug = found.record.slug; // An alternate's cues are not metadata; its id and label are. const otherTracks = (altTracks ?? []).map((t) => ({ track: t.track, label: trackLabel(t.track), cueCount: t.cues.length, })); const lines: string[] = [ JSON.stringify( { ...meta, channelName: found.ch.name, cueCount: cues?.length ?? 0, ...(otherTracks.length > 0 ? { otherTracks } : {}), }, null, 2, ), ]; // ── stats/ ── const stat = await lookupStat(source, slug); if (stat) { const engagement = [ stat.viewCount != null ? `${stat.viewCount.toLocaleString()} views` : "", stat.likeCount != null ? `${stat.likeCount.toLocaleString()} likes` : "", stat.commentCount != null ? `${stat.commentCount.toLocaleString()} comments` : "", ].filter(Boolean); const statLines = [ `## Stats`, engagement.length > 0 ? `- engagement: ${engagement.join(", ")}` : "", `- platform state: ${stat.status}`, `- media type: ${stat.mediaType}`, stat.duration ? `- duration: ${formatDuration(stat.duration)}` : "", stat.cueCount != null ? `- transcript cues: ${stat.cueCount}` : "", stat.downloadedDate ? `- downloaded: ${formatDate(stat.downloadedDate)}` : "", stat.transcribedDate ? `- transcribed: ${formatDate(stat.transcribedDate)}` : "", // Coverage is the one stat that changes how the transcript may be USED, // so it is phrased as a warning rather than a number: a transcript that // stops at 41% will answer "he never said X" wrongly and confidently. coverageNote(stat.coverage), ].filter(Boolean); lines.push(statLines.join("\n")); } // ── duplicates.json ── lines.push(await describeOtherCopies(source, slug)); // ── digests/ ── lines.push(await describeDigest(source, found.ch, videoId, slug)); return text(lines.filter(Boolean).join("\n\n")); } // The transcript-coverage caveat, or "" when coverage is fine/unknown. The // threshold is deliberately generous: values a little under 1 are normal // (trailing silence), while a genuinely truncated download lands far below. function coverageNote(coverage: number | null): string { if (coverage == null) return ""; if (coverage >= 0.9) return ""; return ( `- ⚠ TRANSCRIPT COVERS ONLY ${Math.round(coverage * 100)}% OF THE RUNTIME — ` + `the download was truncated. Quotes from the uncovered tail are missing, so ` + `do NOT conclude from this transcript that something was never said.` ); } async function lookupStat( source: ShardSource, slug: string, ): Promise { if (typeof source.statsIndex !== "function") return undefined; try { return (await source.statsIndex()).get(slug); } catch { return undefined; } } // Other uploads of the SAME recording, and — the load-bearing part — whether a // timestamp may legitimately be carried across to them. // // `aligned` is the gate, and absent means NOT MEASURED, which is treated as not // aligned. A mirror with a different intro matches on text at shifted times, so // mapping a citation into it produces a link that looks perfectly plausible and // points at the wrong moment of a different upload. Refusing to guess is the // only correct behaviour, and saying so is how the caller learns not to. async function describeOtherCopies( source: ShardSource, slug: string, ): Promise { if (typeof source.duplicateIndex !== "function") return ""; let membership; try { membership = (await source.duplicateIndex()).get(slug); } catch { return ""; } if (!membership || membership.siblings.length === 0) return ""; const rows = membership.siblings.map((s) => { const timing = s.aligned ? `timings ALIGNED${ s.offsetSeconds != null ? ` (max observed offset ${s.offsetSeconds}s)` : "" } — a moment link may be mapped across` : `⚠ timing NOT measured/aligned — do NOT map timestamps into this copy; ` + `open it at 0:00 and locate the moment again`; return ( `- ${s.id} (${s.channel}, ${s.platform}, ${formatDuration(s.duration)}, ` + `uploaded ${formatDate(s.uploadDate)}` + `${s.hasTranscript ? "" : ", NO transcript"}) — ${timing}` ); }); const caveats = [ membership.contained ? `- ⚠ this is a CLIP/EXCERPT relationship — the copies overlap only ` + `partially, so nothing may be mapped across wholesale` : "", membership.needsReview ? `- ⚠ UNCONFIRMED SUSPECT — matched on title and duration only; nothing ` + `compared the actual content, so treat "same recording" as a claim, ` + `not a fact` : "", membership.isCanonical ? "" : `- note: the canonical copy of this recording is ${membership.canonicalSlug}`, ].filter(Boolean); return [ `## Other copies of this recording (${membership.siblings.length})`, ...rows, ...caveats, ].join("\n"); } // The AI digest for a video, when one exists. Sparse by design — a manifest // lists only the digested videos — so absence is the common case and is stated // plainly rather than left as silence. async function describeDigest( source: ShardSource, ch: ChannelRef, videoId: string, slug: string, ): Promise { if ( typeof source.digestsManifest !== "function" || typeof source.digestPage !== "function" ) { return ""; } let manifest; try { manifest = await source.digestsManifest(ch); } catch { return ""; } if (!manifestHasDigest(manifest, videoId)) return ""; const page = manifest!.slugToPage[videoId]; let records; try { records = await source.digestPage(ch, page); } catch { return ""; } const digest = records.find((r) => r.id === videoId || r.slug === slug); if (!digest) return ""; const out: string[] = ["## AI digest"]; // A borrowed digest describes ANOTHER upload. Presenting it as native is the // failure that looks like success — every chapter reads plausibly while // describing a different video — so this leads, before any content. if (digest.derivedFrom) { out.push( `- ⚠ BORROWED: these chapters/tags were generated for ${digest.derivedFrom.slug}, ` + `a different upload of the same recording, and shared onto this one` + (digest.derivedFrom.offsetSeconds ? ` (offset ${digest.derivedFrom.offsetSeconds}s)` : "") + `. They describe that video's timeline.`, ); } if (digest.chapters.length > 0) { out.push(`### Chapters (${digest.chapters.length})`); for (const c of digest.chapters) { // `start` is the cue-snapped seconds; `clock` is the model's raw string // and is never re-parsed to seek. out.push( `- [${formatDuration(c.start)}] ${c.title}` + (c.decidedBy === "human" ? " _(human-written)_" : ""), ); } } if (digest.tags.length > 0) { out.push( `### Topic tags\n` + digest.tags .map((t) => t.tag + (t.decidedBy === "human" ? " (human)" : "")) .join(", "), ); } return out.length > 1 ? out.join("\n") : ""; } // ─── Source discovery: list_sources / resolve_source ─── // A one-line human description of a source spec's target (kind + where it // points), for the list_sources / resolve_source reports. function describeSpec(spec: SourceSpec | undefined): string { if (!spec) return "(pinned source)"; switch (spec.kind) { case "hub": return spec.sites && spec.sites.length > 0 ? `hub ${spec.url} (subset: ${spec.sites.join(", ")})` : `hub ${spec.url} (all members)`; case "remote": return `remote ${spec.url}`; case "local": return `local ${spec.dir}`; } } // Report the corpus this call read (the default when `source` was omitted) and, // when a hub is in play, its member sites as ready-to-paste handles. There is // no active source to report — only a default and whatever a caller names. async function handleListSources( registry: SourceRegistry, resolved: ResolvedSource, ): Promise { const isDefault = resolved.handle === registry.defaultHandle; const lines: string[] = [ `Corpus read by this call: ${resolved.label}`, ` handle: ${resolved.handle}` + (isDefault ? " (the default)" : ""), ` target: ${describeSpec(resolved.spec)}`, ]; if (!isDefault) { lines.push( ` default (used when a call omits source): ${registry.defaultHandle}`, ); } // A cited-only site would otherwise look like an empty corpus here. const reports = reportsLine(await sourceReports(resolved.source)); if (reports) lines.push(` reports: ${reports}`); const hubUrl = resolved.spec.kind === "hub" ? resolved.spec.url : registry.hubUrl(); let sites: HubSite[] = []; if (hubUrl) { try { sites = await registry.listHubSites(hubUrl); } catch (e) { lines.push(`\n(could not list hub members: ${(e as Error).message})`); } } if (sites.length > 0) { // When the resolved corpus is a hub subset, mark who is actually in it. const spec = resolved.spec; const subset = spec.kind === "hub" && spec.sites && spec.sites.length > 0 ? new Set(spec.sites) : undefined; const rows = sites.map((site) => { const mark = subset ? (subset.has(site.siteId) ? "✓ " : " ") : ""; return ( ` - ${mark}${site.siteId} · ${site.title} · ${site.url}\n` + ` source: remote:${site.url.replace(/\/+$/, "")}` ); }); lines.push( `\nHub member sites (${sites.length})` + (subset ? ` — ✓ = in the subset this call read` : "") + `:\n${rows.join("\n")}`, ); lines.push( `\nA subset of them: source:"hub:${hubUrl}#${sites .slice(0, 2) .map((x) => x.siteId) .join(",")}".`, ); } lines.push( `\nPass a corpus per call, in each tool's own \`source\` argument — ` + `'default', 'local:', 'remote:', 'hub:', or ` + `'hub:#'. Nothing is switched or remembered: a call ` + `without \`source\` always reads the default. resolve_source turns a ` + `loose reference into the handle to pass. Read-only throughout.`, ); return text(lines.join("\n")); } // Resolve a reference to its canonical handle and verify the corpus can be // read. Mutates nothing — the point is to hand back a string to pass as // `source`, not to select anything. async function handleResolveSource( resolved: ResolvedSource, ): Promise { const lines = [ `Handle: ${resolved.handle}`, ` label: ${resolved.label}`, ` target: ${describeSpec(resolved.spec)}`, ]; for (const n of resolved.notes) lines.push(` note: ${n}`); try { const channels = await resolved.source.listChannels(); const { groups } = await resolved.source.loadGroups(); lines.push( ` reachable: yes — ${channels.length} channel(s)` + (groups.length > 0 ? `, ${groups.length} group(s)` : ""), ); const reports = reportsLine(await sourceReports(resolved.source)); if (reports) lines.push(` reports: ${reports}`); // Named here so a caller learns whether a `tags` filter is even available // BEFORE it writes one and reads the empty result as an answer. The two // states are different and both are said: some tags, or none at all. const tags = await loadSourceTags(resolved.source); lines.push( tags.length > 0 ? ` curated tags: ${tags.length} — ` + tags.map((t) => `${t.id} (${t.count})`).join(", ") + ` · list_tags for the breakdown` : ` curated tags: none published (a \`tags\` filter matches nothing here)`, ); } catch (e) { return errorText( `${lines.join("\n")}\n reachable: NO — ${(e as Error).message}`, ); } lines.push( ``, `Nothing was switched. Pass source:"${resolved.handle}" on each call ` + `that should read this corpus.`, ); return text(lines.join("\n")); } // The old switching tool, kept one release as an unadvertised alias so a habit // (or a stored transcript) doesn't hard-fail. It resolves its target and hands // back the handle; it cannot switch anything, because there is nothing to // switch. async function handleLegacyUseSource( registry: SourceRegistry, args: Record, ): Promise { const sites = strArray(args.sites); const token = (typeof args.site === "string" && args.site.trim()) || (typeof args.remote === "string" && `remote:${args.remote.trim()}`) || (typeof args.local === "string" && `local:${args.local.trim()}`) || (typeof args.hub === "string" && `hub:${args.hub.trim()}`) || (sites ? `sites:${sites.join(",")}` : ""); if (!token) { return errorText( "use_source is gone: there is no active source to switch. Pass the " + "corpus per call in each tool's `source` argument instead, and use " + "resolve_source to turn a URL or site name into a handle.", ); } let target: ResolvedSource; try { target = await registry.resolve(token); } catch (e) { return errorText(`use_source: ${(e as Error).message}`); } return text( `use_source no longer switches anything — this server holds no active ` + `source (a persisted one silently sent every call to the wrong corpus, ` + `so it was removed).\n\n` + `That target's handle is: ${target.handle}\n` + ` target: ${describeSpec(target.spec)}\n` + (target.notes.length > 0 ? ` note: ${target.notes.join("; ")}\n` : "") + `\nPass source:"${target.handle}" on each call that should read it.`, ); } // ─── open_link: paste a share link → preview, adjust, apply ─── // Map the snake_cased tool `overrides` object onto the LinkOverrides shape. function parseOverrides(raw: unknown): LinkOverrides | undefined { if (!raw || typeof raw !== "object") return undefined; const o = raw as Record; const bool = (k: string): boolean | undefined => o[k] === true ? true : undefined; const str = (k: string): string | undefined => typeof o[k] === "string" ? (o[k] as string) : undefined; const scope = str("query_scope"); return { clearFilters: bool("clear_filters"), clearAvailability: bool("clear_availability"), clearType: bool("clear_type"), clearAge: bool("clear_age"), clearDates: bool("clear_dates"), clearChannels: bool("clear_channels"), channels: strArray(o.channels), dateFrom: str("date_from"), dateTo: str("date_to"), query: str("query"), regex: o.regex === true ? true : undefined, queryScope: scope === "transcripts" || scope === "chat" || // "posts" was declared in the tool schema and dropped here, so a caller // asking to re-target a link at the post corpus silently got transcripts. // Everything downstream (LayerScope, runSearchSpec's posts pass) already // supported it. scope === "posts" || scope === "metadata" || scope === "description" || scope === "tags" ? scope : undefined, }; } type OriginProbe = { kind: "hub" | "remote"; siteCount?: number; channelCount?: number }; type ChannelScope = { channels: ChannelRef[]; all: boolean; matched: string[]; unknown: string[]; }; // Resolve a decoded link's `fc` channel NAMES against a live source's channels // (by display name or slug, case-insensitive). No channel filter → whole corpus. async function resolveLinkChannels( source: ShardSource, decoded: DecodedLink, ): Promise { const all = await source.listChannels(); if (!decoded.hasChannelFilter) { return { channels: all, all: true, matched: [], unknown: [] }; } const want = decoded.channelNames.map((n) => n.toLowerCase()); const matches = (c: ChannelRef): boolean => want.includes(c.name.toLowerCase()) || want.includes(c.slug.toLowerCase()); const channels = all.filter(matches); const matched = channels.map((c) => c.name); const unknown = decoded.channelNames.filter( (n) => !all.some( (c) => c.name.toLowerCase() === n.toLowerCase() || c.slug.toLowerCase() === n.toLowerCase(), ), ); return { channels, all: false, matched, unknown }; } // Render the human-readable plan for a decoded link. function buildLinkPlan( decoded: DecodedLink, probe: OriginProbe, scope: ChannelScope, handle: string, searched: boolean, ): string { const lines: string[] = []; lines.push(searched ? "## Share link" : "## Share-link plan (dry run)"); const kindNote = probe.kind === "hub" ? `hub (${probe.siteCount ?? "?"} member site(s))` : `single site${probe.channelCount != null ? ` (${probe.channelCount} channel(s))` : ""}`; lines.push(`- Source: ${decoded.origin} — ${kindNote}`); lines.push(`- Corpus handle: \`${handle}\` — pass this as \`source\` on the follow-up calls`); lines.push(`- Query: ${renderQueryTree(decoded.tree)} _(from ${decoded.querySource})_`); if (scope.all) { lines.push(`- Scope: whole corpus`); } else { const names = scope.matched.length ? scope.matched.join(", ") : "(none)"; lines.push(`- Scope: ${scope.channels.length} channel(s): ${names}`); if (scope.unknown.length > 0) { lines.push(` - ⚠ channel name(s) not found in this corpus: ${scope.unknown.join(", ")}`); } if (scope.channels.length === 0) { lines.push( ` - ⚠ the link selects 0 channels here — nothing will match. Re-call ` + `with overrides.clear_channels to search the whole corpus.`, ); } } const filterLines = describeFilters(decoded.filters); if (filterLines.length > 0) { lines.push(`- Filters:`); for (const f of filterLines) lines.push(` - ${f}`); } else { lines.push(`- Filters: none`); } for (const w of decoded.warnings) lines.push(`- Note: ${w}`); if (!searched) { lines.push( ``, `Dry run — no search was performed. Re-call without dry_run to search, ` + `or pass overrides to adjust the plan first.`, ); } return lines.join("\n"); } // A ScopedSnippet line: a linked `[m:ss]` for timed scopes, or a scope-tagged // note for the non-timed scopes (metadata / description / tags). function renderScopedSnippet( source: ShardSource, hit: SpecHit, s: ScopedSnippet, ): string { if (s.seconds > 0) { const tag = s.scope === "transcripts" ? s.track ? `${inTrackLabel(s.track)} ` : "" : `${s.track ?? s.scope} `; return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`; } return ` - [${s.scope}] ${s.text}`; } // open_link results (and get_transcript) stay inline-linked for now — a // link_style:"base" form for them is future work. function renderSpecResults(source: ShardSource, hits: SpecHit[]): string { return hits .map((h) => { const head = `### ${h.title}\n` + `- video_id: ${h.videoId} | channel: ${h.channelName}` + (h.siteTitle ? ` | site: ${h.siteTitle}` : "") + ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` + (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : ""); const snips = h.snippets .map((s) => renderScopedSnippet(source, h, s)) .join("\n"); return snips ? `${head}\n${snips}` : head; }) .join("\n\n"); } async function handleOpenLink( registry: SourceRegistry, callerCorpus: ResolvedSource, args: Record, ): Promise { const linkStr = typeof args.link === "string" ? args.link.trim() : ""; if (!linkStr) return errorText("link is required"); let decoded: DecodedLink; try { decoded = decodeShareLink(linkStr); } catch (e) { return errorText(`could not parse link as a URL: ${(e as Error).message}`); } decoded = applyLinkOverrides(decoded, parseOverrides(args.overrides)); // The link's ORIGIN decides the corpus, not the caller's `source` — a share // link is a pointer to a specific deployment. An explicit `source` still // wins, for the case where the same corpus is reachable at another handle. const explicitSource = typeof args.source === "string" && args.source.trim() !== ""; let target: ResolvedSource; if (explicitSource) { target = callerCorpus; } else { try { target = await registry.resolve(decoded.origin); } catch (e) { return errorText( `could not resolve the link's origin ${decoded.origin}: ${(e as Error).message}`, ); } } const source = target.source; const probe: OriginProbe = target.spec.kind === "hub" ? { kind: "hub" } : { kind: "remote" }; let scope: ChannelScope; try { scope = await resolveLinkChannels(source, decoded); } catch (e) { return errorText( `could not read the corpus at ${target.handle}: ${(e as Error).message}`, ); } probe.channelCount = scope.all ? scope.channels.length : undefined; // dry_run stops here: the plan, with no search. The default is to do the // whole job in this one call — the old two-phase apply:true cost a round // trip and, worse, mutated a global. const dryRun = args.dry_run === true; const plan = buildLinkPlan(decoded, probe, scope, target.handle, !dryRun); if (dryRun) return text(plan); const limit = typeof args.limit === "number" ? args.limit : 20; const offset = typeof args.offset === "number" ? args.offset : 0; const result = await runSearchSpec( source, scope.channels, { tree: decoded.tree, filters: decoded.filters, aliases: await source.loadAliases(), }, { limit, offset }, ); const aliasNote = describeFiredAliases(result.firedAliases); const rangeStart = result.total === 0 ? 0 : result.offset + 1; const rangeEnd = result.offset + result.hits.length; const footer = `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` + `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` + `page(s) across ${result.scanned.channels} channel(s)` + (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") + (aliasNote ? `; ${aliasNote}` : "") + ")"; const body = result.hits.length === 0 ? result.total === 0 ? "No matches for this link's search." : `No matches in this page (offset ${result.offset} is past the ${result.total} total).` : `${result.total} video(s) matched:\n\n${renderSpecResults(source, result.hits)}`; const banner = incompletePageBanner(result, "open_link"); return text(`${plan}\n\n---\n\n${banner}${body}${footer}`); } // ─── sweep_plan / ask_plan: a pasted request in, a resolved plan out ─── // Pre-resolve the facts the model would otherwise burn round-trips on: // canonicalise the corpus, validate the channel/group tokens against the live // channel list, and offer the group roster when no scope was given. Tolerant — // an unreadable corpus just means fewer resolved facts, not a failed plan. async function buildPlanContext( resolved: ResolvedSource, req: PromptRequest, ): Promise { const ctx: PlanContext = { corpus: resolved.handle, notes: [...resolved.notes] }; let channels: ChannelRef[]; let groups: ChannelGroup[]; try { channels = await resolved.source.listChannels(); groups = (await resolved.source.loadGroups()).groups; } catch (e) { ctx.notes = [ ...(ctx.notes ?? []), `could not read ${resolved.handle} to validate the scope: ${(e as Error).message}`, ]; return ctx; } const known = (token: string): boolean => { const want = token.toLowerCase(); return channels.some( (c) => c.slug.toLowerCase() === want || c.key.toLowerCase() === want || c.name.toLowerCase() === want, ); }; ctx.knownChannels = req.channels.filter(known); ctx.unknownChannels = req.channels.filter((c) => !known(c)); const knownGroup = (token: string): boolean => { const want = token.toLowerCase(); return groups.some( (g) => g.id.toLowerCase() === want || (g.name.trim() !== "" && g.name.toLowerCase() === want), ); }; ctx.knownGroups = req.groups.filter(knownGroup); ctx.unknownGroups = req.groups.filter((g) => !knownGroup(g)); if (req.channels.length === 0 && req.groups.length === 0 && groups.length > 0) { const counts = new Map(); for (const c of channels) { const gid = resolveChannelGroupId( c.groupId, groups, (await resolved.source.loadGroups()).defaultGroupId, ); counts.set(gid, (counts.get(gid) ?? 0) + 1); } ctx.availableGroups = sortGroups(groups).map((g) => ({ id: g.id, name: g.name || g.id, channels: counts.get(g.id) ?? 0, })); } ctx.notes = [ ...(ctx.notes ?? []), `${channels.length} channel(s) in ${resolved.handle}`, ]; return ctx; } // Both plan tools. The request arrives as ONE structured JSON string, so a URL // with `?a=b&c=d` and a full sentence of punctuation survive intact — which is // the whole reason these are tools and not prompt arguments. async function handlePlan( kind: "sweep" | "ask", registry: SourceRegistry, callerCorpus: ResolvedSource, args: Record, ): Promise { const raw = typeof args.request === "string" ? args.request : ""; if (raw.trim() === "") { return errorText( `${kind}_plan needs a request: what to ${kind === "sweep" ? "sweep for" : "ask"}, ` + `in plain English, optionally with a share link.`, ); } const req = parsePromptRequest(raw); // A `source=` inside the request text wins over the tool's own `source` // argument, since it is the more specific statement of intent. let corpus = callerCorpus; if (req.source) { try { corpus = await registry.resolve(req.source); } catch (e) { req.warnings.push( `source=${req.source} could not be resolved (${(e as Error).message}) — ` + `using ${callerCorpus.handle}.`, ); } } const ctx = await buildPlanContext(corpus, req); return text( kind === "sweep" ? buildSweepInstructions(req, ctx) : buildAskInstructions(req, ctx), ); } // ─── Prompts: the first-class `sweep` entry point ─── // A single slash command that drives Claude Code to run the browser's // "corpus sweep" on plan usage: enumerate a query's full match set, batch the // windowed transcripts, and fold findings into a markdown report it maintains // with its own Write/Edit tools. The MCP stays read-only; the report is a file // in Claude's cwd. Mirrors the browser's accumulationSystemPrompt discipline. const PROMPTS: Prompt[] = [ { name: "sweep", description: "IN CLAUDE CODE, USE /sweep INSTEAD — this form's arguments get " + "word-split by the slash-command tokenizer and everything past the " + "ninth word is dropped; it will refuse rather than sweep the wrong " + "thing. This form is for clients that give each argument its own field " + "(Claude Desktop, Cursor). Sweep a query across the corpus (or a chosen " + "group/channels): enumerate every matching video, batch-read the " + "transcripts, and fold cited, cross-referenced findings into a markdown " + "report — on plan usage, no API key. With no scope arg it lists the " + "channel groups and asks you to pick a group/channels (or confirm " + "'all') before sweeping.", arguments: [ { name: "query", description: "Term or phrase to sweep for. Optional when a `link` is given (the " + "link supplies the query).", required: false, }, { name: "link", description: "An archilyzer viewer share URL to seed the sweep from — its origin " + "(source), query tree, and filters are decoded and applied via " + "open_link before enumerating.", required: false, }, { name: "channel", description: "Optional channel slug/name to restrict the sweep to.", required: false, }, { name: "channels", description: "Optional comma-separated channel slugs/names to restrict the sweep to.", required: false, }, { name: "group", description: "Optional channel group (id or name, e.g. 'other' or 'Extended " + "Universe') to restrict the sweep to its member channels.", required: false, }, { name: "directive", description: "What to extract (default: 'key claims & contradictions').", required: false, }, { name: "batch_size", description: "Videos per batch (default 8).", required: false, }, { name: "parse_model", description: "Model to request for the per-batch extractor subagents (default " + "'haiku'). The extractors only quote verbatim, so the cheapest " + "model wins; all synthesis stays with the orchestrator.", required: false, }, { name: "report_path", description: "Report file to write (default ./sweep-report.md).", required: false, }, ], }, ]; // The MCP prompt form. Its declared arguments go through the SAME parser and // validators as the tools, so a form-based client (Claude Desktop, Cursor — // where each argument gets its own field and multi-word values are fine) gets // the identical instructions, and a bad value gets a named warning instead of // being rendered into the text. function buildSweepPrompt(args: Record): { description: string; messages: { role: "user"; content: { type: "text"; text: string } }[]; } { // Refuse a shredded request rather than sweeping for something nobody asked // for. A typed argument holding prose ("link" = "search", "batch_size" = // "Why") is the fingerprint of Claude Code's slash-command tokenizer, which // whitespace-splits this prompt's arguments and drops the overflow. A // form-based client, where each argument has its own field, only trips this // by genuinely typing a bad value — which also deserves an error. const { problems, reassembled } = validateSweepArguments(args); if (problems.length > 0) { const named = problems .map((p) => ` - \`${p.arg}\` = "${p.value}" — ${p.why}`) .join("\n"); return { description: "sweep: the request did not arrive intact", messages: [ { role: "user" as const, content: { type: "text" as const, text: `Do NOT run a sweep. Tell me, briefly, that the request did not ` + `arrive intact, and show me this:\n\n` + `The \`sweep\` prompt received values that cannot be what they ` + `claim to be:\n${named}\n\n` + `In Claude Code this almost always means the slash-command ` + `tokenizer word-split the line: it splits on whitespace, zips ` + `the words onto this prompt's nine declared arguments in order, ` + `and **silently drops everything past the ninth word**. It is ` + `not quote-aware.\n\n` + (reassembled ? `The words that survived, in order, were:\n\n ${reassembled}\n\n` + `Anything after them was discarded.\n\n` : "") + `**Use \`/sweep\` instead** — it passes the whole line through ` + `untouched:\n\n` + ` /sweep ${reassembled || ""}\n\n` + `(\`/ask\` is the same thing for a question answered in the ` + `conversation rather than a report file. If \`/sweep\` is not in ` + `the slash list, the commands live in \`.claude/commands/\` and ` + `are picked up when a session starts — restart the session; ` + `restarting the MCP server only reloads its tools.)`, }, }, ], }; } const req = requestFromArguments(args); if (!req.query && !req.link) { throw new Error("sweep requires a query argument (or a link)"); } // The prompt form has no live corpus to resolve against — it is rendered // before any tool call — so the plan names the server default and the model // resolves the rest as step 1. const ctx: PlanContext = { corpus: "default" }; const subject = req.query ? `"${req.query}"` : "the share link's search"; return { description: `Corpus sweep for ${subject} → ${req.reportPath ?? DEFAULT_REPORT_PATH}`, messages: [ { role: "user" as const, content: { type: "text" as const, text: buildSweepInstructions(req, ctx), }, }, ], }; }