import { test } from "node:test"; import assert from "node:assert/strict"; // Client AND InMemoryTransport must come from the SAME package: each bundles // its own copy with private state, so a pair split across packages never links. import { Client, InMemoryTransport } from "@modelcontextprotocol/client"; import type { ChannelTranscriptsManifest, ChannelSubsManifest, } from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import type { ChannelPostsManifest, Post, } from "yt-dlp-transcript-common/lib/posts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import type { VideoStat } from "yt-dlp-transcript-common/lib/stats"; import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups"; import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; import type { ChannelGroups, ChannelRef, ShardSource, VideoAvailability, } from "./source"; import { searchTranscripts, findPost, getThread, getWindowedTranscript, buildMatcher, resolveSelectedChannels, runSearchSpec, type SearchFilters, } from "./search"; import { createServer } from "./server"; import { SourceRegistry } from "./sourceRegistry"; import { MISSING_STATES, VIDEO_STATES, type VideoState, } from "yt-dlp-transcript-common/lib/availability"; import { parseShareV1 } from "yt-dlp-transcript-common/components/shareUrl"; // Every filter facet kept (the "no-op" filter) — override fields per test. const KEEP_ALL: SearchFilters = { videos: true, livestreams: true, allAges: true, restricted: true, states: new Set(VIDEO_STATES), }; // Availability kept-set helper: `keep("available", "deleted")` etc. const keep = (...states: VideoState[]): SearchFilters => ({ ...KEEP_ALL, states: new Set(states), }); // ─── A tiny in-memory ShardSource for the tests ─── function cues(...pairs: [number, string][]): Cue[] { return pairs.map(([start, text]) => ({ start, end: start + 3, text })); } function vid( id: string, title: string, channelSlug: string, cs: Cue[], extra: Partial = {}, ): TranscriptDetail { return { // Real records carry a channel-prefixed slug (`/`); mirror // that so viewer-link + availability joins are exercised. slug: `${channelSlug}/${id}`, id, channelSlug, title, uploadDate: "20240101", duration: 300, channel: channelSlug === "chan-a" ? "Channel A" : "Channel B", description: "", tags: [], isLivestream: false, ageRestricted: false, platform: "youtube", webpageUrl: `https://example.test/${id}`, cues: cs, ...extra, }; } const K_CUPS_ALIAS: SearchAlias = { id: "k-cups", label: "K-Cups", triggers: ["k cups"], suggestion: "(k|cake)[ -]?cup", useRegex: true, enabled: true, }; // Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup". // Extra metadata on a1/a3/a4/a5 exercises the description/tags scopes and the // type/age/date filters without changing what matches the "coffee" cue search. const CHAN_A: TranscriptDetail[] = [ vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), { description: "a pour over brewing guide", tags: ["espresso", "beans"], // CURATED tags (lib/curatedTags.ts) — the operator's vocabulary, which is // a different field from the yt-dlp keywords one line above and is // deliberately set on the same record so a test that confuses the two // fails loudly. curatedTags: ["eva-collab"], }), vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])), vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), { isLivestream: true, curatedTags: ["eva-in-chat"], }), vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), { ageRestricted: true, }), vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"]), { uploadDate: "20250601", }), vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])), vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])), ]; // Live-chat cues, keyed by video slug — served via the subs shard methods so a // `chat`-scope leaf has something to match. Only b1 has a chat track. const CHAT: Record = { "chan-b/b1": cues([12, "@fan: banana bread anyone"], [15, "@mod: stay on topic"]), }; // Availability overrides, keyed by video slug (default: available). One video // per non-available VideoState so the `fav` filter is exercised across the // whole enum; b1 is the only video left available. const AVAILABILITY: Record = { "chan-a/a1": { state: "deleted" }, "chan-a/a2": { state: "unlisted" }, "chan-a/a3": { state: "private" }, "chan-a/a4": { state: "members_only" }, "chan-a/a5": { state: "maybe_missing" }, }; // Channel B: one long transcript for windowing (a single match at 100s). const CHAN_B: TranscriptDetail[] = [ vid( "b1", "Long one", "chan-b", cues( [0, "intro chatter"], [50, "still warming up"], [98, "right before the moment"], [100, "here is the coffee moment"], [102, "right after the moment"], [160, "much later unrelated"], [220, "the very end"], ), ), ]; // Two groups + a (blank-named) default. chan-a is in "other" (Extended // Universe); chan-b carries an unknown "ghost" groupId that must fold onto the // default group (exactly resolveChannelGroupId's semantics). const STUB_GROUPS: ChannelGroup[] = [ { id: "default", name: "", selectedByDefault: true, order: 0 }, { id: "other", name: "Extended Universe", selectedByDefault: false, order: 1, description: "Related characters", }, ]; // A tiny social channel: one post sharing the "k cups" term with the video // fixtures (so a default search proves BOTH corpora come back), plus a thread // reply for get_thread. const STUB_POSTS: Post[] = [ { id: "p1", slug: "chan-b/p1", channelSlug: "chan-b", author: "tester.bsky.social", authorName: "Tester", createdAt: "2026-03-01T10:00:00.000Z", uploadDate: "20260301", text: "a zephyrpost mentioning something", url: "https://bsky.app/profile/tester.bsky.social/post/p1", platform: "bluesky", isReply: false, isRepost: false, links: [], threadId: "p1", }, { id: "p2", slug: "chan-b/p2", channelSlug: "chan-b", author: "tester.bsky.social", authorName: "Tester", createdAt: "2026-03-01T11:00:00.000Z", uploadDate: "20260301", text: "a zephyrpost reply in the same thread", url: "https://bsky.app/profile/tester.bsky.social/post/p2", platform: "bluesky", isReply: true, isRepost: false, links: [], threadId: "p1", replyTo: { platform: "bluesky", id: "p1" }, }, ]; class StubSource implements ShardSource { readonly label = "stub"; constructor(private aliases: SearchAlias[] = [K_CUPS_ALIAS]) {} async loadAliases(): Promise { return this.aliases; } async loadGroups(): Promise { return { groups: STUB_GROUPS, defaultGroupId: "default" }; } async listChannels(): Promise { return [ { key: "chan-a", slug: "chan-a", name: "Channel A", groupId: "other" }, // "ghost" is not a defined group → folds onto the default group. { key: "chan-b", slug: "chan-b", name: "Channel B", groupId: "ghost" }, ]; } protected pages(ch: ChannelRef): TranscriptDetail[][] { // One record per page so paging exercises multiple shard pages. const recs = ch.slug === "chan-a" ? CHAN_A : CHAN_B; return recs.map((r) => [r]); } async transcriptsManifest(ch: ChannelRef): Promise { const pages = this.pages(ch); const slugToPage: Record = {}; pages.forEach((p, i) => (slugToPage[p[0].id] = i)); return { version: 4, channelSlug: ch.slug, pageCount: pages.length, maxPageBytes: 0, generatedAt: "", slugToPage, }; } async transcriptPage(ch: ChannelRef, page: number): Promise { return this.pages(ch)[page] ?? []; } publicOrigin(): string | null { return null; // a stub has no viewer origin } async subsManifest(ch: ChannelRef): Promise { // Only chan-b ships a (single-page) subs shard, with b1 on page 0. if (ch.slug !== "chan-b") return null; return { version: 1, channelSlug: ch.slug, pageCount: 1, maxPageBytes: 0, generatedAt: "", tracks: ["live_chat"], slugToPage: { b1: 0 }, }; } async subsPage(ch: ChannelRef, page: number): Promise { if (ch.slug !== "chan-b" || page !== 0) return []; const slug = "chan-b/b1"; return [ { slug, id: "b1", channelSlug: "chan-b", title: "Long one", uploadDate: "20240101", date: "2024-01-01", duration: "5:00", channel: "Channel B", isLivestream: false, ageRestricted: false, isDeleted: false, isUnlisted: false, platform: "youtube", webpageUrl: "https://example.test/b1", tracks: { live_chat: CHAT[slug] }, }, ]; } // Only chan-b ships posts; every other channel is video-only, which is // what a real mixed corpus looks like. async postsManifest(ch: ChannelRef): Promise { if (ch.slug !== "chan-b") return null; return { version: 1, channelSlug: ch.slug, pageCount: 1, maxPageBytes: 0, generatedAt: "", slugToPage: { p1: 0, p2: 0 }, }; } async postsPage(ch: ChannelRef, page: number): Promise { if (ch.slug !== "chan-b" || page !== 0) return []; return STUB_POSTS; } async availabilityMap(): Promise> { return new Map(Object.entries(AVAILABILITY)); } } // ─── (a) Paging: offset / limit / total / hasMore ─── test("paging: total is stable and hasMore/offset slice the full set", async () => { const src = new StubSource(); // Six "coffee" matches: a1–a5 (Channel A) + b1 (its cue says "coffee moment"), // in corpus order (channel A pages first, then B). const page1 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 0 }); assert.equal(page1.total, 6, "six coffee videos total"); assert.equal(page1.hits.length, 2); assert.deepEqual(page1.hits.map((h) => h.videoId), ["a1", "a2"]); assert.equal(page1.hasMore, true); const page2 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 2 }); assert.deepEqual(page2.hits.map((h) => h.videoId), ["a3", "a4"]); assert.equal(page2.hasMore, true); const page3 = await searchTranscripts(src, { query: "coffee", limit: 2, offset: 4 }); assert.deepEqual(page3.hits.map((h) => h.videoId), ["a5", "b1"]); assert.equal(page3.hasMore, false); assert.equal(page3.total, 6, "total unchanged across pages"); }); // ─── (b) Alias expansion ─── test("alias expansion: 'k cups' matches both the literal and the aliased spelling", async () => { const src = new StubSource(); const res = await searchTranscripts(src, { query: "k cups", limit: 20 }); const ids = res.hits.map((h) => h.videoId).sort(); assert.deepEqual(ids, ["a6", "a7"], "literal 'k cups' and aliased 'cake cup' both match"); assert.equal(res.firedAliases.length, 1); assert.equal(res.firedAliases[0].id, "k-cups"); }); test("alias expansion: use_aliases:false falls back to plain substring only", async () => { const src = new StubSource(); const res = await searchTranscripts(src, { query: "k cups", limit: 20, useAliases: false }); assert.deepEqual(res.hits.map((h) => h.videoId), ["a6"], "only the literal match survives"); assert.equal(res.firedAliases.length, 0); }); test("alias expansion: explicit regex disables alias expansion", async () => { const src = new StubSource(); const res = await searchTranscripts(src, { query: "cake cup", regex: true, limit: 20 }); assert.deepEqual(res.hits.map((h) => h.videoId), ["a7"]); assert.equal(res.firedAliases.length, 0); }); // ─── (c) include_snippets ─── test("include_snippets:false omits snippet text but keeps the match count", async () => { const src = new StubSource(); const withSnips = await searchTranscripts(src, { query: "coffee", limit: 1 }); assert.ok(withSnips.hits[0].snippets.length > 0, "snippets present by default"); const noSnips = await searchTranscripts(src, { query: "coffee", limit: 1, includeSnippets: false, }); assert.equal(noSnips.hits[0].snippets.length, 0, "no snippets"); assert.ok(noSnips.hits[0].matches >= 1, "match count still reported"); }); // ─── Scope: multi-channel + channel groups ─── test("scope: channels union scopes to both; an unknown channel is reported", async () => { const src = new StubSource(); const sel = await resolveSelectedChannels(src, { channels: ["chan-a", "chan-b", "chan-z"], }); assert.deepEqual( sel.channels.map((c) => c.key).sort(), ["chan-a", "chan-b"], "the two real channels are selected", ); assert.deepEqual(sel.unknownChannels, ["chan-z"], "the typo is surfaced"); assert.equal(sel.all, false); }); test("scope: group by id and by name expands to member channels", async () => { const src = new StubSource(); const byId = await resolveSelectedChannels(src, { group: "other" }); assert.deepEqual(byId.channels.map((c) => c.key), ["chan-a"]); assert.deepEqual(byId.matchedGroups.map((g) => g.id), ["other"]); const byName = await resolveSelectedChannels(src, { group: "Extended Universe" }); assert.deepEqual(byName.channels.map((c) => c.key), ["chan-a"], "name resolves the same group"); }); test("scope: an unknown/missing channel groupId folds onto the default group", async () => { const src = new StubSource(); // chan-b's "ghost" groupId isn't a defined group, so it resolves to default. const def = await resolveSelectedChannels(src, { group: "default" }); assert.deepEqual(def.channels.map((c) => c.key), ["chan-b"]); }); test("scope: an unknown group is reported, not silently a whole-corpus scan", async () => { const src = new StubSource(); const sel = await resolveSelectedChannels(src, { group: "nope" }); assert.deepEqual(sel.unknownGroups, ["nope"]); assert.deepEqual(sel.channels, []); assert.equal(sel.all, false); }); test("scope: channels + group combine as a deduped union", async () => { const src = new StubSource(); // chan-a named explicitly AND a member of group "other" → deduped to one. const overlap = await resolveSelectedChannels(src, { channel: "chan-a", group: "other", }); assert.deepEqual(overlap.channels.map((c) => c.key), ["chan-a"]); // chan-b explicit + group "other" (chan-a) → the union of both. const union = await resolveSelectedChannels(src, { channels: ["chan-b"], group: "other", }); assert.deepEqual(union.channels.map((c) => c.key).sort(), ["chan-a", "chan-b"]); }); test("scope: no selector means all channels (unchanged primitive behavior)", async () => { const src = new StubSource(); const sel = await resolveSelectedChannels(src, {}); assert.equal(sel.all, true); assert.deepEqual(sel.channels.map((c) => c.key).sort(), ["chan-a", "chan-b"]); }); test("search: a group scope restricts the scan to its members", async () => { const src = new StubSource(); // Group "other" is just chan-a → the five chan-a coffee videos, not b1. const res = await searchTranscripts(src, { query: "coffee", group: "other", limit: 20 }); assert.equal(res.total, 5); assert.ok(res.hits.every((h) => h.channelSlug === "chan-a")); assert.equal(res.selection.channelCount, 1); assert.deepEqual(res.selection.matchedGroups.map((g) => g.id), ["other"]); }); // ─── (d) getWindowedTranscript ─── test("getWindowedTranscript: windows a bounded region around the match", async () => { const rec = CHAN_B[0]; const { match } = buildMatcher({ query: "coffee" }); const { lines, matchCount } = getWindowedTranscript(rec, match, { before: 30, after: 30 }); assert.equal(matchCount, 1); const body = lines.join("\n"); // Cues within ±30s of the 100s match are kept… assert.ok(body.includes("right before the moment")); assert.ok(body.includes("here is the coffee moment")); assert.ok(body.includes("right after the moment")); // …and cues far outside the window are dropped. assert.ok(!body.includes("intro chatter")); assert.ok(!body.includes("much later unrelated")); assert.ok(!body.includes("the very end")); }); test("getWindowedTranscript: no matches yields no lines", async () => { const rec = CHAN_B[0]; const { match } = buildMatcher({ query: "banana" }); const { lines, matchCount } = getWindowedTranscript(rec, match); assert.equal(matchCount, 0); assert.equal(lines.length, 0); }); test("getWindowedTranscript: a stamp formatter renders the bracket contents", async () => { const rec = CHAN_B[0]; const { match } = buildMatcher({ query: "coffee" }); const { lines } = getWindowedTranscript(rec, match, { before: 30, after: 30, stamp: (clock, seconds) => `${clock}|${Math.floor(seconds)}`, }); assert.ok( lines.includes("[1:40|100] here is the coffee moment"), lines.join("\n"), ); }); test("getWindowedTranscript: maxLines keeps the earliest merged lines", async () => { const rec = CHAN_B[0]; const { match } = buildMatcher({ query: "coffee" }); const { lines, matchCount } = getWindowedTranscript(rec, match, { before: 30, after: 30, maxLines: 2, }); assert.equal(matchCount, 1, "matchCount is unaffected by the line cap"); assert.equal(lines.length, 2); assert.ok(lines[0].includes("right before the moment")); assert.ok(lines[1].includes("here is the coffee moment")); assert.ok(!lines.join("\n").includes("right after the moment")); }); // ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ─── async function allChannels(src: StubSource): Promise { return (await resolveSelectedChannels(src, {})).channels; } function leafTree( query: string, scope: "transcripts" | "chat" | "metadata" | "description" | "tags", extra: Record = {}, ) { return newGroup({ children: [newLeaf({ query, scope, ...extra })] }); } async function specIds( src: StubSource, tree: ReturnType, spec: { filters?: SearchFilters | null; aliases?: SearchAlias[] } = {}, ): Promise { const res = await runSearchSpec(src, await allChannels(src), { tree, ...spec }, { limit: 50 }); return res.hits.map((h) => h.videoId).sort(); } test("spec transcripts leaf: matches cue text (same set as the plain scanner)", async () => { const src = new StubSource(); assert.deepEqual(await specIds(src, leafTree("coffee", "transcripts")), [ "a1", "a2", "a3", "a4", "a5", "b1", ]); }); test("spec metadata leaf: matches title/channel, not cues", async () => { const src = new StubSource(); // "kcup" is in a6's title but no cue (cue says "k cups" with a space). assert.deepEqual(await specIds(src, leafTree("kcup", "metadata")), ["a6"]); }); test("spec description leaf: matches the description only", async () => { const src = new StubSource(); // "brewing" is only in a1's description, in no cue. assert.deepEqual(await specIds(src, leafTree("brewing", "description")), ["a1"]); }); test("spec tags leaf: matches the joined tags", async () => { const src = new StubSource(); assert.deepEqual(await specIds(src, leafTree("espresso", "tags")), ["a1"]); }); test("spec chat leaf: matches live_chat cues fetched from subs shards", async () => { const src = new StubSource(); // "banana" only appears in b1's live chat, in no transcript cue. const res = await runSearchSpec(src, await allChannels(src), { tree: leafTree("banana", "chat"), }, { limit: 50 }); assert.deepEqual(res.hits.map((h) => h.videoId), ["b1"]); assert.equal(res.hits[0].snippets[0].scope, "chat"); assert.equal(res.hits[0].snippets[0].track, "live_chat"); }); test("spec AND: transcripts AND tags narrows to the intersection", async () => { const src = new StubSource(); const tree = newGroup({ op: "AND", children: [ newLeaf({ query: "coffee", scope: "transcripts" }), newLeaf({ query: "espresso", scope: "tags" }), ], }); assert.deepEqual(await specIds(src, tree), ["a1"]); }); test("spec OR: description OR chat unions the two", async () => { const src = new StubSource(); const tree = newGroup({ op: "OR", children: [ newLeaf({ query: "brewing", scope: "description" }), newLeaf({ query: "banana", scope: "chat" }), ], }); assert.deepEqual(await specIds(src, tree), ["a1", "b1"]); }); test("spec negate: coffee AND NOT (espresso tag) drops a1", async () => { const src = new StubSource(); const tree = newGroup({ op: "AND", children: [ newLeaf({ query: "coffee", scope: "transcripts" }), newLeaf({ query: "espresso", scope: "tags", negate: true }), ], }); assert.deepEqual(await specIds(src, tree), ["a2", "a3", "a4", "a5", "b1"]); }); test("spec filter ft: livestreams:false drops the livestream (a3)", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, livestreams: false }, }); assert.deepEqual(ids, ["a1", "a2", "a4", "a5", "b1"]); }); test("spec filter fa: keep only age-restricted → a4", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, allAges: false }, }); assert.deepEqual(ids, ["a4"]); }); test("spec filter tg: a curated tag keeps only the videos carrying it", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, curatedTags: ["eva-collab"] }, }); assert.deepEqual(ids, ["a1"]); }); test("spec filter tg: several tags are ORed, never ANDed", async () => { // a1 carries eva-collab, a3 carries eva-in-chat and neither carries both: // an AND here would return nothing, which is the bug this pins. const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, curatedTags: ["eva-collab", "eva-in-chat"] }, }); assert.deepEqual(ids, ["a1", "a3"]); }); test("spec filter tg: an untagged record carries none and passes none", async () => { // Every record on a site built before corpus spec 4 looks like this, so the // pre-spec-4 path is "an honest empty result", not "everything matches". const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, curatedTags: ["never-assigned"] }, }); assert.deepEqual(ids, []); }); test("spec filter fav: available-only drops every missing state", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: keep("available"), }); assert.deepEqual(ids, ["b1"]); }); test("spec filter fav: missing-only drops the available b1", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: keep(...MISSING_STATES), }); assert.deepEqual(ids, ["a1", "a2", "a3", "a4", "a5"]); }); // Each leaf on its own — the point of the enum is that these are five distinct // states, not one "gone" bucket. for (const [state, id] of [ ["deleted", "a1"], ["unlisted", "a2"], ["private", "a3"], ["members_only", "a4"], ["maybe_missing", "a5"], ] as const) { test(`spec filter fav: ${state}-only keeps just ${id}`, async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: keep(state), }); assert.deepEqual(ids, [id]); }); } // A pre-existing share link (fv=1) knew three buckets. Its `a` token has to // keep unconfirmed videos too: they read as available when the link was // written, so dropping them would silently narrow somebody's saved search. test("spec filter fav: a legacy fv=1 'available' link keeps maybe_missing", async () => { const sel = parseShareV1("?fv=1&fav=a", []); const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, states: sel.states }, }); assert.deepEqual(ids, ["a5", "b1"]); }); test("spec filter fav: a legacy fv=1 'deleted' link also keeps private/members", async () => { const sel = parseShareV1("?fv=1&fav=d", []); const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, states: sel.states }, }); assert.deepEqual(ids, ["a1", "a3", "a4"]); }); test("spec filter dates: fdf bound keeps only the later upload (a5)", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { filters: { ...KEEP_ALL, dateFrom: "20250101" }, }); assert.deepEqual(ids, ["a5"]); }); test("spec paging/total: stable total with an offset/limit slice", async () => { const src = new StubSource(); const chans = await allChannels(src); const tree = leafTree("coffee", "transcripts"); const page = await runSearchSpec(src, chans, { tree }, { limit: 2, offset: 2 }); assert.equal(page.total, 6); assert.deepEqual(page.hits.map((h) => h.videoId), ["a3", "a4"]); assert.equal(page.hasMore, true); }); test("spec aliases: a transcripts leaf expands via curated aliases", async () => { const src = new StubSource(); const res = await runSearchSpec( src, await allChannels(src), { tree: leafTree("k cups", "transcripts"), aliases: await src.loadAliases() }, { limit: 50 }, ); assert.deepEqual(res.hits.map((h) => h.videoId).sort(), ["a6", "a7"]); assert.equal(res.firedAliases.length, 1); assert.equal(res.firedAliases[0].id, "k-cups"); }); test("spec hits carry slug + platform for moment links", async () => { const src = new StubSource(); const res = await runSearchSpec(src, await allChannels(src), { tree: leafTree("coffee", "transcripts"), }, { limit: 1 }); assert.equal(res.hits[0].slug, "chan-a/a1"); assert.equal(res.hits[0].platform, "youtube"); assert.ok(res.hits[0].snippets[0].seconds >= 0); }); // ─── End-to-end through the MCP server (tools + prompts + missing ids) ─── async function connectClient(source: ShardSource): Promise { const server = createServer(source); const [clientTransport, serverTransport] = InMemoryTransport.createLinkedPair(); const client = new Client({ name: "test", version: "0" }, { capabilities: {} }); await Promise.all([ server.connect(serverTransport), client.connect(clientTransport), ]); return client; } function firstText(res: unknown): string { const content = (res as { content: { type: string; text: string }[] }).content; return content.map((c) => c.text).join("\n"); } test("server: search_transcripts footer reports total, has_more, and the alias", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "search_transcripts", arguments: { query: "k cups", limit: 20 }, }); const out = firstText(res); assert.match(out, /total 2 match/); assert.match(out, /has_more: no/); assert.match(out, /scope: whole corpus/); assert.match(out, /expanded via alias K-Cups/); await client.close(); }); test("server: search_transcripts footer names a group scope and its channel count", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", group: "other", limit: 20 }, }); const out = firstText(res); assert.match(out, /total 5 match/); assert.match(out, /scope: group Extended Universe \(1 channel/); await client.close(); }); test("server: search_transcripts footer flags a typo'd group as unknown", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", group: "nope", limit: 20 }, }); const out = firstText(res); assert.match(out, /unknown group\(s\): nope/); await client.close(); }); test("server: list_channels groups channels under headers and lists the groups", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "list_channels", arguments: {} }); const out = firstText(res); // Group headers (blank-named default falls back to its id). assert.match(out, /### Extended Universe \(1 channel\(s\); not selected by default\)/); assert.match(out, /### default \(1 channel\(s\)\)/); assert.match(out, /Channel A/); assert.match(out, /Channel B/); // The compact selector cheat-sheet. assert.match(out, /Groups \(select by id or name\)/); assert.match(out, /other · Extended Universe · 1 channel\(s\) · not selected by default/); await client.close(); }); test("server: get_transcripts windows with a query and reports missing ids", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1", "nope"], query: "coffee" }, }); const out = firstText(res); assert.match(out, /here is the coffee moment/); assert.ok(!out.includes("intro chatter"), "far cues excluded by the window"); assert.match(out, /not found: nope/); await client.close(); }); test("server: get_transcripts without a query returns full markdown", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1"] }, }); const out = firstText(res); assert.match(out, /## Transcript/); assert.ok(out.includes("intro chatter"), "full transcript includes every cue"); assert.ok(out.includes("the very end")); await client.close(); }); test("server: the sweep prompt lists with its arguments and renders the query", async () => { const client = await connectClient(new StubSource()); const list = await client.listPrompts(); const sweep = list.prompts.find((p) => p.name === "sweep"); assert.ok(sweep, "sweep prompt is listed"); const names = (sweep!.arguments ?? []).map((a) => a.name); assert.deepEqual(names.sort(), [ "batch_size", "channel", "channels", "directive", "group", "link", "parse_model", "query", "report_path", ]); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups", channel: "chan-a" }, }); const msg = got.messages[0].content; assert.equal(msg.type, "text"); assert.match((msg as { text: string }).text, /"k cups"/); assert.match((msg as { text: string }).text, /chan-a/); await client.close(); }); test("server: the sweep prompt reflects a supplied group scope", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups", group: "other" }, }); const text = (got.messages[0].content as { text: string }).text; assert.match(text, /scoped to groups "other"/); assert.match(text, /groups: \["other"\]/); await client.close(); }); test("server: the sweep prompt with no scope instructs a group/channel pick first", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups" }, }); const text = (got.messages[0].content as { text: string }).text; assert.match(text, /Choose the scope first/); assert.match(text, /list_channels/); assert.match(text, /confirm \*\*all\*\*/); await client.close(); }); test("server: the sweep prompt mandates linked citations and subagent batches", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups", channel: "chan-a" }, }); const text = (got.messages[0].content as { text: string }).text; // Feature 1: linked citation form (not the old "title + [mm:ss]"). assert.match(text, /\[title @ mm:ss\]\(\)/); assert.ok(!/title \+ \[mm:ss\]/.test(text), "old bare citation form is gone"); // Feature 2: subagent map-reduce + inline fallback. assert.match(text, /Spawn a subagent/); assert.match(text, /Task tool/); assert.match(text, /Fallback/); await client.close(); }); test("server: the sweep prompt accepts a link= seed and drives open_link", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { link: "https://site.example/?q=coffee" }, }); const text = (got.messages[0].content as { text: string }).text; assert.match(text, /open_link/); // One call now does the whole job; dry_run is the opt-in preview. assert.match(text, /dry_run: true/); assert.ok(!text.includes("apply:true"), "the two-phase apply is gone"); assert.match(text, /https:\/\/site\.example\/\?q=coffee/); // A link-only sweep needs no query argument. assert.match(text, /share link/); await client.close(); }); // ─── link_style:"base" — compact stamps, moment_base, max_lines, worklist trim ─── test("server: get_transcripts link_style base emits moment_base + compact stamps", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1"], query: "coffee", link_style: "base" }, }); const out = firstText(res); // StubSource has no viewer origin → the platform (YouTube-param) base. assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m); assert.match(out, /\[1:40\|100\] here is the coffee moment/); assert.ok(!out.includes("](http"), "no full inline links in base excerpts"); assert.match(out, //, "the expansion note is present"); await client.close(); }); test("server: get_transcripts default output stays inline-linked (regression pin)", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1"], query: "coffee" }, }); const out = firstText(res); assert.ok(out.includes("](https://example.test/b1?t=100s)"), out); assert.ok(!out.includes("moment_base"), "no base artifacts in inline mode"); await client.close(); }); test("server: get_transcripts base full transcript uses compact stamps + header base", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1"], link_style: "base" }, }); const out = firstText(res); assert.match(out, /\[0:00\|0\] intro chatter/); assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m); assert.ok(!out.includes("](http"), "no inline links in base full transcripts"); await client.close(); }); test("server: get_transcripts max_lines caps the merged excerpt lines", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b1"], query: "coffee", max_lines: 1 }, }); const out = firstText(res); assert.match(out, /1 matching line\(s\), windowed/); assert.ok(out.includes("right before the moment"), "the earliest line is kept"); assert.ok(!out.includes("right after the moment"), "later lines are dropped"); await client.close(); }); test("server: search_transcripts link_style base emits moment_base + compact snippets", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", link_style: "base", limit: 20 }, }); const out = firstText(res); assert.match(out, /- moment_base: https:\/\/example\.test\/a1\?t=$/m); assert.ok(out.includes(" - [0:10|10] i love coffee"), out); assert.ok(!out.includes("](http"), "no full inline links in base snippets"); assert.match(out, //, "the expansion note is present"); await client.close(); }); test("server: search_transcripts worklist trims the source line (and no moment_base)", async () => { const client = await connectClient(new StubSource()); const withSnips = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", limit: 5 }, }), ); assert.match(withSnips, /- source: /, "source line present by default"); const worklist = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", limit: 5, include_snippets: false, link_style: "base", }, }), ); assert.ok(!worklist.includes("- source:"), "worklist mode drops the source line"); assert.ok(!worklist.includes("moment_base"), "…and emits no moment_base either"); await client.close(); }); test("server: the sweep prompt drives dumb extractors on the cheap model", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups", channel: "chan-a" }, }); const text = (got.messages[0].content as { text: string }).text; assert.match(text, /haiku/, "the default parse model is requested"); assert.match(text, /DUMB EXTRACTOR/); assert.match(text, /link_style/); assert.match(text, /moment_base/); assert.match(text, /\[mm:ss\|seconds\]/); assert.match(text, /40 lines/, "the extractor's hard line budget"); assert.match(text, //, "the expansion rule is spelled out"); await client.close(); }); test("server: the sweep prompt honors parse_model", async () => { const client = await connectClient(new StubSource()); const got = await client.getPrompt({ name: "sweep", arguments: { query: "k cups", channel: "chan-a", parse_model: "sonnet" }, }); const text = (got.messages[0].content as { text: string }).text; assert.match(text, /sonnet/); assert.ok(!/haiku/.test(text), "the default model name is fully replaced"); await client.close(); }); // ─── the social-post corpus ─── test("posts: default content_types searches BOTH corpora", async () => { const src = new StubSource(); // "zephyrpost" only exists in the posts fixtures, so a default search finding // it proves posts are covered without an opt-in. const r = await searchTranscripts(src, { query: "zephyrpost" }); assert.equal(r.total, 2); assert.ok(r.hits.every((h) => h.contentType === "post")); assert.deepEqual( r.hits.map((h) => h.videoId).sort(), ["p1", "p2"], ); }); test("posts: a post hit carries author/date and no timestamps", async () => { const src = new StubSource(); const r = await searchTranscripts(src, { query: "zephyrpost mentioning" }); assert.equal(r.total, 1); const hit = r.hits[0]; assert.equal(hit.contentType, "post"); assert.equal(hit.author, "Tester"); assert.equal(hit.uploadDate, "20260301"); assert.equal(hit.platform, "bluesky"); assert.match(hit.webpageUrl ?? "", /bsky\.app/); // A post has no timeline: seconds stay 0 so nothing can render an @ mm:ss. assert.ok(hit.snippets.every((sn) => sn.seconds === 0)); }); test("posts: content_types can narrow to one corpus", async () => { const src = new StubSource(); const videosOnly = await searchTranscripts(src, { query: "zephyrpost", contentTypes: ["video"], }); assert.equal(videosOnly.total, 0); const postsOnly = await searchTranscripts(src, { query: "coffee", contentTypes: ["post"], }); assert.equal(postsOnly.total, 0, "coffee only appears in transcripts"); }); test("posts: a posts-scope spec leaf matches post text", async () => { const src = new StubSource(); const root = newGroup({ op: "AND", children: [newLeaf({ id: "l1", query: "zephyrpost", scope: "posts" })], }); const channels = await src.listChannels(); const r = await runSearchSpec(src, channels, { tree: root }); assert.equal(r.total, 2); assert.deepEqual(r.hits.map((h) => h.videoId).sort(), ["p1", "p2"]); }); test("posts: a tag filter takes the posts corpus out of the search", async () => { // Curated tags live on video records, so every post fails a tag filter. // Scanning them to drop them all costs a manifest probe and a shard read per // posting channel — and counting them before the drop is the bug the export // viewer had (tag-chips.spec.ts). The corpus is skipped, and the result says // so rather than leaving a caller to read "0 posts" as "searched, unmatched". const src = new StubSource(); const r = await searchTranscripts(src, { query: "zephyrpost", filters: { ...KEEP_ALL, curatedTags: ["eva-collab"] }, }); assert.equal(r.total, 0); assert.equal(r.postsScanned.skippedForTagFilter, true); assert.equal(r.postsScanned.requested, false); assert.equal(r.postsScanned.channels, 0, "no manifest was even probed"); }); test("posts: without a tag filter the same query still searches them", async () => { // The control: posts are silenced BY the tag filter, never by this change. const src = new StubSource(); const r = await searchTranscripts(src, { query: "zephyrpost", filters: { ...KEEP_ALL }, }); assert.equal(r.total, 2); assert.equal(r.postsScanned.skippedForTagFilter, false); assert.equal(r.postsScanned.requested, true); }); test("posts: a posts-scope spec leaf under a tag filter matches nothing", async () => { // The spec path is the other half: an explicit posts leaf carries its own // slug set, so the default-scope exclusion above does not cover it. const src = new StubSource(); const root = newGroup({ op: "AND", children: [newLeaf({ id: "l1", query: "zephyrpost", scope: "posts" })], }); const channels = await src.listChannels(); const r = await runSearchSpec(src, channels, { tree: root, filters: { ...KEEP_ALL, curatedTags: ["eva-collab"] }, }); assert.equal(r.total, 0); }); test("posts: findPost resolves by id and getThread returns the whole thread", async () => { const src = new StubSource(); const found = await findPost(src, "p2"); assert.ok(found, "reply is findable by its own id"); assert.equal(found!.post.id, "p2"); assert.equal(found!.post.isReply, true); const thread = await getThread(src, found!.ch, found!.post); // Root + reply, oldest first. assert.deepEqual(thread.map((p) => p.id), ["p1", "p2"]); }); test("posts: a forum post's thread is its conversation, not the whole forum thread", async () => { // SYNTHETIC forum posts: 3 quotes 2, 4 quotes 3; 5 is unrelated. const mk = (id: string, position: number, quotes: string[] = []): Post => ({ id, slug: `forum/${id}`, channelSlug: "forum", author: `Member${position}`, createdAt: new Date(Date.UTC(2026, 0, 1, 0, position)).toISOString(), uploadDate: "20260101", text: `Post ${position}.`, url: `https://forum.example/posts/${id}/`, platform: "xenforo", isReply: position > 1, isRepost: false, links: [], forum: { host: "forum.example", threadId: "1", position, ...(quotes.length ? { quotes: quotes.map((postId) => ({ postId })) } : {}), }, }); const posts = [mk("5", 5), mk("4", 4, ["3"]), mk("3", 3, ["2"]), mk("2", 2), mk("1", 1)]; const source = { async postsManifest() { return { version: 1, channelSlug: "forum", pageCount: 1, maxPageBytes: 0, generatedAt: "", slugToPage: {} }; }, async postsPage() { return posts; }, } as unknown as Parameters[0]; const thread = await getThread(source, { slug: "forum" } as Parameters[1], posts[2]); assert.deepEqual(thread.map((p) => p.id), ["2", "3", "4"]); }); test("server: get_post returns the post with no timestamps", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_post", arguments: { post_id: "p1" }, }); const out = firstText(res); assert.match(out, /Post by Tester/); assert.match(out, /a zephyrpost mentioning something/); assert.match(out, /post_id: p1/); // The ISO `posted:` line legitimately contains a time; what must NOT appear // is a citable clock marker — a bracketed [m:ss] stamp or an `@ mm:ss`. assert.doesNotMatch(out, /\[\d+:\d{2}/, "no bracketed cue stamps in a post"); assert.doesNotMatch(out, /@\s*\d+:\d{2}/, "no @ mm:ss moment in a post"); await client.close(); }); test("server: get_post names the post's archive page when the source has a viewer", async () => { // A stub has no viewer origin: no archive line. const bare = await connectClient(new StubSource()); assert.doesNotMatch(firstText(await bare.callTool({ name: "get_post", arguments: { post_id: "p1" } })), /- archive:/); await bare.close(); // A site with one: its post modal for /. class SiteSource extends StubSource { publicOrigin(): string | null { return "https://site.example"; } } const site = await connectClient(new SiteSource()); const out = firstText(await site.callTool({ name: "get_post", arguments: { post_id: "p1" } })); const line = /- archive: (\S+)/.exec(out); assert.ok(line, "archive line present"); const u = new URL(line![1]); assert.equal(u.origin, "https://site.example"); assert.equal(u.searchParams.get("vm"), "post"); assert.match(u.searchParams.get("v") ?? "", /\/p1$/); await site.close(); }); test("server: get_thread returns parent + replies", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_thread", arguments: { post_id: "p1" }, }); const out = firstText(res); assert.match(out, /Thread \(2 posts\)/); assert.match(out, /a zephyrpost mentioning something/); assert.match(out, /a zephyrpost reply in the same thread/); await client.close(); }); test("server: search_transcripts renders a post hit without a moment link", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "search_transcripts", arguments: { query: "zephyrpost", limit: 20 }, }); const out = firstText(res); assert.match(out, /post by Tester/); assert.match(out, /posted:/); assert.doesNotMatch(out, /moment_base/); await client.close(); }); // ─── Server-level coverage for the single-record reads and open_link ─── // // These three had no server-level test at all, and open_link's default just // changed from "preview" to "decode and search in one call" — exactly the kind // of change that needs a test underneath it before it lands. // A registry whose every spec resolves to the same StubSource, so a share // link's origin can be resolved without a live fetch. function stubRegistry(): SourceRegistry { return new SourceRegistry( { kind: "remote", url: "https://site.example" }, { build: () => new StubSource(), probe: async () => "remote" as const, }, ); } async function connectRegistry(): Promise { const server = createServer(stubRegistry()); const [ct, st] = InMemoryTransport.createLinkedPair(); const client = new Client({ name: "test", version: "0" }, { capabilities: {} }); await Promise.all([server.connect(st), client.connect(ct)]); return client; } test("server: get_transcript returns the full transcript as markdown", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "get_transcript", arguments: { video_id: "a1" }, }), ); assert.match(out, /Coffee one/); assert.match(out, /i love coffee/); assert.match(out, /\(corpus: /, "every result names its corpus"); await client.close(); }); test("server: get_transcript reports a missing id as an error", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_transcript", arguments: { video_id: "nope" }, }); assert.equal((res as { isError?: boolean }).isError, true); assert.match(firstText(res), /video not found: nope/); await client.close(); }); test("server: get_video_metadata returns metadata without the cue bodies", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" }, }), ); const parsed = JSON.parse(out.slice(0, out.lastIndexOf("}") + 1)); assert.equal(parsed.id, "a1"); assert.equal(parsed.title, "Coffee one"); assert.equal(parsed.channelName, "Channel A"); assert.equal(parsed.cueCount, 1); assert.equal(parsed.cues, undefined, "the transcript body is not included"); await client.close(); }); test("server: get_video_metadata reports a missing id as an error", async () => { const client = await connectClient(new StubSource()); const res = await client.callTool({ name: "get_video_metadata", arguments: { video_id: "nope" }, }); assert.equal((res as { isError?: boolean }).isError, true); assert.match(firstText(res), /video not found: nope/); await client.close(); }); // The "## Stats" block is read straight off the archive's stats pages, so its // "covers only N% — truncated" warning is exactly as good as the stat. Before // stats schema 6 the stat of a video transcribed after it was first indexed // stayed at "no transcript, coverage 0" for good, and this warned that a // complete transcript was truncated. buildStats.test.ts (a) pins the stat now // being recomputed; this pins what the tool then says about it. function statFor( record: TranscriptDetail, s: Pick, ): VideoStat { return { slug: record.slug, id: record.id, channelSlug: record.channelSlug, channel: "Channel A", title: record.title, platform: "youtube", uploadDate: record.uploadDate, downloadedDate: "20260711", timestamp: null, duration: 3600, viewCount: null, likeCount: null, commentCount: null, channelFollowerCount: null, categories: [], tags: [], language: null, isLivestream: false, mediaType: "video", status: "available", ...s, }; } class StatsStubSource extends StubSource { constructor(private stat: VideoStat) { super(); } async statsIndex(): Promise> { return new Map([[this.stat.slug, this.stat]]); } } test("server: get_video_metadata on a recomputed stat reports the cues and no truncation", async () => { const a1 = CHAN_A[0]; const client = await connectClient( new StatsStubSource( statFor(a1, { hasTranscript: true, cueCount: 7461, coverage: 0.998, transcribedDate: "20260918" }), ), ); const out = firstText( await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }), ); assert.match(out, /## Stats/); assert.match(out, /- transcript cues: 7461/); assert.match(out, /- transcribed: /); assert.doesNotMatch(out, /COVERS ONLY/); await client.close(); }); test("server: get_video_metadata still warns for a transcript that really stops early", async () => { const a1 = CHAN_A[0]; const client = await connectClient( new StatsStubSource( statFor(a1, { hasTranscript: true, cueCount: 900, coverage: 0.41, transcribedDate: "20260918" }), ), ); const out = firstText( await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }), ); assert.match(out, /TRANSCRIPT COVERS ONLY 41% OF THE RUNTIME/); await client.close(); }); test("server: open_link decodes AND searches in one call, returning the handle", async () => { const client = await connectRegistry(); const out = firstText( await client.callTool({ name: "open_link", arguments: { link: "https://site.example/?q=coffee" }, }), ); // The plan… assert.match(out, /## Share link/); assert.match(out, /Corpus handle: `remote:https:\/\/site\.example`/); // …and the results, from the same call. assert.match(out, /Coffee one/); assert.match(out, /video\(s\) matched/); await client.close(); }); test("server: open_link dry_run returns the plan and runs no search", async () => { const client = await connectRegistry(); const out = firstText( await client.callTool({ name: "open_link", arguments: { link: "https://site.example/?q=coffee", dry_run: true }, }), ); assert.match(out, /Share-link plan \(dry run\)/); assert.match(out, /no search was performed/); assert.ok(!out.includes("video(s) matched"), "nothing was searched"); await client.close(); }); test("server: open_link honours overrides", async () => { const client = await connectRegistry(); const out = firstText( await client.callTool({ name: "open_link", arguments: { link: "https://site.example/?q=coffee", overrides: { query: "k cups" }, dry_run: true, }, }), ); assert.match(out, /query replaced with "k cups"/); await client.close(); }); test("server: open_link accepts a posts query_scope override", async () => { const client = await connectRegistry(); const out = firstText( await client.callTool({ name: "open_link", arguments: { link: "https://site.example/?q=coffee", overrides: { query: "bluesky", query_scope: "posts" }, dry_run: true, }, }), ); // Declared in the schema but dropped by parseOverrides until now, so a // caller asking for the post corpus silently got transcripts. assert.match(out, /posts/); await client.close(); }); test("server: open_link rejects a link that is not a URL", async () => { const client = await connectRegistry(); const res = await client.callTool({ name: "open_link", arguments: { link: "not a url" }, }); assert.equal((res as { isError?: boolean }).isError, true); await client.close(); }); // ─── W4: coverage is mechanical, and silently-ignored parameters are not ─── test("server: enumerate_matches returns the whole set in one call", async () => { const client = await connectClient(new StubSource()); const enumerated = firstText( await client.callTool({ name: "enumerate_matches", arguments: { query: "coffee", batch_size: 2 }, }), ); // The complete worklist, with the batch arithmetic already done. assert.match(enumerated, /the complete worklist/); assert.match(enumerated, /complete set: yes/); assert.match(enumerated, /batch\(es\) of 2/); assert.match(enumerated, /- a1 \| video \| Channel A \|/); assert.ok(!enumerated.includes("i love coffee"), "no snippet bodies"); // And its total agrees with what search reports — the invariant a sweep // relies on when it claims coverage. const searched = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", limit: 1 }, }), ); const enumTotal = /(\d+) match\(es\) for "coffee"/.exec(enumerated)?.[1]; const searchTotal = /total (\d+) match/.exec(searched)?.[1]; assert.equal(enumTotal, searchTotal); await client.close(); }); test("server: enumerate_matches with no matches says so plainly", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "enumerate_matches", arguments: { query: "zzzznotathing" }, }), ); assert.match(out, /No matches for "zzzznotathing"/); assert.match(out, /complete set: yes/); await client.close(); }); test("server: enumerate_matches shouts when a cap made coverage partial", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "enumerate_matches", arguments: { query: "coffee", max_pages: 1 }, }), ); // The first line, not a footer note. assert.ok(out.startsWith("⚠ COVERAGE PARTIAL"), out.slice(0, 60)); assert.match(out, /are a SAMPLE, not the full set/); assert.match(out, /a PARTIAL worklist/); assert.ok(!out.includes("the complete worklist")); assert.match(out, /complete set: NO — capped/); await client.close(); }); test("server: an incomplete search page warns ABOVE the hits", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", limit: 2 }, }), ); assert.ok(out.startsWith("⚠ INCOMPLETE PAGE"), out.slice(0, 60)); assert.match(out, /Do NOT report a count/); assert.match(out, /enumerate_matches/); // The old footer is still there, so existing consumers keep working. assert.match(out, /has_more: yes/); await client.close(); }); test("server: a complete search page carries no banner", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", limit: 50 }, }), ); assert.ok(!out.includes("INCOMPLETE PAGE")); assert.match(out, /has_more: no/); await client.close(); }); test("server: an empty posts corpus is named as empty, not as zero pages", async () => { const client = await connectClient(new StubSource()); // chan-a is video-only, so a posts-scoped search there has no index at all. const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "anything", channels: ["chan-a"], content_types: ["post"], }, }), ); assert.match(out, /the post corpus is EMPTY here, not merely unmatched/); await client.close(); }); test("server: a posts corpus that WAS searched reports its own scan counts", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "zephyrpost", channels: ["chan-b"], content_types: ["post"], }, }), ); assert.match(out, /posts: scanned \d+ page\(s\) across 1 posting channel\(s\)/); assert.ok(!out.includes("EMPTY here")); await client.close(); }); test("server: get_transcripts queries[] merges windows and counts each query", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["a1"], queries: ["coffee", "zzzznotathing"] }, }), ); // Both queries are reported per video, so the one that matched nothing is // visible rather than silently absorbed into the merged excerpt. assert.match(out, /"coffee": 1/); assert.match(out, /"zzzznotathing": 0/); assert.match(out, /i love coffee/); assert.match(out, /queries: "coffee", "zzzznotathing"/); await client.close(); }); test("server: get_transcripts content_types reaches the post corpus", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["p1"] }, }), ); // p1 is a post id; the declared content_types was never read before, so // this used to come back "not found". assert.match(out, /zephyrpost/); assert.ok(!out.includes("not found: p1")); await client.close(); }); test("server: get_transcripts video-only scope does not fall through to posts", async () => { const client = await connectClient(new StubSource()); const out = firstText( await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["p1"], content_types: ["video"] }, }), ); assert.match(out, /not found: p1/); await client.close(); }); test("server: the unadvertised channel/group singulars are still parsed", async () => { const client = await connectClient(new StubSource()); // Dropped from the advertised schema in favour of the plural forms, but a // stored habit (or an old transcript) must not hard-fail. const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "coffee", channel: "chan-a" }, }), ); assert.match(out, /scope: 1 channel/); await client.close(); }); // ─── Alternate tracks (lib/captionTracks.ts) ─── // chan-b plus b2: an en-orig primary and an uploaded `en` that says a word the // primary never does. const ALT_REC = vid( "b2", "Two tracks", "chan-b", cues([5, "the harbor bridge opened"]), { track: "en-orig", altTracks: [{ track: "en", cues: cues([6, "the harbor bridge opened"], [90, "a zeppelin flew over"]) }], }, ); class AltTrackSource extends StubSource { protected override pages(ch: ChannelRef): TranscriptDetail[][] { const base = super.pages(ch); return ch.slug === "chan-b" ? [...base, [ALT_REC]] : base; } } test("searchTranscripts: a word only an alternate track holds is found there, the track named", async () => { const src = new AltTrackSource(); const res = await searchTranscripts(src, { query: "zeppelin" }); assert.equal(res.hits.length, 1); assert.equal(res.hits[0].videoId, "b2"); assert.deepEqual( res.hits[0].snippets.map((s) => [s.seconds, s.track]), [[90, "en"]], ); // Said by both tracks at the same moment: once, from the primary. const both = await searchTranscripts(src, { query: "harbor bridge" }); assert.deepEqual(both.hits[0].snippets.map((s) => s.track), [undefined]); }); test("server: a hit from an alternate track says which one", async () => { const client = await connectClient(new AltTrackSource()); const out = firstText( await client.callTool({ name: "search_transcripts", arguments: { query: "zeppelin" } }), ); assert.match(out, /\[in uploaded captions \[1:30\]\(https:\/\/example.test\/b2\?t=90s\)\] a zeppelin flew over/); await client.close(); }); test("server: get_transcript reads the primary by default and another track by `track`", async () => { const client = await connectClient(new AltTrackSource()); const primary = firstText( await client.callTool({ name: "get_transcript", arguments: { video_id: "b2" } }), ); assert.doesNotMatch(primary, /zeppelin/); assert.match(primary, /track: en-orig \(original audio captions\) — the primary/); assert.match(primary, /other tracks: en \(uploaded captions\) — pass track to read one/); const en = firstText( await client.callTool({ name: "get_transcript", arguments: { video_id: "b2", track: "en" } }), ); assert.match(en, /a zeppelin flew over/); assert.match(en, /track: en \(uploaded captions\)/); const bad = await client.callTool({ name: "get_transcript", arguments: { video_id: "b2", track: "en-GB" }, }); assert.match(firstText(bad), /has no track "en-GB"; its tracks are: en-orig \(original audio captions\), en \(uploaded captions\)/); // get_transcripts with a query windows a match only the alternate holds. const batch = firstText( await client.callTool({ name: "get_transcripts", arguments: { video_ids: ["b2"], query: "zeppelin" }, }), ); assert.match(batch, /1 matching line\(s\) only in uploaded captions \(track en\), windowed/); assert.match(batch, /a zeppelin flew over/); await client.close(); });