// Tests for the four things the speed/honesty work added, each aimed at the // specific failure it was built to prevent: // // 1. the page cache — a batch of ids on one shard page must parse it ONCE // 2. filter-first planning — a filtered query must read only the pages that // can hold a match, and must never skip a page it // is not certain about // 3. duplicate collapsing — a recording mirrored across platforms counts once // 4. cap honesty — a truncated scan must NAME the channels it never // reached, because it truncates in channel order // // The page-cache test drives a real LocalSource over a temp dir, because the // cache lives in the source implementations — a stub would prove nothing. import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdtemp, mkdir, writeFile, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest"; import type { ChannelSubsManifest } from "yt-dlp-transcript-common/lib/manifest"; import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts"; import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs"; import type { ChannelPostsManifest, Post } from "yt-dlp-transcript-common/lib/posts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; import { VIDEO_STATES } from "yt-dlp-transcript-common/lib/availability"; import { LocalSource, type ChannelGroups, type ChannelRef, type DuplicateIndex, type ShardSource, type VideoAvailability, type VideoIndex, } from "./source"; import { buildScanPlan, searchTranscripts, type SearchFilters } from "./search"; // ─── fixtures ─── function cues(...pairs: [number, string][]): Cue[] { return pairs.map(([start, text]) => ({ start, end: start + 3, text })); } function video( channelSlug: string, id: string, opts: { uploadDate?: string; isLivestream?: boolean; deleted?: boolean; text?: string; curatedTags?: string[]; } = {}, ): TranscriptDetail { return { id, slug: `${channelSlug}/${id}`, channelSlug, channel: channelSlug, title: `video ${id}`, uploadDate: opts.uploadDate ?? "20240601", duration: 600, isLivestream: opts.isLivestream ?? false, ageRestricted: false, platform: "youtube", webpageUrl: `https://www.youtube.com/watch?v=${id}`, description: "", tags: [], ...(opts.curatedTags ? { curatedTags: opts.curatedTags } : {}), cues: cues([10, opts.text ?? "they filed a lawsuit today"]), }; } // A source whose pages are declared per channel, counting every page read so a // test can assert exactly which shard pages a query opened. class CountingSource implements ShardSource { readonly label = "counting"; readonly reads: string[] = []; constructor( private readonly channelPages: Record, private readonly index?: Map ? V : never>, private readonly duplicates?: DuplicateIndex, ) {} async listChannels(): Promise { return Object.keys(this.channelPages).map((slug) => ({ key: slug, slug, name: slug, })); } async transcriptsManifest(ch: ChannelRef): Promise { const pages = this.channelPages[ch.slug] ?? []; const slugToPage: Record = {}; pages.forEach((page, i) => { for (const rec of page) slugToPage[rec.id] = i; }); return { version: 1, channelSlug: ch.slug, pageCount: pages.length, maxPageBytes: 0, generatedAt: "", slugToPage, }; } async transcriptPage(ch: ChannelRef, page: number): Promise { this.reads.push(`${ch.slug}:${page}`); return this.channelPages[ch.slug]?.[page] ?? []; } async loadAliases(): Promise { return []; } async loadGroups(): Promise { return { groups: [], defaultGroupId: "default" }; } publicOrigin(): string | null { return null; } async subsManifest(): Promise { return null; } async subsPage(): Promise { return []; } async postsManifest(): Promise { return null; } async postsPage(): Promise { return []; } async availabilityMap(): Promise> { return this.index ?? new Map(); } videoIndex(): Promise { // Deliberately still a method when the map is absent — an EMPTY index is a // different statement from no index at all, and the planner must treat the // empty case as "know nothing", not "nothing matches". return Promise.resolve((this.index ?? new Map()) as VideoIndex); } duplicateIndex(): Promise { return Promise.resolve(this.duplicates ?? new Map()); } } function indexed( rec: TranscriptDetail, state: "available" | "deleted" = "available", ): [string, NonNullable>] { return [ rec.slug, { state, id: rec.id, channelSlug: rec.channelSlug, title: rec.title, uploadDate: rec.uploadDate, isLivestream: rec.isLivestream === true, ageRestricted: rec.ageRestricted === true, ...(rec.curatedTags && rec.curatedTags.length > 0 ? { curatedTags: rec.curatedTags } : {}), }, ]; } const ALL_STATES: SearchFilters = { videos: true, livestreams: true, allAges: true, restricted: true, states: new Set(VIDEO_STATES), }; // ─── 1. the page cache ─── test("a batch of ids sharing one shard page parses that page exactly once", async () => { const dir = await mkdtemp(path.join(tmpdir(), "mcp-cache-")); try { const chDir = path.join(dir, "transcripts", "chan"); await mkdir(chDir, { recursive: true }); const page = [video("chan", "a1"), video("chan", "a2"), video("chan", "a3")]; const slugToPage: Record = { a1: 0, a2: 0, a3: 0 }; await writeFile( path.join(dir, "corpus.json"), JSON.stringify({ site: { id: "s", title: "S", url: "https://example.test/" }, channels: [{ slug: "chan", name: "Chan", videoCount: 3 }], }), ); await writeFile( path.join(chDir, "manifest.json"), JSON.stringify({ version: 1, channelSlug: "chan", pageCount: 1, maxPageBytes: 0, generatedAt: "", slugToPage, }), ); await writeFile(path.join(chDir, "page-0000.json"), JSON.stringify(page)); const source = new LocalSource(dir); const ch = (await source.listChannels())[0]; // Identity is the assertion: a second parse would produce a different // array. Same reference ⇒ the bytes were read and parsed once. const first = await source.transcriptPage(ch, 0); const second = await source.transcriptPage(ch, 0); assert.equal(first, second, "second read of the same page must be the cached one"); // Concurrent callers coalesce onto ONE in-flight read rather than racing. const [a, b, c] = await Promise.all([ source.transcriptPage(ch, 0), source.transcriptPage(ch, 0), source.transcriptPage(ch, 0), ]); assert.equal(a, b); assert.equal(b, c); assert.equal(a, first); // Manifests are cached outright — this is what kills the ~300 manifest // reads an un-hinted 20-id batch used to cost. assert.equal( await source.transcriptsManifest(ch), await source.transcriptsManifest(ch), ); // …and refresh is the escape hatch for a corpus rebuilt under a running // server: it must drop the page cache too, not just the channel list. await source.listChannels({ refresh: true }); assert.notEqual( await source.transcriptPage(ch, 0), first, "refresh must drop cached pages", ); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a local corpus cites its own composed site, not the platform", async () => { const dir = await mkdtemp(path.join(tmpdir(), "mcp-origin-")); try { await writeFile( path.join(dir, "corpus.json"), JSON.stringify({ site: { id: "s", title: "S", url: "https://hasanalyzer.pages.dev/" }, channels: [{ slug: "chan", name: "Chan" }], }), ); const source = new LocalSource(dir); // Unknown until the channel list is read — every render path does that // first, and guessing an origin before reading corpus.json would be // inventing one. assert.equal(source.publicOrigin(), null); await source.listChannels(); assert.equal(source.publicOrigin(), "https://hasanalyzer.pages.dev"); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a local corpus with no corpus.json keeps falling back to platform links", async () => { const dir = await mkdtemp(path.join(tmpdir(), "mcp-origin-none-")); try { await mkdir(path.join(dir, "transcripts"), { recursive: true }); const source = new LocalSource(dir); await source.listChannels(); assert.equal(source.publicOrigin(), null); } finally { await rm(dir, { recursive: true, force: true }); } }); // ─── 2. filter-first planning ─── test("a date-scoped query plans only the pages that can hold a match", async () => { const p0 = [video("chan", "old1", { uploadDate: "20220101" })]; const p1 = [video("chan", "new1", { uploadDate: "20240301" })]; const p2 = [video("chan", "old2", { uploadDate: "20210101" })]; const source = new CountingSource( { chan: [p0, p1, p2] }, new Map([indexed(p0[0]), indexed(p1[0]), indexed(p2[0])]), ); const channels = await source.listChannels(); const plan = await buildScanPlan(source, channels, { ...ALL_STATES, dateFrom: "20240101", dateTo: "20241231", }); assert.equal(plan.pruned, true); assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); assert.equal(plan.pagesPlanned, 1); assert.equal(plan.pagesTotal, 3); // …and the scan really only opens that page. const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], filters: { ...ALL_STATES, dateFrom: "20240101", dateTo: "20241231" }, }); assert.deepEqual(source.reads, ["chan:1"]); assert.equal(result.total, 1); assert.equal(result.hits[0].videoId, "new1"); assert.equal(result.coverage.pruned, true); assert.equal(result.coverage.pagesPlanned, 1); }); test("a video the index has never heard of still gets its page scanned", async () => { const p0 = [video("chan", "known", { uploadDate: "20200101" })]; const p1 = [video("chan", "ghost", { uploadDate: "20200101" })]; // The index knows only `known`, and would exclude it by date. `ghost` is // absent — which must mean "read the page and let the record decide", not // "skip it". This is the rule that makes a stale or partial summaries set // cost time instead of correctness. const source = new CountingSource( { chan: [p0, p1] }, new Map([indexed(p0[0])]), ); const channels = await source.listChannels(); const filters: SearchFilters = { ...ALL_STATES, dateFrom: "20240101" }; const plan = await buildScanPlan(source, channels, filters); assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); assert.equal(plan.unknownVideos, 1); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], filters, }); // The page was read, and the record predicate — not the index — dropped it. assert.deepEqual(source.reads, ["chan:1"]); assert.equal(result.total, 0); }); test("an unfiltered query never reads the index and never prunes", async () => { const p0 = [video("chan", "a1")]; const p1 = [video("chan", "a2")]; const source = new CountingSource({ chan: [p0, p1] }, new Map()); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], }); assert.deepEqual(source.reads, ["chan:0", "chan:1"]); assert.equal(result.coverage.pruned, false); assert.equal(result.total, 2); }); test("an all-permissive filter is treated as no filter, not as a reason to plan", async () => { const p0 = [video("chan", "a1")]; const source = new CountingSource({ chan: [p0] }, new Map()); const plan = await buildScanPlan(source, await source.listChannels(), ALL_STATES); assert.equal(plan.pruned, false, "a filter that excludes nothing must not trigger an index read"); assert.equal(plan.pagesPlanned, 1); }); test("a states filter reaches only the pages holding videos in those states", async () => { const p0 = [video("chan", "live1")]; const p1 = [video("chan", "gone1", { deleted: true })]; const p2 = [video("chan", "live2")]; const source = new CountingSource( { chan: [p0, p1, p2] }, new Map([ indexed(p0[0], "available"), indexed(p1[0], "deleted"), indexed(p2[0], "available"), ]), ); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], filters: { ...ALL_STATES, states: new Set(["deleted"]) }, }); assert.deepEqual(source.reads, ["chan:1"]); assert.equal(result.total, 1); assert.equal(result.hits[0].videoId, "gone1"); }); test("a curated-tag filter reaches only the pages holding tagged videos", async () => { const p0 = [video("chan", "plain")]; const p1 = [video("chan", "collab", { curatedTags: ["eva-collab"] })]; const p2 = [video("chan", "chatty", { curatedTags: ["eva-in-chat"] })]; const source = new CountingSource( { chan: [p0, p1, p2] }, new Map([indexed(p0[0]), indexed(p1[0]), indexed(p2[0])]), ); const channels = await source.listChannels(); const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] }; const plan = await buildScanPlan(source, channels, filters); assert.equal(plan.pruned, true); assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); assert.equal(plan.unknownVideos, 0); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], filters, }); assert.deepEqual(source.reads, ["chan:1"]); assert.equal(result.total, 1); assert.equal(result.hits[0].videoId, "collab"); }); test("a video the index has no tags for still gets its page read", async () => { // The pre-spec-4 case in miniature: the summaries set is older than the // transcripts and has never heard of `ghost`. "I don't know about this // video" must mean READ THE PAGE and let the record predicate decide — the // planner's one invariant — not "it carries no tags, skip it". `plain` IS // known and genuinely untagged, so it is correctly pruned. const p0 = [video("chan", "plain")]; const p1 = [video("chan", "ghost", { curatedTags: ["eva-collab"] })]; const source = new CountingSource( { chan: [p0, p1] }, new Map([indexed(p0[0])]), ); const channels = await source.listChannels(); const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] }; const plan = await buildScanPlan(source, channels, filters); assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); assert.equal(plan.unknownVideos, 1); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], filters, }); assert.deepEqual(source.reads, ["chan:1"]); assert.equal(result.total, 1, "the record the index missed still matched"); }); test("an empty curated-tag list is not a filter and does not plan", async () => { const p0 = [video("chan", "a1")]; const source = new CountingSource({ chan: [p0] }, new Map()); const plan = await buildScanPlan(source, await source.listChannels(), { ...ALL_STATES, curatedTags: [], }); assert.equal(plan.pruned, false); assert.equal(plan.pagesPlanned, 1); }); // ─── 3. duplicate collapsing ─── test("a recording mirrored across channels is counted once and its mirror named", async () => { const original = video("chanA", "orig"); const mirror = video("chanB", "copy"); const membership = { clusterId: "c1", canonicalSlug: "chanA/orig", contained: false, needsReview: false, }; const duplicates: DuplicateIndex = new Map([ [ "chanA/orig", { ...membership, isCanonical: true, siblings: [] }, ], [ "chanB/copy", { ...membership, isCanonical: false, siblings: [] }, ], ]); const source = new CountingSource( { chanA: [[original]], chanB: [[mirror]] }, new Map(), duplicates, ); const collapsed = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], }); assert.equal(collapsed.total, 1, "two uploads of one recording are one recording"); assert.equal(collapsed.duplicates.collapsed, 1); assert.equal(collapsed.duplicates.clusters, 1); // The kept row is the canonical member, and the copy is NAMED rather than // silently dropped. assert.equal(collapsed.hits[0].videoId, "orig"); assert.deepEqual( collapsed.hits[0].mirrors?.map((m) => m.videoId), ["copy"], ); // Opting out returns one row per upload, unchanged. const raw = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], collapseDuplicates: false, }); assert.equal(raw.total, 2); assert.equal(raw.duplicates.collapsed, 0); }); test("when only the mirror matches, the mirror is kept rather than dropped", async () => { // The canonical copy is not in the result set at all. Preferring an absent // canonical would delete the only surviving evidence — which is exactly the // material a 'what did the deleted videos say' question is after. const mirror = video("chanB", "copy"); const duplicates: DuplicateIndex = new Map([ [ "chanB/copy", { clusterId: "c1", canonicalSlug: "chanA/orig", isCanonical: false, contained: false, needsReview: false, siblings: [], }, ], ]); const source = new CountingSource({ chanB: [[mirror]] }, new Map(), duplicates); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], }); assert.equal(result.total, 1); assert.equal(result.hits[0].videoId, "copy"); assert.equal(result.duplicates.collapsed, 0); }); test("a corpus with no duplicates report reports that it could not check", async () => { const source = new CountingSource({ chan: [[video("chan", "a1")]] }, new Map()); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], }); // available:true here because the stub implements the member and returns an // empty map — "checked, none" — which is a different claim from a source // that ships no duplicates layer at all. assert.equal(result.duplicates.collapsed, 0); assert.equal(result.total, 1); }); // ─── 4. cap honesty ─── test("a capped scan names the channels it finished and the ones it never reached", async () => { const source = new CountingSource({ chanA: [[video("chanA", "a1")], [video("chanA", "a2")]], chanB: [[video("chanB", "b1")]], chanC: [[video("chanC", "c1")]], }); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], maxPages: 1, }); assert.equal(result.truncated, true); assert.equal(result.scanned.pages, 1); // The cap cut inside chanA, so chanB and chanC were never opened — and the // sample is therefore channel-biased, not random. Naming them is what makes // "partial" actionable instead of merely alarming. assert.deepEqual(result.coverage.channelsCompleted, []); assert.equal(result.coverage.channelStopped?.channel, "chanA"); assert.deepEqual(result.coverage.channelsNotReached, ["chanB", "chanC"]); }); test("an uncapped scan reports every channel as completed and none unreached", async () => { const source = new CountingSource({ chanA: [[video("chanA", "a1")]], chanB: [[video("chanB", "b1")]], }); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], }); assert.equal(result.truncated, false); assert.deepEqual(result.coverage.channelsCompleted, ["chanA", "chanB"]); assert.deepEqual(result.coverage.channelsNotReached, []); }); // ─── exclude ─── test("exclude drops a video that also contains the excluded term", async () => { const plain = video("chan", "keep", { text: "they filed a lawsuit today" }); const worldCup = video("chan", "drop", { text: "the lawsuit came up during the world cup", }); const source = new CountingSource({ chan: [[plain, worldCup]] }); const result = await searchTranscripts(source, { query: "lawsuit", contentTypes: ["video"], exclude: ["world cup"], }); assert.equal(result.total, 1); assert.equal(result.hits[0].videoId, "keep"); });