import { test } from "node:test"; import assert from "node:assert/strict"; import type { TranscriptDetail } from "../transcripts"; import type { SubsDetail } from "../subs"; import type { Post } from "../posts"; import type { ChannelTranscriptsManifest } from "../manifest"; import type { ArchiveReader, ChannelRef } from "../archive/reader"; import { newLeaf } from "../searchQuery"; import { runLeafPipeline, type LeafFetchers, type LeafProgress } from "./leafPipeline"; // ─── an in-memory archive, and fetchers derived from it ─── // // The leaf pipeline takes its fetches as parameters, which is the whole point // of the move: `components/searchPipeline.ts` binds them to the react-query // caches, and here they are bound to an ArchiveReader instead. Nothing in the // module under test knows the difference, and the page reads are countable — // which is how "one page read serves every slug on it" is stated as a fact // rather than hoped for. const CH: ChannelRef = { key: "c", slug: "c", name: "C" }; type Archive = { pages: TranscriptDetail[][]; subs: Record; posts: Record; reads: string[]; }; function record(id: string, cues: { start: number; text: string }[]): TranscriptDetail { return { id, slug: `c/${id}`, title: `video ${id}`, channel: "C", uploadDate: "20250101", cues: cues.map((c) => ({ ...c, end: c.start + 2 })), description: `about ${id}`, tags: [id, "shared-tag"], } as TranscriptDetail; } function makeReader(a: Archive): ArchiveReader { const slugToPage: Record = {}; a.pages.forEach((page, i) => { for (const r of page) slugToPage[r.id] = i; }); const manifest: ChannelTranscriptsManifest = { version: 1, pageCount: a.pages.length, slugToPage, } as ChannelTranscriptsManifest; return { label: "in-memory", listChannels: async () => [CH], transcriptsManifest: async () => { a.reads.push("manifest"); return manifest; }, transcriptPage: async (_ch, page) => { a.reads.push(`page-${page}`); return a.pages[page] ?? []; }, loadAliases: async () => [], loadGroups: async () => ({ groups: [], defaultGroupId: "default" }), publicOrigin: () => null, subsManifest: async () => null, subsPage: async () => [], postsManifest: async () => null, postsPage: async () => [], availabilityMap: async () => new Map(), }; } // The adapter a caller writes once: slug -> record, through the reader's // manifest/page walk. Page-level, never per-record — a per-record fetch is a // bench regression by construction. function fetchersFor(a: Archive): LeafFetchers { const reader = makeReader(a); const pageCache = new Map>(); return { transcript: async (slug) => { const id = slug.split("/")[1] ?? slug; const manifest = await reader.transcriptsManifest(CH); const page = manifest.slugToPage[id]; if (page === undefined) throw new Error(`no such slug: ${slug}`); let p = pageCache.get(page); if (!p) { p = reader.transcriptPage(CH, page); pageCache.set(page, p); } const found = (await p).find((r) => r.id === id); if (!found) throw new Error(`no such record: ${slug}`); return found; }, subs: async (slug) => { const d = a.subs[slug]; if (!d) throw new Error(`no subs: ${slug}`); return d; }, post: async (slug) => { const p = a.posts[slug]; if (!p) throw new Error(`no post: ${slug}`); return p; }, }; } function archive(over: Partial = {}): Archive { return { pages: [], subs: {}, posts: {}, reads: [], ...over }; } async function run( leaf: ReturnType, slugs: string[], fetchers: LeafFetchers, opts: { hitLimit?: number } = {}, ): Promise<{ final: LeafProgress[]; slugs: string[]; hits: Map }> { const seen: LeafProgress[] = []; const controller = runLeafPipeline({ leaf, scopeSlugs: slugs, initialHitLimit: opts.hitLimit ?? 1000, concurrency: 2, flushIntervalMs: 0, emit: (p) => seen.push(p), fetchers, }); const result = await controller.done; return { final: seen, slugs: [...result.slugs].sort(), hits: result.hits as Map, }; } test("a transcripts leaf matches cues over an injected reader, one read per page", async () => { const a = archive({ pages: [ [record("v1", [{ start: 0, text: "a needle here" }]), record("v2", [{ start: 5, text: "nothing" }])], [record("v3", [{ start: 9, text: "another NEEDLE" }])], ], }); const r = await run(newLeaf({ query: "needle" }), ["c/v1", "c/v2", "c/v3"], fetchersFor(a)); assert.deepEqual(r.slugs, ["c/v1", "c/v3"]); assert.equal(a.reads.filter((x) => x.startsWith("page-")).length, 2); const hits = r.hits.get("c/v1") as { start: number; text: string; scope: string }[]; assert.equal(hits.length, 1); assert.equal(hits[0].start, 0); assert.equal(hits[0].scope, "transcripts"); }); test("the leaf's id is stamped on every hit so the tree can bucket them", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const r = await run(newLeaf({ id: "leaf-7", query: "needle" }), ["c/v1"], fetchersFor(a)); const hits = r.hits.get("c/v1") as { leafId: string }[]; assert.equal(hits[0].leafId, "leaf-7"); }); test("contributeHits:false still narrows the slug set but carries no hits", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const r = await run( newLeaf({ query: "needle", contributeHits: false }), ["c/v1"], fetchersFor(a), ); assert.deepEqual(r.slugs, ["c/v1"]); assert.equal(r.hits.size, 0); }); test("description and tags scopes read the same fetched record", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "silence" }])]] }); const f = fetchersFor(a); const desc = await run(newLeaf({ query: "about v1", scope: "description" }), ["c/v1"], f); assert.deepEqual(desc.slugs, ["c/v1"]); const tags = await run(newLeaf({ query: "shared-tag", scope: "tags" }), ["c/v1"], f); assert.deepEqual(tags.slugs, ["c/v1"]); }); test("a chat leaf reads ONLY the live_chat track", async () => { const a = archive({ subs: { "c/v1": { id: "v1", slug: "c/v1", tracks: { live_chat: [{ start: 4, end: 5, text: "needle in chat" }], en: [{ start: 9, end: 10, text: "needle in captions" }], }, } as unknown as SubsDetail, }, }); const r = await run(newLeaf({ query: "needle", scope: "chat" }), ["c/v1"], fetchersFor(a)); const hits = r.hits.get("c/v1") as { track: string; start: number; scope: string }[]; assert.equal(hits.length, 1, "the en track is not a chat hit"); assert.equal(hits[0].track, "live_chat"); assert.equal(hits[0].scope, "chat"); assert.equal(hits[0].start, 4); }); test("a posts leaf matches the body and stamps start 0", async () => { const a = archive({ posts: { "c/p1": { id: "p1", slug: "c/p1", text: "a needle in a post" } as unknown as Post, "c/p2": { id: "p2", slug: "c/p2", text: "nothing here" } as unknown as Post, }, }); const r = await run(newLeaf({ query: "needle", scope: "posts" }), ["c/p1", "c/p2"], fetchersFor(a)); assert.deepEqual(r.slugs, ["c/p1"]); const hits = r.hits.get("c/p1") as { start: number }[]; assert.equal(hits[0].start, 0); }); test("a failing fetch is skipped, not fatal, and progress still completes", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const r = await run(newLeaf({ query: "needle" }), ["c/v1", "c/missing"], fetchersFor(a)); assert.deepEqual(r.slugs, ["c/v1"]); const last = r.final[r.final.length - 1]; assert.equal(last.done, true); assert.equal(last.processed, 2, "the failed slug still counts as processed"); }); test("the hit cap stops the scan and reports capped; setHitLimit resumes it", async () => { const a = archive({ pages: [ Array.from({ length: 8 }, (_, i) => record(`v${i}`, [{ start: i, text: "needle" }]), ), ], }); const slugs = Array.from({ length: 8 }, (_, i) => `c/v${i}`); const fetchers = fetchersFor(a); const seen: LeafProgress[] = []; const controller = runLeafPipeline({ leaf: newLeaf({ query: "needle" }), scopeSlugs: slugs, initialHitLimit: 2, concurrency: 1, flushIntervalMs: 0, emit: (p) => seen.push(p), fetchers, }); const first = await controller.done; assert.ok(first.slugs.size <= 3, "stopped near the cap, not at the end"); assert.ok(seen.some((p) => p.capped), "…and said so"); }); test("an empty query settles immediately with nothing, rather than scanning", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const r = await run(newLeaf({ query: " " }), ["c/v1"], fetchersFor(a)); assert.deepEqual(r.slugs, []); assert.deepEqual(a.reads, []); }); test("a metadata leaf is not this module's job and resolves empty", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const r = await run(newLeaf({ query: "video", scope: "metadata" }), ["c/v1"], fetchersFor(a)); assert.deepEqual(r.slugs, []); assert.deepEqual(a.reads, []); }); test("cancel settles `done` with what had been seen so far", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }])]] }); const controller = runLeafPipeline({ leaf: newLeaf({ query: "needle" }), scopeSlugs: ["c/v1"], initialHitLimit: 100, concurrency: 1, flushIntervalMs: 0, emit: () => {}, fetchers: fetchersFor(a), }); controller.cancel(); const r = await controller.done; assert.ok(r.slugs instanceof Set); }); test("a regex leaf compiles once; a malformed pattern matches nothing", async () => { const a = archive({ pages: [[record("v1", [{ start: 0, text: "needle" }]), record("v2", [{ start: 0, text: "noodle" }])]], }); const f = fetchersFor(a); const ok = await run(newLeaf({ query: "n[eo]+dle", useRegex: true }), ["c/v1", "c/v2"], f); assert.deepEqual(ok.slugs, ["c/v1", "c/v2"]); const bad = await run(newLeaf({ query: "n([edle", useRegex: true }), ["c/v1"], f); assert.deepEqual(bad.slugs, []); }); test("a transcripts leaf reads a record's alternate tracks too, and names the track of a hit only one holds", async () => { const rec = { ...record("v1", [{ start: 3, text: "the harbor bridge" }]), track: "en-orig", altTracks: [ { track: "en", cues: [ { start: 4, end: 6, text: "the harbor bridge" }, { start: 100, end: 102, text: "a zeppelin overhead" }, ], }, ], } as TranscriptDetail; const a = archive({ pages: [[rec, record("v2", [{ start: 0, text: "nothing" }])]] }); const only = await run(newLeaf({ query: "zeppelin" }), ["c/v1", "c/v2"], fetchersFor(a)); assert.deepEqual(only.slugs, ["c/v1"]); const hits = only.hits.get("c/v1") as { start: number; track?: string; scope: string }[]; assert.deepEqual(hits.map((h) => [h.start, h.track, h.scope]), [[100, "en", "transcripts"]]); // A word both tracks say near the same moment: one hit, the primary's. const both = await run(newLeaf({ query: "harbor" }), ["c/v1"], fetchersFor(a)); const bh = both.hits.get("c/v1") as { track?: string }[]; assert.deepEqual(bh.map((h) => h.track), [undefined]); });