import { test } from "node:test"; import assert from "node:assert/strict"; import type { DisplaySummary } from "yt-dlp-transcript-common/lib/transcripts"; import type { LayerHit } from "yt-dlp-transcript-common/components/searchPipeline"; import type { TreeProgress } from "yt-dlp-transcript-common/lib/searchEval"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { extractKeywords, rankResults, buildContext, buildSearchRoot, } from "./askRetrieval"; import { isLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; function summary(over: Partial): DisplaySummary { return { slug: "s", id: "s", channelSlug: "chan", title: "Title", uploadDate: "20200101", date: "2020-01-01", duration: "1:00", channel: "Channel", isLivestream: false, ageRestricted: false, isDeleted: false, isUnlisted: false, platform: "youtube", webpageUrl: "https://example.com/s", ...over, }; } function hit(over: Partial): LayerHit { return { leafId: "l1", scope: "transcripts", start: 0, text: "x", ...over }; } function progress( slugs: string[], hits: Record, capped = false, ): TreeProgress { return { slugs: new Set(slugs), hits: new Map(Object.entries(hits)), leafStates: new Map(), groupStates: new Map(), done: true, capped, }; } test("extractKeywords drops stopwords, dedupes, keeps content terms", () => { const kw = extractKeywords("What did they say about the travel ban and travel?"); assert.deepEqual(kw, ["travel", "ban"]); }); test("extractKeywords ignores words shorter than 3 chars", () => { // "ai"/"ml" (2 chars) are dropped; "models" survives — no phrase fallback. assert.deepEqual(extractKeywords("Is AI or ML in the models?"), ["models"]); }); test("extractKeywords falls back to the phrase when nothing survives", () => { // all stopwords -> fall back to the trimmed lowercased phrase assert.deepEqual(extractKeywords("what about that"), ["what about that"]); }); test("extractKeywords caps the number of terms", () => { const q = "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo"; assert.equal(extractKeywords(q, 4).length, 4); }); test("rankResults orders by distinct-term coverage then hit density", () => { const summaries = [ summary({ slug: "a", id: "a", title: "A" }), summary({ slug: "b", id: "b", title: "B" }), summary({ slug: "c", id: "c", title: "C" }), ]; // a: 2 distinct leaves (highest coverage) // b: 1 leaf but 3 hits // c: 1 leaf, 1 hit const p = progress( ["a", "b", "c"], { a: [hit({ leafId: "l1", start: 10 }), hit({ leafId: "l2", start: 20 })], b: [ hit({ leafId: "l1", start: 5 }), hit({ leafId: "l1", start: 6 }), hit({ leafId: "l1", start: 7 }), ], c: [hit({ leafId: "l1", start: 1 })], }, ); const ranked = rankResults(p, summaries); assert.deepEqual( ranked.map((r) => r.videoId), ["a", "b", "c"], ); }); test("rankResults ties break toward the more recent upload", () => { const summaries = [ summary({ slug: "old", id: "old", uploadDate: "20190101" }), summary({ slug: "new", id: "new", uploadDate: "20210101" }), ]; const p = progress( ["old", "new"], { old: [hit({ leafId: "l1", start: 1 })], new: [hit({ leafId: "l1", start: 1 })], }, ); assert.deepEqual( rankResults(p, summaries).map((r) => r.videoId), ["new", "old"], ); }); test("rankResults skips slugs missing from summaries and honors the limit", () => { const summaries = [summary({ slug: "a", id: "a" })]; const p = progress(["a", "ghost"], { a: [hit({ leafId: "l1", start: 1 })], ghost: [hit({ leafId: "l1", start: 1 })], }); const ranked = rankResults(p, summaries, { limit: 5 }); assert.equal(ranked.length, 1); assert.equal(ranked[0].videoId, "a"); }); test("rankResults picks diverse snippets and formats timestamps", () => { const summaries = [summary({ slug: "a", id: "a" })]; const p = progress(["a"], { a: [ hit({ leafId: "l1", start: 5, text: " first hit " }), hit({ leafId: "l2", start: 125, text: "second hit" }), ], }); const [v] = rankResults(p, summaries, { snippetsPerVideo: 2 }); assert.deepEqual( v.snippets.map((s) => s.clock), ["0:05", "2:05"], ); assert.equal(v.snippets[0].text, "first hit"); }); const LOLI_ALIAS: SearchAlias = { id: "loli", label: "loli", triggers: ["loli", "lolly", "loly"], suggestion: "\\blol(i|ly)", useRegex: true, }; const PLATNER_ALIAS: SearchAlias = { id: "graham-platner", label: "Graham Platner", triggers: ["graham platner"], suggestion: "\\bgra\\w+ plat\\w+", useRegex: true, }; test("buildSearchRoot: a single-word alias fires and drops the plain leaf", () => { const { root } = buildSearchRoot("lolly banana", ["lolly", "banana"], [LOLI_ALIAS]); const leaves = root.children.filter(isLeaf); const aliasLeaf = leaves.find((l) => l.id === "t#a0")!; // "lolly" matches the alias trigger → a regex leaf uses the suggestion. assert.equal(aliasLeaf.query, "\\blol(i|ly)"); assert.equal(aliasLeaf.useRegex, true); // The covered "lolly" keyword is dropped; only "banana" remains as a plain leaf. assert.ok(!leaves.some((l) => l.query === "lolly")); const banana = leaves.find((l) => l.query === "banana")!; assert.equal(banana.useRegex, false); }); test("buildSearchRoot: a MULTI-WORD alias fires on the full phrase (the bug)", () => { // The whole-query phrase contains "graham platner" → the regex leaf appears. const { root } = buildSearchRoot( "graham platner", ["graham", "platner"], [PLATNER_ALIAS], ); const leaves = root.children.filter(isLeaf); const aliasLeaf = leaves.find((l) => l.useRegex); assert.ok(aliasLeaf, "expected a regex alias leaf"); assert.equal(aliasLeaf!.query, "\\bgra\\w+ plat\\w+"); // Both trigger tokens are covered → no bare "graham"/"platner" plain leaves. assert.ok(!leaves.some((l) => l.query === "graham" || l.query === "platner")); }); test("buildSearchRoot: a fragment does NOT fire a multi-word alias", () => { // Searching just "graham" can't satisfy the two-token trigger → no regex leaf. const { root, firedT, firedM } = buildSearchRoot("graham", ["graham"], [PLATNER_ALIAS]); const leaves = root.children.filter(isLeaf); assert.ok(!leaves.some((l) => l.useRegex)); assert.ok(leaves.some((l) => l.query === "graham")); // Nothing fired → no aliases surfaced for the answer prompt. assert.equal(firedT.length, 0); assert.equal(firedM.length, 0); }); test("buildSearchRoot: reports the aliases that fired (for focused answer context)", () => { const { firedT, firedM } = buildSearchRoot("lolly banana", ["lolly", "banana"], [LOLI_ALIAS]); // The matching alias fires in both scopes (it has no scope restriction). assert.deepEqual( firedT.map((a) => a.id), ["loli"], ); assert.deepEqual( firedM.map((a) => a.id), ["loli"], ); }); test("buildContext numbers videos and indents snippets", () => { const ctx = buildContext([ { key: "a", videoId: "a", title: "Travel Bans", channel: "Rekieta", uploadDate: "20200101", url: "https://example.com/a", snippets: [{ clock: "0:05", seconds: 5, text: "about travel bans" }], }, ]); assert.match(ctx, /\[1\] "Travel Bans" — Rekieta/); assert.match(ctx, /\[0:05\] about travel bans/); }); test("buildContext honours a numberOf override (global registry index)", () => { const videos = [ { key: "a", videoId: "a", title: "First", channel: "Rekieta", uploadDate: "20200101", snippets: [{ clock: "0:05", seconds: 5, text: "alpha" }], }, { key: "b", videoId: "b", title: "Second", channel: "Rekieta", uploadDate: "20200101", snippets: [{ clock: "1:00", seconds: 60, text: "beta" }], }, ]; // Number the batch as if it were entries 41/42 of a large sweep's registry. const index = new Map([ ["a", 41], ["b", 42], ]); const ctx = buildContext(videos, (v) => index.get(v.key)!); assert.match(ctx, /\[41\] "First" — Rekieta/); assert.match(ctx, /\[42\] "Second" — Rekieta/); // The default 1-based numbering is NOT used. assert.doesNotMatch(ctx, /\[1\] "First"/); });