import { test } from "node:test"; import assert from "node:assert/strict"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import { accumulationSystemPrompt, applyReportPatch, buildAccumulationContent, buildApiMessages, buildGroundedContent, chunk, collectPriorPool, mergeReportSources, renderAliasGlossary, renderUsedAliasContext, reportNumbering, gatherSystemPrompt, answerSystemPrompt, serializeContext, parseContext, type UiMessage, } from "./askConversation"; import type { RetrievedVideo } from "./askRetrieval"; const ALIASES: SearchAlias[] = [ { id: "platner", label: "Graham Platner", triggers: ["platner", "plattner", "platter"], suggestion: "plat+ner", useRegex: true, note: "often mis-transcribed", }, ]; function video(over: Partial = {}): RetrievedVideo { return { key: "a", videoId: "a", title: "Platner interview", channel: "Rekieta", uploadDate: "20200101", snippets: [{ clock: "0:05", seconds: 5, text: "he said" }], ...over, }; } test("buildApiMessages replays prior turns as BARE text (no re-embedded excerpts)", () => { const prior: UiMessage[] = [ { role: "user", content: "tell me about Platner" }, { role: "assistant", content: "He is ...", sources: [video()] }, ]; const msgs = buildApiMessages(prior); assert.equal(msgs.length, 2); // The user turn replays as the bare question — excerpts are carried forward via // the cumulative pool, not re-embedded per turn (the "trapped in context" fix). assert.equal(msgs[0].content, "tell me about Platner"); assert.doesNotMatch(msgs[0].content, /excerpt|\[1\]/); assert.equal(msgs[1].content, "He is ..."); }); test("buildApiMessages drops error turns and empty pending assistants", () => { const prior: UiMessage[] = [ { role: "user", content: "q" }, { role: "assistant", content: "boom", error: true }, { role: "assistant", content: "" }, // pending, no content yet ]; assert.equal(buildApiMessages(prior).length, 1); }); test("collectPriorPool dedupes videos by key in first-seen order, merging snippets", () => { const prior: UiMessage[] = [ { role: "assistant", content: "a1", sources: [ video({ key: "a", snippets: [{ clock: "0:05", seconds: 5, text: "five" }] }), video({ key: "b", title: "Second", snippets: [{ clock: "0:10", seconds: 10, text: "ten" }] }), ], }, // A later turn expands video "a" with a new snippet + repeats "b". { role: "assistant", content: "a2", sources: [ video({ key: "a", snippets: [{ clock: "0:50", seconds: 50, text: "fifty" }] }), ], }, { role: "assistant", content: "boom", error: true, sources: [video({ key: "z" })] }, ]; const pool = collectPriorPool(prior); // First-seen order: a, then b. Error turn's "z" is skipped. assert.deepEqual(pool.map((v) => v.key), ["a", "b"]); // Video "a" merged both turns' snippets (deduped/sorted by seconds). assert.deepEqual( pool[0].snippets.map((sn) => sn.seconds), [5, 50], ); }); test("buildGroundedContent appends numbered excerpts, or just the question", () => { assert.equal(buildGroundedContent("hi", []), "hi"); const grounded = buildGroundedContent("who?", [video()]); assert.match(grounded, /who\?/); assert.match(grounded, /\[1\] "Platner interview" — Rekieta/); }); test("buildGroundedContent tiers a large set: index of all + excerpts for the top 12", () => { const few = [video({ key: "k0", title: "T0" })]; assert.doesNotMatch(buildGroundedContent("q", few), /matching videos/); const many = Array.from({ length: 15 }, (_, i) => video({ key: `k${i}`, title: `T${i}`, channel: "C", uploadDate: "20200102" }), ); const tiered = buildGroundedContent("q", many); // Index lists EVERY video (with a formatted date), numbered. assert.match(tiered, /All 15 matching videos/); assert.match(tiered, /\[15\] "T14" — C · 2020-01-02/); // Full excerpts only for the top 12, and the excerpt numbering matches the index. assert.match(tiered, /Full excerpts for the top 12/); assert.match(tiered, /\[1\] "T0" — C\n {4}\[0:05\] he said/); // A tail video (13th+) appears in the index but NOT as a full excerpt block. assert.doesNotMatch(tiered, /\[13\] "T12" — C\n {4}\[/); }); test("renderAliasGlossary directs searching the full term + carries the note", () => { const g = renderAliasGlossary(ALIASES); // Directive to search the full term, not a fragment. assert.match(g, /Graham Platner: when your search concerns this, search the full term "platner"/); assert.match(g, /NOT a fragment/); // Extra triggers surface as variant spellings. assert.match(g, /also written: plattner, platter/); // The alias note is included. assert.match(g, /often mis-transcribed/); assert.equal(renderAliasGlossary([]), ""); assert.equal( renderAliasGlossary([{ ...ALIASES[0], enabled: false }]), "", ); }); test("gatherSystemPrompt tailors the tail per mode and includes the glossary", () => { const scripted = gatherSystemPrompt(ALIASES, "scripted", 4); assert.match(scripted, /SEARCH: /); assert.match(scripted, /at most 4 searches/); assert.match(scripted, /Graham Platner/); const native = gatherSystemPrompt(ALIASES, "native", 3); assert.match(native, /search_transcripts tool/); assert.match(native, /finish tool/); }); test("serializeContext + parseContext round-trip through role markers", () => { const msgs = [ { role: "user" as const, content: "tell me about Platner\n\n[1] excerpt" }, { role: "assistant" as const, content: "He is ..." }, ]; const text = serializeContext(msgs); assert.match(text, /^USER:\n/); assert.match(text, /\nASSISTANT:\n/); assert.deepEqual(parseContext(text), msgs); }); test("parseContext survives pruning and falls back to a single user block", () => { // A human deletes the assistant turn — still parses to the remaining user turn. assert.deepEqual(parseContext("USER:\nonly the question"), [ { role: "user", content: "only the question" }, ]); // No markers at all → treat the whole blob as one user message. assert.deepEqual(parseContext("just some pasted notes"), [ { role: "user", content: "just some pasted notes" }, ]); assert.deepEqual(parseContext(" "), []); }); test("applyReportPatch appends a new section to the end, preserving others", () => { const before = "## Findings\n\nAlpha is discussed."; const after = applyReportPatch(before, "Open Questions", "Who said beta?"); // The existing section is preserved… assert.match(after, /## Findings\n\nAlpha is discussed\./); // …and the new one is appended with a `## ` heading and its content. assert.match(after, /## Open Questions\n\nWho said beta\?/); // New section comes AFTER the existing one. assert.ok(after.indexOf("## Findings") < after.indexOf("## Open Questions")); }); test("applyReportPatch replaces an existing section's body, leaving neighbours", () => { const before = "## Findings\n\nOld body.\n\n## Timeline\n\n- 2020: a thing"; const after = applyReportPatch(before, "Findings", "New body with more detail."); assert.match(after, /## Findings\n\nNew body with more detail\./); // The old body is gone… assert.doesNotMatch(after, /Old body/); // …and the untouched neighbouring section survives intact. assert.match(after, /## Timeline\n\n- 2020: a thing/); }); test("applyReportPatch preserves a preamble and appends into empty reports", () => { // First write into an empty report is just the section block. const first = applyReportPatch("", "Findings", "Alpha."); assert.equal(first, "## Findings\n\nAlpha."); // A preamble ahead of any heading is kept when a new section is appended. const withPreamble = applyReportPatch( "Research notes on the archive.\n\n## Findings\n\nAlpha.", "Sources", "[1] video", ); assert.match(withPreamble, /^Research notes on the archive\./); assert.match(withPreamble, /## Sources\n\n\[1\] video/); }); test("applyReportPatch matches the heading case-insensitively", () => { const before = "## Findings\n\nOld."; const after = applyReportPatch(before, "findings", "Updated."); // Matched despite the case difference → replaced, not appended (one heading). assert.equal((after.match(/## Findings/g) ?? []).length, 1); assert.match(after, /## Findings\n\nUpdated\./); assert.doesNotMatch(after, /Old\./); }); test("renderUsedAliasContext lists ONLY the used aliases with their variant spellings", () => { const c = renderUsedAliasContext(ALIASES); // Framed for reading excerpts, not for searching. assert.match(c, /may contain AI-transcription/); // The used alias, its variant spellings joined, and its note. assert.match(c, /Graham Platner: platner, plattner, platter — often mis-transcribed/); // Nothing used → empty string; disabled aliases are dropped; deduped by id. assert.equal(renderUsedAliasContext([]), ""); assert.equal(renderUsedAliasContext([{ ...ALIASES[0], enabled: false }]), ""); const dupCount = ( renderUsedAliasContext([ALIASES[0], ALIASES[0]]).match(/Graham Platner:/g) ?? [] ).length; assert.equal(dupCount, 1); }); test("chunk splits into fixed-size batches; last is the remainder; empty → []", () => { assert.deepEqual(chunk([1, 2, 3, 4, 5], 2), [[1, 2], [3, 4], [5]]); // Exact multiple → no trailing empty batch. assert.deepEqual(chunk([1, 2, 3, 4], 2), [[1, 2], [3, 4]]); // A single full batch when size ≥ length. assert.deepEqual(chunk([1, 2, 3], 10), [[1, 2, 3]]); // Empty input → no batches. assert.deepEqual(chunk([], 3), []); // Degenerate size ≤ 0 → one all-in batch (never an infinite loop). assert.deepEqual(chunk([1, 2], 0), [[1, 2]]); }); test("buildAccumulationContent emits the directive, a batch label, and numbered excerpts", () => { const videos = [ video({ key: "a", title: "First", snippets: [{ clock: "0:05", seconds: 5, text: "alpha here" }] }), video({ key: "b", title: "Second", snippets: [{ clock: "1:00", seconds: 60, text: "beta there" }] }), ]; const out = buildAccumulationContent("major contradictions", videos, 3, 12); // The directive leads. assert.match(out, /^major contradictions/); // The batch label names position within the sweep. assert.match(out, /Batch 3 of 12/); // Excerpts are numbered from [1] (restarting per batch) with their clocks. assert.match(out, /\[1\] "First" — Rekieta/); assert.match(out, /\[0:05\] alpha here/); assert.match(out, /\[2\] "Second" — Rekieta/); // The label steers the model to the precise-moment citation form. assert.match(out, /\[n @ mm:ss\]/); }); test("buildAccumulationContent numbers by a global registry index when given numberOf", () => { const videos = [ video({ key: "a", title: "First" }), video({ key: "b", title: "Second" }), ]; const registry = mergeReportSources( [video({ key: "x", title: "Earlier" }), video({ key: "y", title: "Also earlier" })], videos, ); const out = buildAccumulationContent( "d", videos, 2, 3, reportNumbering(registry), ); // The batch's videos are global entries 3 and 4 (after the two prior sources). assert.match(out, /\[3\] "First" — Rekieta/); assert.match(out, /\[4\] "Second" — Rekieta/); }); test("mergeReportSources appends new videos, dedupes by key, and trims snippet text", () => { const base = mergeReportSources([], [ video({ key: "a", title: "A", snippets: [{ clock: "0:05", seconds: 5, text: "keep me?" }] }), ]); // Snippet text is trimmed for storage, but clock/seconds survive. assert.equal(base[0].snippets[0].text, ""); assert.equal(base[0].snippets[0].seconds, 5); assert.equal(base[0].snippets[0].clock, "0:05"); const merged = mergeReportSources(base, [ // "a" reappears with a new moment → merged into the existing entry. video({ key: "a", title: "A", snippets: [{ clock: "1:00", seconds: 60, text: "x" }] }), // "b" is new → appended after "a". video({ key: "b", title: "B" }), ]); assert.deepEqual(merged.map((v) => v.key), ["a", "b"]); // "a" now carries both moments (5s + 60s), deduped/merged by time. assert.deepEqual( merged[0].snippets.map((s) => s.seconds).sort((x, y) => x - y), [5, 60], ); }); test("reportNumbering maps a video to its 1-based registry index by key", () => { const registry = mergeReportSources([], [ video({ key: "a" }), video({ key: "b" }), video({ key: "c" }), ]); const numberOf = reportNumbering(registry); assert.equal(numberOf(video({ key: "a" }), 0), 1); assert.equal(numberOf(video({ key: "c" }), 0), 3); // A key not in the registry falls back to the local index + 1. assert.equal(numberOf(video({ key: "zzz" }), 6), 7); }); test("accumulationSystemPrompt carries the update_report directive + focused alias block", () => { const p = accumulationSystemPrompt(ALIASES); // Reduce-into-report framing, native tools named, and searching forbidden. assert.match(p, /update_report tool/); assert.match(p, /Do NOT search/); assert.match(p, /fetch_context tool/); assert.match(p, /finish tool/); // Precise-moment, globally-stable citation form. assert.match(p, /\[n @ mm:ss\]/); assert.match(p, /GLOBAL and stable/); // The FOCUSED used-alias block (not the full "search the full term" glossary). assert.match(p, /TERMS USED IN THIS SEARCH/); assert.match(p, /Graham Platner: platner, plattner, platter/); assert.doesNotMatch(p, /search the full term/); // No aliases → just the base, no dangling block. assert.doesNotMatch(accumulationSystemPrompt([]), /TERMS USED IN THIS SEARCH/); }); test("applyReportPatch composes across sweep chunks: fold section A then B → both survive", () => { // Batch 1 folds in "Claims"; batch 2 folds in "Contradictions" — the reduce // threads the growing report, so both sections coexist afterwards. let report = ""; report = applyReportPatch(report, "Claims", "Alpha claims X. [1]"); report = applyReportPatch(report, "Contradictions", "Beta disputes X. [2]"); assert.match(report, /## Claims\n\nAlpha claims X\. \[1\]/); assert.match(report, /## Contradictions\n\nBeta disputes X\. \[2\]/); assert.ok(report.indexOf("## Claims") < report.indexOf("## Contradictions")); // A later batch merging MORE into an existing section replaces just that body. report = applyReportPatch(report, "Claims", "Alpha claims X and Y. [1][3]"); assert.match(report, /## Claims\n\nAlpha claims X and Y\. \[1\]\[3\]/); assert.doesNotMatch(report, /Alpha claims X\. \[1\]$/m); // The neighbour is untouched by the update. assert.match(report, /## Contradictions\n\nBeta disputes X\. \[2\]/); }); test("answerSystemPrompt asks for Markdown + citations and carries the FOCUSED used-alias block", () => { const p = answerSystemPrompt(ALIASES); assert.match(p, /Markdown/); // Precise-moment citation form: [n @ mm:ss] (backward-tolerant of a bare [n]). assert.match(p, /\[n @ mm:ss\]/); // The focused block (not the full glossary) — no "search the full term" directive. assert.match(p, /TERMS USED IN THIS SEARCH/); assert.match(p, /Graham Platner: platner, plattner, platter/); assert.doesNotMatch(p, /search the full term/); // No used aliases → just the base prompt, no dangling block. assert.doesNotMatch(answerSystemPrompt([]), /TERMS USED IN THIS SEARCH/); });