import { test } from "node:test"; import assert from "node:assert/strict"; import { buildSweepInstructions, buildAskInstructions } from "./instructions"; import { parsePromptRequest } from "./promptRequest"; import { TOOLS } from "./server"; const CORPUS = "remote:https://site.example"; function sweep(request: string, ctxExtra = {}): string { return buildSweepInstructions(parsePromptRequest(request), { corpus: CORPUS, ...ctxExtra, }); } function ask(request: string, ctxExtra = {}): string { return buildAskInstructions(parsePromptRequest(request), { corpus: CORPUS, ...ctxExtra, }); } // ─── The cheap guard against a whole class of drift ─── test("every tool the instructions name actually exists", () => { const known = new Set(TOOLS.map((t) => t.name)); // Every `backticked_snake_case` token in any generated instruction that // looks like a tool name must be a real tool. Renaming a tool without // updating the instructions is otherwise invisible until a sweep fails. const texts = [ sweep("coffee"), sweep("coffee channels=chan-a batch_size=4"), sweep("https://site.example/?q=a follow this"), sweep("coffee", { unknownChannels: ["ghost"] }), sweep("coffee", { availableGroups: [{ id: "other", name: "Other", channels: 3 }], }), ask("what did they say"), ask("https://site.example/?q=a what about this"), ]; const suspects = new Set(); for (const text of texts) { for (const m of text.matchAll(/`([a-z][a-z0-9]*(?:_[a-z0-9]+)+)`/g)) { suspects.add(m[1]); } } // Argument names share the shape, so only judge tokens that name a tool. const NOT_TOOLS = new Set([ "batch_size", "content_types", "link_style", "max_pages", "moment_base", "dry_run", "parse_model", "report_path", "date_from", "date_to", "media_type", ]); const named = [...suspects].filter((s) => !NOT_TOOLS.has(s)); assert.ok(named.length > 0, "the instructions should name some tools"); for (const name of named) { assert.ok(known.has(name), `instructions name a nonexistent tool: ${name}`); } }); // ─── The corpus handle is threaded everywhere ─── test("the sweep instructions pin the corpus on every call, subagents included", () => { const text = sweep("coffee channels=chan-a"); assert.match(text, /\*\*Corpus: `remote:https:\/\/site\.example`\.\*\*/); assert.match(text, /Pass `source: "remote:https:\/\/site\.example"` on every/); assert.match(text, /Pass the corpus handle into every subagent/); // The literal scope args a call should carry. assert.match(text, /source: "remote:https:\/\/site\.example", channels: \["chan-a"\]/); }); test("the ask instructions pin the corpus too", () => { const text = ask("what did they say about coffee"); assert.match(text, /Corpus: `remote:https:\/\/site\.example`/); assert.match(text, /what did they say about coffee/); }); // ─── Coverage discipline ─── test("the sweep enumerates in one call and never instructs paging", () => { const text = sweep("coffee channels=chan-a"); assert.match(text, /enumerate_matches/); assert.match(text, /COVERAGE PARTIAL/); assert.ok( !/offset \+= limit/.test(text), "the paging instruction is gone — it was ignored and is the expensive path", ); assert.match(text, /ceil\(N \/ 8\)/); }); test("a validated batch size reaches the arithmetic, and a bad one cannot", () => { assert.match(sweep("coffee channels=c batch_size=12"), /ceil\(N \/ 12\)/); const bad = sweep("coffee channels=c batch_size=Why"); assert.match(bad, /ceil\(N \/ 8\)/); assert.ok(!bad.includes("ceil(N / Why)"), "the 'Why' bug must be impossible"); assert.match(bad, /^⚠ batch_size "Why"/m); }); test("warnings render as a leading block", () => { const text = sweep("coffee flavour=vanilla"); assert.ok(text.startsWith("⚠ "), text.slice(0, 40)); assert.match(text, /not a recognised setting/); }); // ─── Scope handling ─── test("an unresolvable channel stops the sweep instead of widening it", () => { const text = sweep("coffee channels=ghost", { unknownChannels: ["ghost"] }); assert.match(text, /Fix the scope first/); assert.match(text, /channel "ghost"/); // And it does not go on to instruct an enumeration over everything. assert.ok(!/Enumerate the complete worklist/.test(text)); }); test("no scope means pick-first, with the roster inlined when known", () => { const withRoster = sweep("coffee", { availableGroups: [ { id: "other", name: "Extended Universe", channels: 7 }, { id: "core", name: "Core", channels: 2 }, ], }); assert.match(withRoster, /Choose the scope first/); assert.match(withRoster, /other · Extended Universe · 7 channel\(s\)/); // The roster was pre-resolved, so no round-trip is asked for. assert.ok(!/Call `list_channels`/.test(withRoster)); const without = sweep("coffee"); assert.match(without, /Choose the scope first/); assert.match(without, /Call `list_channels`/); assert.match(without, /confirm \*\*all\*\*/); }); test("an explicit scope skips the pick-first step", () => { const text = sweep("coffee groups=other", { knownGroups: ["other"] }); assert.ok(!/Choose the scope first/.test(text)); assert.match(text, /scoped to groups "other"/); }); // ─── Links ─── test("a link seeds a dry-run decode, then a single real call", () => { const text = sweep("https://site.example/?qt=abc&fav=deleted what happened"); assert.match(text, /open_link/); assert.match(text, /dry_run: true/); assert.match(text, /https:\/\/site\.example\/\?qt=abc&fav=deleted/); assert.ok(!text.includes("apply:true")); }); // ─── Citations ─── test("both builders mandate the same citation forms", () => { for (const text of [sweep("coffee channels=c"), ask("coffee")]) { assert.match(text, /\[title @ mm:ss\]/); assert.match(text, /\[post by , \]/); assert.match(text, /never (take|takes|with) a? ?`@ mm:ss`/); } }); test("the ask plan pushes multi-query reads rather than re-reading per term", () => { const text = ask("what did they say about coffee and tea"); assert.match(text, /get_transcripts/); assert.match(text, /`queries`/); assert.match(text, /enumerate_matches/); }); test("resolved context notes are surfaced", () => { const text = sweep("coffee channels=c", { notes: ["42 channel(s) in remote:https://site.example"], }); assert.match(text, /Context already resolved for you/); assert.match(text, /42 channel\(s\)/); }); test("the sweep reads operator corrections from a sibling manifest before the first batch", () => { const text = sweep("coffee channels=chan-a"); assert.match(text, /video\.manifest\.json/); assert.match(text, /`correction` field/); // Before the batches, not after: a correction changes what the extractor // writes, so it has to be in hand when the first batch lands. const at = text.indexOf("Honour operator corrections"); const batch = text.indexOf("upsert"); assert.ok(at > 0 && batch > at, `corrections step at ${at}, batch step at ${batch}`); assert.match(text, /never re-asserted/); }); // ─── Media for a citation ─── test("both builders send clip media through fetch_clip, never a hand-run yt-dlp", () => { for (const text of [sweep("coffee channels=c"), ask("coffee")]) { assert.match(text, /`fetch_clip`/); assert.match(text, /NEVER run yt-dlp yourself/); // The corpus rides along, so a Rumble embed id maps to the editor's dir. assert.match(text, /pass source: "remote:https:\/\/site\.example" so a Rumble id/); // The whole recording is on offer, but the window is the default. assert.match(text, /full: true when the ask genuinely needs the whole recording/); // Only the operator falls back to yt-dlp. assert.match(text, /the operator's fallback, not yours/); } // In the sweep, before Finish: a report that needs a clip asks while the // citation is in hand, not after the summary is written. const s = sweep("coffee channels=c"); const clip = s.indexOf("Media for a cited moment"); const finish = s.indexOf("**Finish.**"); assert.ok(clip > 0 && finish > clip, `clip step at ${clip}, Finish at ${finish}`); // In the ask, after the answer's citation rule. const a = ask("coffee"); const cite = a.indexOf("Answer with citations"); const askClip = a.indexOf("Media for a cited moment"); assert.ok(cite > 0 && askClip > cite, `citations at ${cite}, clip step at ${askClip}`); });