// ─── Parsing a pasted request into a sweep/ask plan ─── // // Claude Code tokenises an MCP prompt's arguments as // `text.trim().split(/\s+/)` zipped against the declared argument names. It is // not quote-aware, the last argument does not absorb the remainder, and tokens // past the declared count are dropped in silence. With nine declared arguments // a real request became link="This" channel="search" group="deleted" // directive="videos." batch_size="Why" — and everything from the actual // question onward vanished. // // So the entry point for Claude Code is a TOOL (structured JSON arguments, // which arrive intact) taking one free-text `request`, and this module is the // parser behind it. It is pure and exhaustively tested; the MCP prompt form // routes through the same validators so a form-based client (Claude Desktop, // Cursor) gets a named error instead of mangled arithmetic. // // The rules that matter: // - a key=value pair is honoured ONLY for a closed whitelist, so an unknown // `x=y` stays in the prose instead of being silently eaten; // - a near-miss key gets a typo hint and still stays in the prose; // - every ignored, corrected, or defaulted input lands in `warnings`, which // the caller renders as a ⚠ block at the top of its output. export type PromptRequest = { // The first http(s) URL in the input, cleaned of wrappers and trailing // punctuation. A share link seeds the search; anything else is just prose. link?: string; // Everything that was not a recognised key=value pair or the link: the // question, in the user's own words, punctuation intact. query: string; channels: string[]; groups: string[]; batchSize: number; parseModel: string; reportPath?: string; directive?: string; source?: string; contentTypes?: ("video" | "post")[]; regex: boolean; warnings: string[]; }; export const DEFAULT_BATCH_SIZE = 8; export const DEFAULT_PARSE_MODEL = "haiku"; export const DEFAULT_DIRECTIVE = "key claims & contradictions"; export const DEFAULT_REPORT_PATH = "./sweep-report.md"; // The closed whitelist: canonical key → accepted aliases. Anything outside it // is prose, never a parameter. const KEYS: Record = { channels: ["channel", "chan", "chans"], groups: ["group"], batch_size: ["batch", "batchsize", "batch_sz"], parse_model: ["model", "parsemodel", "extractor_model"], report: ["report_path", "reportpath", "path", "out", "output"], directive: ["extract", "extracting"], source: ["corpus"], content_types: ["content_type", "types", "type"], regex: ["re"], }; const CANONICAL = new Map(); for (const [canon, aliases] of Object.entries(KEYS)) { CANONICAL.set(canon, canon); for (const a of aliases) CANONICAL.set(a, canon); } // Edit distance, bounded: we only care whether it is ≤ 1. function withinOneEdit(a: string, b: string): boolean { if (a === b) return true; const [short, long] = a.length <= b.length ? [a, b] : [b, a]; if (long.length - short.length > 1) return false; let i = 0; let j = 0; let edits = 0; while (i < short.length && j < long.length) { if (short[i] === long[j]) { i++; j++; continue; } if (++edits > 1) return false; if (short.length === long.length) i++; j++; } return edits + (long.length - j) + (short.length - i) <= 1; } function typoHint(key: string): string | undefined { for (const known of CANONICAL.keys()) { if (withinOneEdit(key.toLowerCase(), known)) return known; } return undefined; } function countChar(s: string, c: string): number { let n = 0; for (const x of s) if (x === c) n++; return n; } // Pull the first http(s) URL out of the text, undoing the ways a URL gets // wrapped when a human pastes it: , "url", [label](url), and a trailing // sentence mark. A closing bracket is only stripped when it does not balance // one inside the URL, so a URL that legitimately contains brackets survives. function extractLink(text: string): { link?: string; rest: string } { // A URL that is the VALUE of a recognised setting (source=remote:https://…) // is not the request's link — skip to the next candidate. const re = /https?:\/\/\S+/gi; let m: RegExpExecArray | null = null; for (let hit = re.exec(text); hit; hit = re.exec(text)) { const lineStart = text.lastIndexOf(" ", hit.index - 1) + 1; const prefix = text.slice(lineStart, hit.index).replace(/^["']|["']$/g, ""); const kv = /^([A-Za-z][A-Za-z0-9_]*)=\S*$/.exec(prefix); if (kv && CANONICAL.has(kv[1].toLowerCase())) continue; m = hit; break; } if (!m) return { rest: text }; const start = m.index; let cand = m[0]; const before = start > 0 ? text[start - 1] : ""; for (let guard = 0; guard < 50; guard++) { const last = cand[cand.length - 1]; if (!last) break; if (".,;:!?".includes(last)) { cand = cand.slice(0, -1); continue; } if (last === ">" && before === "<") { cand = cand.slice(0, -1); continue; } if ((last === '"' || last === "'") && (before === last || countChar(cand, last) % 2 === 1)) { cand = cand.slice(0, -1); continue; } const opener = last === ")" ? "(" : last === "]" ? "[" : last === "}" ? "{" : ""; if (opener && countChar(cand, opener) < countChar(cand, last)) { cand = cand.slice(0, -1); continue; } break; } // Remove the whole markdown wrapper when the URL sat inside one, so the // label doesn't survive as a stray `[Search results](` in the prose. let from = start; let to = start + m[0].length; if (before === "(") { const open = text.lastIndexOf("[", start); const close = text.indexOf(")", to - 1); if (open !== -1 && close !== -1 && text[start - 1] === "(" && text[open] === "[") { from = open; to = close + 1; } } else if (before === "<") { from = start - 1; } const rest = `${text.slice(0, from)} ${text.slice(to)}`; return { link: cand, rest }; } // Whitespace-split, but keep a quoted run together and drop the quotes, so // directive="key claims & contradictions" survives as ONE token. function tokenize(text: string): string[] { const out: string[] = []; let cur = ""; let quote: string | null = null; let sawQuote = false; const push = (): void => { if (cur !== "" || sawQuote) out.push(cur); cur = ""; sawQuote = false; }; for (const ch of text) { if (quote) { if (ch === quote) { quote = null; continue; } cur += ch; continue; } if (ch === '"' || ch === "'") { quote = ch; sawQuote = true; continue; } if (/\s/.test(ch)) { push(); continue; } cur += ch; } push(); return out.filter((t) => t !== ""); } function splitList(v: string): string[] { return v .split(",") .map((x) => x.trim()) .filter((x) => x !== ""); } // ─── Validators. Each one either accepts, or falls back and says why. ─── function validBatchSize(raw: string, warnings: string[]): number { const n = Number(raw); if (!Number.isInteger(n) || n < 1 || n > 20) { warnings.push( `batch_size "${raw}" is not an integer between 1 and 20 — using ` + `${DEFAULT_BATCH_SIZE}.`, ); return DEFAULT_BATCH_SIZE; } return n; } function validParseModel(raw: string, warnings: string[]): string { if (raw === "" || /\s/.test(raw)) { warnings.push( `parse_model "${raw}" is not a single token — using ` + `"${DEFAULT_PARSE_MODEL}".`, ); return DEFAULT_PARSE_MODEL; } return raw; } function validReportPath(raw: string, warnings: string[]): string | undefined { if (raw === "" || /\s/.test(raw)) { warnings.push(`report path "${raw}" is not a single token — ignored.`); return undefined; } if (raw.includes("..")) { warnings.push(`report path "${raw}" escapes the working directory — ignored.`); return undefined; } if (!raw.toLowerCase().endsWith(".md")) { warnings.push(`report path "${raw}" is not a .md file — ignored.`); return undefined; } return raw; } function validContentTypes( raw: string, warnings: string[], ): ("video" | "post")[] | undefined { const wanted = splitList(raw.toLowerCase()); const out = wanted.filter((t): t is "video" | "post" => t === "video" || t === "post"); const bad = wanted.filter((t) => t !== "video" && t !== "post"); if (bad.length > 0) { warnings.push( `content_types value(s) ignored (expected 'video' and/or 'post'): ${bad.join(", ")}.`, ); } return out.length > 0 ? out : undefined; } // Parse one free-text request. Never throws: a bad value becomes a default // plus a warning, because failing a whole sweep over a typo'd batch size is // worse than running it at 8 and saying so. export function parsePromptRequest(input: string): PromptRequest { const warnings: string[] = []; const text = (input ?? "").trim(); const { link, rest } = extractLink(text); const tokens = tokenize(rest); const seen = new Map(); const prose: string[] = []; for (const tok of tokens) { const eq = tok.indexOf("="); if (eq <= 0) { prose.push(tok); continue; } const rawKey = tok.slice(0, eq); const value = tok.slice(eq + 1); // Only a plausible key shape is even considered — this keeps prose like // "x=y+2" or "a==b" out of the parameter path entirely. if (!/^[A-Za-z][A-Za-z0-9_]*$/.test(rawKey)) { prose.push(tok); continue; } const canon = CANONICAL.get(rawKey.toLowerCase()); if (!canon) { const hint = typoHint(rawKey); warnings.push( `"${rawKey}=" is not a recognised setting, so it was left in the ` + `question text` + (hint ? ` — did you mean ${hint}=?` : "") + `.`, ); prose.push(tok); continue; } if (seen.has(canon)) { warnings.push( `${canon} was given more than once — using the last value ("${value}").`, ); } seen.set(canon, value); } const req: PromptRequest = { ...(link ? { link } : {}), query: prose.join(" ").replace(/\s+/g, " ").trim(), channels: seen.has("channels") ? splitList(seen.get("channels")!) : [], groups: seen.has("groups") ? splitList(seen.get("groups")!) : [], batchSize: seen.has("batch_size") ? validBatchSize(seen.get("batch_size")!, warnings) : DEFAULT_BATCH_SIZE, parseModel: seen.has("parse_model") ? validParseModel(seen.get("parse_model")!, warnings) : DEFAULT_PARSE_MODEL, regex: seen.get("regex") === "true" || seen.get("regex") === "1", warnings, }; if (seen.has("report")) { const p = validReportPath(seen.get("report")!, warnings); if (p) req.reportPath = p; } if (seen.has("directive")) { const d = seen.get("directive")!.trim(); if (d !== "") req.directive = d; } if (seen.has("source")) { const s = seen.get("source")!.trim(); if (s !== "") req.source = s; } if (seen.has("content_types")) { const t = validContentTypes(seen.get("content_types")!, warnings); if (t) req.contentTypes = t; } if (!req.link && req.query === "") { warnings.push( "no query and no link — say what to sweep for, or paste a share link.", ); } return req; } // The `sweep` prompt's declared argument names, IN ORDER. Claude Code zips the // user's whitespace-split words against exactly this list, so the order is also // the recipe for reassembling what they typed when it has shredded a request. export const SWEEP_PROMPT_ARGS = [ "query", "link", "channel", "channels", "group", "directive", "batch_size", "parse_model", "report_path", ] as const; export type ArgumentProblem = { arg: string; value: string; why: string }; // Hard validation for the PROMPT form. The tool path is forgiving (a bad value // becomes a default plus a ⚠, because failing a whole sweep over a typo is // worse than running it at 8 and saying so). The prompt path must not be: a // typed argument holding prose is the fingerprint of Claude Code's tokenizer // having word-split the request, and continuing would run a sweep for // something the user never asked for. export function validateSweepArguments(args: Record): { problems: ArgumentProblem[]; reassembled: string; } { const str = (k: string): string | undefined => { const v = args[k]; return typeof v === "string" && v.trim() !== "" ? v.trim() : undefined; }; const problems: ArgumentProblem[] = []; const bad = (arg: string, value: string, why: string): void => { problems.push({ arg, value, why }); }; const link = str("link"); if (link && !/^ 20) { bad("batch_size", batch, "is not an integer between 1 and 20"); } } const report = str("report_path") ?? str("report"); if (report && (!report.toLowerCase().endsWith(".md") || report.includes(".."))) { bad("report_path", report, "is not a safe .md path"); } const model = str("parse_model"); if (model && !/^[A-Za-z0-9._-]+$/.test(model)) { bad("parse_model", model, "is not a model name"); } // Whatever words did land, back in the order they were typed. const reassembled = SWEEP_PROMPT_ARGS.map((a) => str(a) ?? "") .filter((v) => v !== "") .join(" "); return { problems, reassembled }; } // The MCP prompt's declared arguments, routed through the SAME validators, so // a form-based client gets an error naming the offending argument rather than // a rendered `ceil(N / Why)`. export function requestFromArguments( args: Record, ): PromptRequest { const warnings: string[] = []; const str = (k: string): string | undefined => { const v = args[k]; return typeof v === "string" && v.trim() !== "" ? v.trim() : undefined; }; const rawLink = str("link"); const link = rawLink ? extractLink(rawLink).link : undefined; if (rawLink && !link) { warnings.push(`link "${rawLink}" is not an http(s) URL — ignored.`); } const channels = [ ...(str("channel") ? [str("channel")!] : []), ...(str("channels") ? splitList(str("channels")!) : []), ]; const groups = [ ...(str("group") ? [str("group")!] : []), ...(str("groups") ? splitList(str("groups")!) : []), ]; const batchRaw = str("batch_size"); const modelRaw = str("parse_model"); const reportRaw = str("report_path") ?? str("report"); const typesRaw = str("content_types"); const req: PromptRequest = { ...(link ? { link } : {}), query: str("query") ?? "", channels, groups, batchSize: batchRaw ? validBatchSize(batchRaw, warnings) : DEFAULT_BATCH_SIZE, parseModel: modelRaw ? validParseModel(modelRaw, warnings) : DEFAULT_PARSE_MODEL, regex: args.regex === true || str("regex") === "true", warnings, }; if (reportRaw) { const p = validReportPath(reportRaw, warnings); if (p) req.reportPath = p; } if (str("directive")) req.directive = str("directive"); if (str("source")) req.source = str("source"); if (typesRaw) { const t = validContentTypes(typesRaw, warnings); if (t) req.contentTypes = t; } return req; } // The ⚠ block a caller puts at the top of its output. Empty when clean. export function renderWarnings(warnings: string[]): string { if (warnings.length === 0) return ""; return `⚠ ${warnings.join("\n⚠ ")}\n\n`; }