// Bring-your-own-key chat providers, called DIRECTLY from the browser. The // export site is fully static — there is no backend and we host no inference. // The visitor supplies their own API key; requests go browser → provider only. // // Each provider is CORS-enabled for browser use: // - Anthropic: requires the `anthropic-dangerous-direct-browser-access` header // - OpenAI: chat-completions is CORS-open with a Bearer key // - Gemini: Generative Language API takes the key as a query param // // Everything streams so the answer renders as it arrives. import { acquire, noteRateLimited, parseRetryAfterMs, withRateLimitRetry, } from "./rateLimit"; export type Provider = "anthropic" | "openai" | "gemini"; export type ChatRole = "user" | "assistant"; export type ChatMessage = { role: ChatRole; content: string }; export type ProviderInfo = { id: Provider; label: string; defaultModel: string; // A few known-good model ids to offer; the field stays free-text so a user can // enter anything their key supports. models: string[]; keyHint: string; keyUrl: string; }; export const PROVIDERS: Record = { anthropic: { id: "anthropic", label: "Anthropic (Claude)", defaultModel: "claude-haiku-4-5", models: ["claude-haiku-4-5", "claude-sonnet-5", "claude-opus-4-8"], keyHint: "sk-ant-…", keyUrl: "https://console.anthropic.com/settings/keys", }, openai: { id: "openai", label: "OpenAI", defaultModel: "gpt-4o-mini", models: ["gpt-4o-mini", "gpt-4o"], keyHint: "sk-…", keyUrl: "https://platform.openai.com/api-keys", }, gemini: { id: "gemini", label: "Google Gemini", defaultModel: "gemini-2.5-flash", models: ["gemini-2.5-flash", "gemini-2.5-pro"], keyHint: "AIza…", keyUrl: "https://aistudio.google.com/app/apikey", }, }; // Normalized provider finish reason. `length` = truncated at the output cap // (Anthropic max_tokens / OpenAI length / Gemini MAX_TOKENS); `safety` = blocked // (content_filter / SAFETY / RECITATION); `stop` = a normal completion. export type FinishReason = "length" | "stop" | "safety" | "other"; // One captured provider API call, for the debug export. `request`/`response` are // the raw-ish payloads (system + messages / accumulated text for answers; the // full body / parsed JSON for gather rounds). An API key is NEVER included. export type DebugCall = { phase: "gather" | "answer"; provider: string; model?: string; request: unknown; response: unknown; finishReason?: FinishReason; rawFinish?: string; blockReason?: string; usage?: unknown; ms: number; }; // Map a provider's raw finish/stop string to our normalized FinishReason. export function normFinish(raw: string): FinishReason { const r = raw.toLowerCase(); if (r === "max_tokens" || r === "length") return "length"; if ( r === "safety" || r === "recitation" || r === "content_filter" || r === "prohibited_content" || r === "blocklist" || r === "spii" ) { return "safety"; } if (r === "end_turn" || r === "stop" || r === "stop_sequence" || r === "tool_use") { return "stop"; } return "other"; } export type AskOptions = { provider: Provider; apiKey: string; model?: string; system: string; messages: ChatMessage[]; maxTokens?: number; signal?: AbortSignal; onDelta?: (chunk: string) => void; // Optional capture of the completed call (finish reason, usage, payloads) for // the debug export. Never receives the API key. onDebug?: (rec: DebugCall) => void; }; // Stream a completion from the chosen provider, invoking onDelta for each text // chunk and resolving with the full concatenated answer. Paced by the shared // rate limiter: `acquire` spaces this call to the model's RPM, and the dispatch // is wrapped in `withRateLimitRetry` so a 429 is honoured (retry the provider's // hint) rather than surfacing as a hard error mid-answer. One insertion covers // streamed answers AND the scripted gather loop (via askOnce). export async function askStream(opts: AskOptions): Promise { const model = opts.model || PROVIDERS[opts.provider]?.defaultModel || ""; await acquire(opts.provider, model, opts.signal); return withRateLimitRetry( () => { switch (opts.provider) { case "anthropic": return askAnthropic(opts); case "openai": return askOpenAI(opts); case "gemini": return askGemini(opts); default: throw new Error(`unknown provider: ${opts.provider}`); } }, { provider: opts.provider, model, signal: opts.signal }, ); } // Non-streaming single-shot completion, used by the scripted search loop to get // a short SEARCH:/DONE decision. Reuses the streaming machinery and collects the // full text; callers pass a small `maxTokens` to keep decision turns cheap. export function askOnce(opts: AskOptions): Promise { return askStream({ ...opts, onDelta: undefined }); } // ─── shared SSE plumbing ─── async function* sseLines( res: Response, signal?: AbortSignal, ): AsyncGenerator { if (!res.body) throw new Error("no response body"); const reader = res.body.getReader(); const decoder = new TextDecoder(); let buf = ""; try { while (true) { if (signal?.aborted) throw new DOMException("aborted", "AbortError"); const { done, value } = await reader.read(); if (done) break; buf += decoder.decode(value, { stream: true }); // SSE events are separated by a blank line; yield each `data:` payload. let idx: number; while ((idx = buf.indexOf("\n")) >= 0) { const line = buf.slice(0, idx).trim(); buf = buf.slice(idx + 1); if (line.startsWith("data:")) yield line.slice(5).trim(); } } } finally { reader.releaseLock(); } } // Thrown when a request fails because the conversation exceeds the model's // context window (a provider 400 with a token/context-length signature). Kept // distinct from a generic error — and from ToolsUnavailableError — so the chat // surfaces an actionable message instead of retrying the oversized request. export class ContextTooLargeError extends Error { constructor(providerMsg?: string) { super( "The conversation is too large for this model's context window. Start a " + "new chat, or open the Context panel and trim earlier turns, to continue." + (providerMsg ? `\n\n(${providerMsg})` : ""), ); this.name = "ContextTooLargeError"; } } // Heuristically detect a "context length exceeded" error from a provider's 400 // response body. Covers Anthropic ("prompt is too long"), OpenAI // ("context_length_exceeded" / "maximum context length"), and Gemini ("input // token count … exceeds"). Case-insensitive; matches the common phrasings. export function isContextLengthError(detail: string): boolean { const d = detail.toLowerCase(); return ( d.includes("context_length_exceeded") || d.includes("context length") || d.includes("maximum context") || d.includes("prompt is too long") || d.includes("too many tokens") || d.includes("input token count") || d.includes("reduce the length") || (d.includes("token") && d.includes("exceed")) ); } // Thrown when a request fails in a way that is transient and RESUMABLE rather // than fatal — a rate-limit / quota hit (HTTP 429). The retry wrapper catches it // to re-issue the request; when retries are exhausted the sweep driver catches it // and PAUSES (checkpoints the report + remaining batches) instead of aborting, so // the run can resume once the limit resets. `reason` is a short human string for // the paused banner; `retryAfterMs` carries the provider's own retry hint (when // present) so a retry waits exactly as long as asked. // // Hosted here (beside ContextTooLargeError) rather than in nativeTools/shared so // ensureOk can throw it without a shared↔askProvider import cycle; shared // re-exports it for its existing importers. export class PausableError extends Error { reason: string; retryAfterMs?: number; constructor(message: string, reason: string, retryAfterMs?: number) { super(message); this.name = "PausableError"; this.reason = reason; this.retryAfterMs = retryAfterMs; } } async function ensureOk( res: Response, provider: string, model: string, ): Promise { if (res.ok) return; let detail = ""; try { detail = await res.text(); } catch { /* ignore */ } const msg = `${provider} request failed (${res.status}). ${detail.slice(0, 300)}`; if (res.status === 400 && isContextLengthError(detail)) { throw new ContextTooLargeError(msg); } // Rate-limit / quota → resumable. Parse the provider's retry hint, step the // session RPM estimate down, and throw PausableError so the retry wrapper can // re-issue this request BEFORE any SSE body has been consumed (no duplicate // text). Previously a streaming 429 was a generic hard error. if (res.status === 429) { const retryAfterMs = parseRetryAfterMs(res.headers, detail) ?? undefined; noteRateLimited(provider.toLowerCase() as Provider, model); throw new PausableError(msg, "usage limit", retryAfterMs); } throw new Error(msg); } function emit(full: string[], chunk: string, onDelta?: (c: string) => void): void { if (!chunk) return; full.push(chunk); onDelta?.(chunk); } // ─── Anthropic ─── async function askAnthropic(opts: AskOptions): Promise { const t0 = Date.now(); const model = opts.model || PROVIDERS.anthropic.defaultModel; const res = await fetch("https://api.anthropic.com/v1/messages", { method: "POST", signal: opts.signal, headers: { "content-type": "application/json", "x-api-key": opts.apiKey, "anthropic-version": "2023-06-01", "anthropic-dangerous-direct-browser-access": "true", }, body: JSON.stringify({ model, max_tokens: opts.maxTokens ?? 1024, system: opts.system, messages: opts.messages.map((m) => ({ role: m.role, content: m.content })), stream: true, }), }); await ensureOk(res, "Anthropic", model); const full: string[] = []; let finishReason: FinishReason | undefined; let rawFinish: string | undefined; let usage: unknown; for await (const data of sseLines(res, opts.signal)) { if (!data || data === "[DONE]") continue; let evt: unknown; try { evt = JSON.parse(data); } catch { continue; } const e = evt as { type?: string; delta?: { type?: string; text?: string; stop_reason?: string }; usage?: unknown; }; if (e.type === "content_block_delta" && e.delta?.type === "text_delta") { emit(full, e.delta.text ?? "", opts.onDelta); } else if (e.type === "message_delta") { // The terminal event carries stop_reason ("max_tokens" = truncated) + usage. if (e.delta?.stop_reason) { rawFinish = e.delta.stop_reason; finishReason = normFinish(rawFinish); } if (e.usage) usage = e.usage; } } const text = full.join(""); opts.onDebug?.({ phase: "answer", provider: "anthropic", model, request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens }, response: text, finishReason, rawFinish, usage, ms: Date.now() - t0, }); return text; } // ─── OpenAI ─── async function askOpenAI(opts: AskOptions): Promise { const t0 = Date.now(); const model = opts.model || PROVIDERS.openai.defaultModel; const res = await fetch("https://api.openai.com/v1/chat/completions", { method: "POST", signal: opts.signal, headers: { "content-type": "application/json", authorization: `Bearer ${opts.apiKey}`, }, body: JSON.stringify({ model, stream: true, stream_options: { include_usage: true }, // Only cap when asked (the scripted search loop passes a small value); // otherwise let the provider default so answers aren't truncated. ...(opts.maxTokens ? { max_tokens: opts.maxTokens } : {}), messages: [ { role: "system", content: opts.system }, ...opts.messages.map((m) => ({ role: m.role, content: m.content })), ], }), }); await ensureOk(res, "OpenAI", model); const full: string[] = []; let finishReason: FinishReason | undefined; let rawFinish: string | undefined; let usage: unknown; for await (const data of sseLines(res, opts.signal)) { if (!data || data === "[DONE]") continue; let evt: unknown; try { evt = JSON.parse(data); } catch { continue; } const e = evt as { choices?: { delta?: { content?: string }; finish_reason?: string | null }[]; usage?: unknown; }; const chunk = e.choices?.[0]?.delta?.content; if (chunk) emit(full, chunk, opts.onDelta); const fr = e.choices?.[0]?.finish_reason; if (fr) { rawFinish = fr; finishReason = normFinish(fr); } if (e.usage) usage = e.usage; } const text = full.join(""); opts.onDebug?.({ phase: "answer", provider: "openai", model, request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens }, response: text, finishReason, rawFinish, usage, ms: Date.now() - t0, }); return text; } // ─── Google Gemini ─── async function askGemini(opts: AskOptions): Promise { const t0 = Date.now(); const model = opts.model || PROVIDERS.gemini.defaultModel; const url = `https://generativelanguage.googleapis.com/v1beta/models/` + `${encodeURIComponent(model)}:streamGenerateContent?alt=sse&key=${encodeURIComponent(opts.apiKey)}`; const res = await fetch(url, { method: "POST", signal: opts.signal, headers: { "content-type": "application/json" }, body: JSON.stringify({ system_instruction: { parts: [{ text: opts.system }] }, // Gemini uses role "model" for assistant turns. contents: opts.messages.map((m) => ({ role: m.role === "assistant" ? "model" : "user", parts: [{ text: m.content }], })), ...(opts.maxTokens ? { generationConfig: { maxOutputTokens: opts.maxTokens } } : {}), }), }); await ensureOk(res, "Gemini", model); const full: string[] = []; let finishReason: FinishReason | undefined; let rawFinish: string | undefined; let blockReason: string | undefined; let usage: unknown; for await (const data of sseLines(res, opts.signal)) { if (!data) continue; let evt: unknown; try { evt = JSON.parse(data); } catch { continue; } const e = evt as { candidates?: { content?: { parts?: { text?: string }[] }; finishReason?: string; }[]; promptFeedback?: { blockReason?: string }; usageMetadata?: unknown; }; const cand = e.candidates?.[0]; const parts = cand?.content?.parts ?? []; for (const p of parts) emit(full, p.text ?? "", opts.onDelta); // finishReason MAX_TOKENS = truncated; SAFETY/RECITATION = blocked (often // with no parts at all → an empty answer, which we must surface, not swallow). if (cand?.finishReason) { rawFinish = cand.finishReason; finishReason = normFinish(cand.finishReason); } if (e.promptFeedback?.blockReason) blockReason = e.promptFeedback.blockReason; if (e.usageMetadata) usage = e.usageMetadata; } const text = full.join(""); opts.onDebug?.({ phase: "answer", provider: "gemini", model, request: { system: opts.system, messages: opts.messages, maxTokens: opts.maxTokens }, response: text, finishReason, rawFinish, blockReason, usage, ms: Date.now() - t0, }); return text; }