From d25a44825f2c0eb10f45383cb905519842ebe8c9 Mon Sep 17 00:00:00 2001 From: imvir Date: Mon, 7 Sep 2026 18:35:44 +0530 Subject: [PATCH 1/2] fix(gateway): OpenCode Go/Zen session headers, per-model endpoints, max effort, failover Fixes #10. - Send stable x-opencode-session + distinct User-Agent on every Go/Zen request (fixes 400 MissingSessionID); mint per-conversation ids so first requests carry it. - Route by model endpoint instead of forcing /chat/completions: new GatewayResponsesAdapter (Responses API) and GatewayMessagesAdapter (Anthropic protocol over gateway base). AnthropicAdapter gained protected hooks; temperature/top_p omitted when thinking is on. - Always maximum reasoning effort on Go/Zen paths (chat: reasoning_effort=max, responses: reasoning.effort=high, messages: max-budget thinking); transparent one-shot retry with low effort on gateway [1210] thinking errors. - Router: immediate failover (no retry) on deterministic 400/401/403/404/410; ERROR-level hint for billing failures (CreditsError/RegionError); lastError recorded on all break paths. - Replace end-of-life NVIDIA default stepfun-ai/step-3.7-flash (410 Gone since 2026-08-28) with live minimaxai/minimax-m3; reasoning-effort hot-reloads via reloadRouter. --- proxy/models.json | 21 ++- proxy/package-lock.json | 4 +- proxy/src/adapters/anthropic.ts | 90 ++++++---- proxy/src/adapters/gateway-messages.ts | 34 ++++ proxy/src/adapters/gateway-responses.ts | 226 ++++++++++++++++++++++++ proxy/src/adapters/openai.ts | 18 +- proxy/src/adapters/opencode-go.ts | 127 +++++++++++-- proxy/src/adapters/zen.ts | 82 +++++++-- proxy/src/engine.ts | 2 + proxy/src/index.ts | 27 ++- proxy/src/opencode-endpoints.ts | 141 +++++++++++++++ proxy/src/router.ts | 60 +++++++ proxy/test/gateway-protocols.test.ts | 157 ++++++++++++++++ proxy/test/opencode-go-session.test.ts | 187 ++++++++++++++++++++ proxy/test/provider-adapters.test.ts | 8 +- proxy/test/router-classify.test.ts | 174 ++++++++++++++++++ 16 files changed, 1273 insertions(+), 85 deletions(-) create mode 100644 proxy/src/adapters/gateway-messages.ts create mode 100644 proxy/src/adapters/gateway-responses.ts create mode 100644 proxy/src/opencode-endpoints.ts create mode 100644 proxy/test/gateway-protocols.test.ts create mode 100644 proxy/test/opencode-go-session.test.ts create mode 100644 proxy/test/router-classify.test.ts diff --git a/proxy/models.json b/proxy/models.json index 2b20f3e..570bf37 100644 --- a/proxy/models.json +++ b/proxy/models.json @@ -9,7 +9,7 @@ "_title_model": "gpt-oss-120b-medium", "_fallback_model": "", "_default_provider": "nvidia", - "_default_model": "stepfun-ai/step-3.7-flash", + "_default_model": "minimaxai/minimax-m3", "_compaction_enabled": true, "_compaction_threshold": 0.8, "_compaction_model": "", @@ -24,7 +24,7 @@ "deepseek-v4-flash-free": 128000, "mimo-v2.5-free": 128000, "north-mini-code-free": 64000, - "stepfun-ai/step-3.7-flash": 128000 + "minimaxai/minimax-m3": 128000 }, "_provider_models": { "gemini-3.5-flash": { @@ -32,7 +32,10 @@ }, "gemini-3.1-pro": { "opencode-go": "deepseek-v4-pro", - "zen": ["deepseek-v4-flash-free", "mimo-v2.5-free"] + "zen": [ + "deepseek-v4-flash-free", + "mimo-v2.5-free" + ] }, "claude-sonnet-4-6-thinking": { "zen": "mimo-v2.5-free" @@ -41,13 +44,16 @@ "zen": "mimo-v2.5-free" }, "gpt-oss-120b-medium": { - "nvidia": ["stepfun-ai/step-3.7-flash", "deepseek-v4-pro"] + "nvidia": [ + "minimaxai/minimax-m3", + "deepseek-v4-pro" + ] }, "gemini-3.5-flash-low": { "zen": "deepseek-v4-flash-free" }, "gemini-3.5-flash-extra-low": { - "nvidia": "stepfun-ai/step-3.7-flash" + "nvidia": "minimaxai/minimax-m3" }, "gemini-3-flash-agent": { "zen": "north-mini-code-free" @@ -59,7 +65,8 @@ "zen": "north-mini-code-free" }, "default": { - "nvidia": "stepfun-ai/step-3.7-flash" + "nvidia": "minimaxai/minimax-m3", + "zen": "muse-spark-1.3-contributor-free" } } -} +} \ No newline at end of file diff --git a/proxy/package-lock.json b/proxy/package-lock.json index da84e76..180fc38 100644 --- a/proxy/package-lock.json +++ b/proxy/package-lock.json @@ -1,12 +1,12 @@ { "name": "@12errh/antigravity-proxy", - "version": "1.0.8", + "version": "1.0.11", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@12errh/antigravity-proxy", - "version": "1.0.8", + "version": "1.0.11", "dependencies": { "ai": "^6.0.193", "better-sqlite3": "^12.10.0", diff --git a/proxy/src/adapters/anthropic.ts b/proxy/src/adapters/anthropic.ts index df828a3..06f0562 100644 --- a/proxy/src/adapters/anthropic.ts +++ b/proxy/src/adapters/anthropic.ts @@ -6,14 +6,49 @@ import { parseToolArgs } from '../utils/parse-tool-args.js'; export class AnthropicAdapter implements ModelAdapter { provider = 'anthropic'; - private baseUrl: string; - private apiKey: string; + protected baseUrl: string; + protected apiKey: string; constructor(baseUrl: string, apiKey: string) { this.baseUrl = baseUrl.replace(/\/+$/, ''); this.apiKey = apiKey; } + /** Endpoint path — '/messages' for Anthropic and compatible gateways. */ + protected endpointPath(): string { + return '/messages'; + } + + /** Request headers. Overridden by gateway adapters (Bearer + session). */ + protected buildHeaders(_config?: Record): Record { + return { + 'Content-Type': 'application/json', + 'x-api-key': this.apiKey, + 'anthropic-version': '2023-06-01', + }; + } + + /** + * Thinking budget for `thinking: { type: 'enabled', budget_tokens }`. + * Returns null when thinking should be off. Overridden by gateway adapters + * that always run maximum thinking. + */ + protected thinkingBudget(maxTokens: number, config?: Record): number | null { + const providerOptions = (config as any)?.providerOptions; + const reasoningEffort: string | undefined = providerOptions?.openai?.reasoningEffort; + if (!reasoningEffort) return null; + // Approximate budget_tokens by effort level. Anthropic requires + // budget_tokens >= 1024, and <= max_tokens. + const budgetByLevel: Record = { + low: Math.min(2048, maxTokens - 1), + medium: Math.min(8192, maxTokens - 1), + high: Math.min(16384, maxTokens - 1), + }; + const budget = budgetByLevel[reasoningEffort] ?? Math.min(8192, maxTokens - 1); + // Ensure we respect Anthropic's minimum (1024) and stay below max_tokens. + return Math.max(1024, budget); + } + async *stream( model: string, messages: OpenAIMessage[], @@ -27,7 +62,7 @@ export class AnthropicAdapter implements ModelAdapter { ? [{ role: 'system' as const, content: system }, ...messages] : messages; const body = this.buildRequest(model, finalMessages, tools, config); - const response = await this.fetchResponse(body, signal); + const response = await this.fetchResponse(body, signal, config); const reader = response.body!.getReader(); const decoder = new TextDecoder(); @@ -100,7 +135,7 @@ export class AnthropicAdapter implements ModelAdapter { } } - private buildRequest( + protected buildRequest( model: string, messages: OpenAIMessage[], tools?: Record, @@ -109,9 +144,10 @@ export class AnthropicAdapter implements ModelAdapter { const systemMessages = messages.filter(m => m.role === 'system'); const nonSystemMessages = messages.filter(m => m.role !== 'system'); + const maxTokens = (config?.maxTokens as number) || 4096; const body: Record = { model, - max_tokens: (config?.maxTokens as number) || 4096, + max_tokens: maxTokens, stream: true, messages: this.convertMessages(nonSystemMessages), }; @@ -125,38 +161,26 @@ export class AnthropicAdapter implements ModelAdapter { input_schema: tool.parameters || { type: 'object', properties: {} }, })); } - if (config?.temperature != null) body.temperature = config.temperature; - if (config?.topP != null) body.top_p = config.topP; - if ((config as any)?.stopSequences?.length) body.stop_sequences = (config as any).stopSequences; // A3: translate OpenAI-style `reasoningEffort` to Anthropic's `thinking`. // Antigravity sets providerOptions.openai.reasoningEffort = 'low'|'medium'|'high' // when the user wants visible chain-of-thought. Anthropic uses a different // shape: `thinking: { type: 'enabled', budget_tokens: N }`. - // We only set this when reasoning is requested (avoids changing behavior - // for non-reasoning requests). - const providerOptions = (config as any)?.providerOptions; - const reasoningEffort: string | undefined = providerOptions?.openai?.reasoningEffort; - if (reasoningEffort) { - // Approximate budget_tokens by effort level. Anthropic requires - // budget_tokens >= 1024, and <= max_tokens. - const maxTokens = (body.max_tokens as number) || 4096; - const budgetByLevel: Record = { - low: Math.min(2048, maxTokens - 1), - medium: Math.min(8192, maxTokens - 1), - high: Math.min(16384, maxTokens - 1), - }; - const budget = budgetByLevel[reasoningEffort] - ?? Math.min(8192, maxTokens - 1); - // Ensure we respect Anthropic's minimum (1024) and stay below max_tokens. - const safeBudget = Math.max(1024, budget); - (body as any).thinking = { type: 'enabled', budget_tokens: safeBudget }; + const budget = this.thinkingBudget(maxTokens, config); + if (budget != null) { + (body as any).thinking = { type: 'enabled', budget_tokens: budget }; + } else { + // Anthropic rejects temperature/top_p alongside thinking — only send + // them for non-reasoning requests. + if (config?.temperature != null) body.temperature = config.temperature; + if (config?.topP != null) body.top_p = config.topP; } + if ((config as any)?.stopSequences?.length) body.stop_sequences = (config as any).stopSequences; return body; } - private convertMessages(messages: OpenAIMessage[]): any[] { + protected convertMessages(messages: OpenAIMessage[]): any[] { const result: any[] = []; for (const m of messages) { if (m.role === 'tool') { @@ -202,20 +226,16 @@ export class AnthropicAdapter implements ModelAdapter { return result; } - private async fetchResponse(body: Record, signal?: AbortSignal): Promise { - const response = await poolFetch(`${this.baseUrl}/messages`, { + protected async fetchResponse(body: Record, signal?: AbortSignal, config?: Record): Promise { + const response = await poolFetch(`${this.baseUrl}${this.endpointPath()}`, { method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'x-api-key': this.apiKey, - 'anthropic-version': '2023-06-01', - }, + headers: this.buildHeaders(config), body: JSON.stringify(body), signal, }); if (!response.ok) { const err = await response.text().catch(() => 'unknown'); - throw new Error(`[anthropic] API error ${response.status}: ${err}`); + throw new Error(`[${this.provider}] API error ${response.status}: ${err}`); } return response; } diff --git a/proxy/src/adapters/gateway-messages.ts b/proxy/src/adapters/gateway-messages.ts new file mode 100644 index 0000000..81990d7 --- /dev/null +++ b/proxy/src/adapters/gateway-messages.ts @@ -0,0 +1,34 @@ +/** + * Anthropic Messages API over an OpenCode gateway base URL (Go /messages, + * Zen /messages). Same protocol as AnthropicAdapter, but authenticates like + * the gateway expects (Bearer + stable x-opencode-session + distinct UA) + * and always runs maximum thinking. + */ + +import { randomUUID } from 'crypto'; +import { AnthropicAdapter } from './anthropic.js'; + +export class GatewayMessagesAdapter extends AnthropicAdapter { + constructor(baseUrl: string, apiKey: string, gatewayProvider: string) { + super(baseUrl, apiKey); + this.provider = gatewayProvider; + } + + protected override buildHeaders(config?: Record): Record { + const sessionId = (config as any)?.providerOptions?.sessionId || randomUUID(); + return { + 'Content-Type': 'application/json', + 'Authorization': `Bearer ${this.apiKey}`, + 'x-opencode-session': String(sessionId), + 'User-Agent': 'antigravity-proxy/1.0.7', + }; + } + + /** + * Always maximum thinking: largest budget that still leaves headroom for + * output (budget must stay below max_tokens). + */ + protected override thinkingBudget(maxTokens: number): number | null { + return Math.max(1024, Math.min(16384, maxTokens - 1024)); + } +} diff --git a/proxy/src/adapters/gateway-responses.ts b/proxy/src/adapters/gateway-responses.ts new file mode 100644 index 0000000..510bede --- /dev/null +++ b/proxy/src/adapters/gateway-responses.ts @@ -0,0 +1,226 @@ +/** + * OpenAI Responses API over an OpenCode gateway base URL (Go /responses, + * Zen /responses). Used for models the gateways do NOT serve over + * chat/completions (e.g. grok-4.6, gpt-5.6-luna, muse-spark contributors). + * Always runs maximum reasoning effort (`reasoning.effort: 'high'` — the + * highest Responses-API value with broad support). + */ + +import { randomUUID } from 'crypto'; +import type { OpenAIMessage } from '../mapper.js'; +import type { StreamChunk, ModelAdapter } from './types.js'; +import { poolFetch } from '../http-pool.js'; +import { parseToolArgs } from '../utils/parse-tool-args.js'; + +/** Convert OpenAI-style messages to Responses `input` items. */ +export function toResponsesInput(messages: OpenAIMessage[], system?: string): { input: any[]; instructions?: string } { + const systemTexts: string[] = []; + if (system) systemTexts.push(system); + const input: any[] = []; + for (const m of messages) { + if (m.role === 'system') { + if (typeof m.content === 'string' && m.content) systemTexts.push(m.content); + continue; + } + if (m.role === 'tool') { + input.push({ + type: 'function_call_output', + call_id: m.tool_call_id || '', + output: typeof m.content === 'string' ? m.content : JSON.stringify(m.content ?? ''), + }); + continue; + } + if (m.role === 'assistant' && m.tool_calls && m.tool_calls.length > 0) { + if (m.content) { + input.push({ type: 'message', role: 'assistant', content: contentToString(m.content) }); + } + for (const tc of m.tool_calls) { + input.push({ type: 'function_call', call_id: tc.id, name: tc.function.name, arguments: tc.function.arguments }); + } + continue; + } + input.push({ type: 'message', role: m.role, content: toInputContent(m.content) }); + } + return { input, instructions: systemTexts.length > 0 ? systemTexts.join('\n') : undefined }; +} + +function contentToString(content: OpenAIMessage['content']): string { + if (typeof content === 'string') return content; + if (Array.isArray(content)) { + return content.map((p: any) => (typeof p === 'string' ? p : p.text || '')).join(''); + } + return ''; +} + +function toInputContent(content: OpenAIMessage['content']): any { + if (typeof content === 'string' || content == null) return content ?? ''; + if (!Array.isArray(content)) return ''; + const blocks: any[] = []; + for (const p of content as any[]) { + if (typeof p === 'string') { + blocks.push({ type: 'input_text', text: p }); + } else if (p.type === 'text' && p.text) { + blocks.push({ type: 'input_text', text: p.text }); + } else if (p.type === 'image_url' && p.image_url?.url) { + blocks.push({ type: 'input_image', image_url: p.image_url.url }); + } + } + return blocks.length > 0 ? blocks : ''; +} + +export function buildResponsesRequest( + model: string, + messages: OpenAIMessage[], + tools?: Record, + config?: Record, + system?: string, +): Record { + const { input, instructions } = toResponsesInput(messages, system); + const body: Record = { + model, + input, + stream: true, + // Maximum reasoning effort. 'high' is the highest Responses-API value + // with broad model support. Temperature/top_p are deliberately omitted: + // reasoning models restrict them. + reasoning: { effort: 'high' }, + }; + if (instructions) body.instructions = instructions; + if (tools && Object.keys(tools).length > 0) { + body.tools = Object.entries(tools).map(([name, tool]: [string, any]) => ({ + type: 'function', + name, + description: tool.description || '', + parameters: tool.parameters || { type: 'object', properties: {} }, + })); + } + if (config?.maxTokens) body.max_output_tokens = config.maxTokens; + return body; +} + +/** + * Handle one parsed Responses SSE event. Returns the chunks it produces + * (possibly empty). Throws on terminal error events. + */ +export function handleResponsesEvent(event: any): StreamChunk[] { + if (!event || typeof event.type !== 'string') return []; + switch (event.type) { + case 'response.output_text.delta': + return event.delta ? [{ type: 'text', content: event.delta }] : []; + case 'response.reasoning_summary_text.delta': + return event.delta ? [{ type: 'thought', content: event.delta }] : []; + case 'response.output_item.done': { + const item = event.item; + if (item?.type === 'function_call') { + let args: Record = {}; + try { + args = parseToolArgs(item.arguments || '{}'); + } catch { /* keep empty */ } + return [{ type: 'tool-call', name: item.name || 'unknown', args }]; + } + return []; + } + case 'response.failed': { + const msg = event.response?.error?.message || 'response failed'; + throw new Error(msg); + } + case 'error': + throw new Error(event.message || event.error || 'responses error'); + default: + return []; + } +} + +/** Extract chunks from a complete (non-streaming) Responses object. */ +export function responsesObjectToChunks(data: any): StreamChunk[] { + const chunks: StreamChunk[] = []; + for (const item of data?.output || []) { + if (item?.type === 'message' && Array.isArray(item.content)) { + for (const part of item.content) { + if ((part.type === 'output_text') && part.text) chunks.push({ type: 'text', content: part.text }); + } + } else if (item?.type === 'function_call') { + let args: Record = {}; + try { + args = parseToolArgs(item.arguments || '{}'); + } catch { /* keep empty */ } + chunks.push({ type: 'tool-call', name: item.name || 'unknown', args }); + } + } + return chunks; +} + +export class GatewayResponsesAdapter implements ModelAdapter { + provider: string; + private baseUrl: string; + private apiKey: string; + + constructor(baseUrl: string, apiKey: string, gatewayProvider: string) { + this.baseUrl = baseUrl.replace(/\/+$/, ''); + this.apiKey = apiKey; + this.provider = gatewayProvider; + } + + private buildHeaders(config?: Record): Record { + const sessionId = (config as any)?.providerOptions?.sessionId || randomUUID(); + return { + 'Content-Type': 'application/json', + 'Authorization': `Bearer ${this.apiKey}`, + 'x-opencode-session': String(sessionId), + 'User-Agent': 'antigravity-proxy/1.0.7', + }; + } + + async *stream( + model: string, + messages: OpenAIMessage[], + tools?: Record, + config?: Record, + signal?: AbortSignal, + system?: string, + ): AsyncGenerator { + const body = buildResponsesRequest(model, messages, tools, config, system); + const response = await poolFetch(`${this.baseUrl}/responses`, { + method: 'POST', + headers: this.buildHeaders(config), + body: JSON.stringify(body), + signal, + }); + if (!response.ok) { + const err = await response.text().catch(() => 'unknown'); + throw new Error(`[${this.provider}] API error ${response.status}: ${err}`); + } + + const contentType = (response.headers.get('content-type') || '').toLowerCase(); + if (!contentType.includes('text/event-stream')) { + const data = await response.json() as any; + for (const chunk of responsesObjectToChunks(data)) yield chunk; + return; + } + + const reader = response.body!.getReader(); + const decoder = new TextDecoder(); + let buffer = ''; + try { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + buffer += decoder.decode(value, { stream: true }); + const lines = buffer.split('\n'); + buffer = lines.pop() || ''; + for (const line of lines) { + const trimmed = line.trim(); + if (!trimmed.startsWith('data: ')) continue; + const data = trimmed.slice(6).trim(); + if (data === '[DONE]') return; + let event: any; + try { event = JSON.parse(data); } catch { continue; } + if (event.type === 'response.completed' || event.type === 'response.incomplete') return; + for (const chunk of handleResponsesEvent(event)) yield chunk; + } + } + } finally { + reader.releaseLock(); + } + } +} diff --git a/proxy/src/adapters/openai.ts b/proxy/src/adapters/openai.ts index 7d76d9c..4555b87 100644 --- a/proxy/src/adapters/openai.ts +++ b/proxy/src/adapters/openai.ts @@ -59,7 +59,7 @@ export class OpenAICompatAdapter implements ModelAdapter { ? [{ role: 'system' as const, content: system }, ...messages] : messages; const body = this.buildRequest(model, finalMessages, tools, config); - const response = await this.fetchWithRetry(body, signal); + const response = await this.fetchWithRetry(body, signal, config); if (!this.isStreaming(response)) { const data = await response.json() as any; @@ -204,7 +204,7 @@ export class OpenAICompatAdapter implements ModelAdapter { // Providers that support OpenAI-specific params like reasoning_effort // (kept for potential future per-provider guards; effort is now driven by model config) - private static REASONING_PROVIDERS = new Set(['openai', 'zen', 'opencode-go', 'nvidia', 'openrouter', 'groq']); + private static REASONING_PROVIDERS = new Set(['openai', 'zen', 'opencode-go', 'nvidia', 'openrouter', 'groq', 'minimax']); protected buildRequest( model: string, @@ -256,13 +256,17 @@ export class OpenAICompatAdapter implements ModelAdapter { }); } - protected async fetchWithRetry(body: Record, signal?: AbortSignal): Promise { + protected buildHeaders(config?: Record): Record { + return { + 'Content-Type': 'application/json', + 'Authorization': `Bearer ${this.apiKey}`, + }; + } + + protected async fetchWithRetry(body: Record, signal?: AbortSignal, config?: Record): Promise { const response = await poolFetch(`${this.baseUrl}/chat/completions`, { method: 'POST', - headers: { - 'Content-Type': 'application/json', - 'Authorization': `Bearer ${this.apiKey}`, - }, + headers: this.buildHeaders(config), body: JSON.stringify(body), signal, }); diff --git a/proxy/src/adapters/opencode-go.ts b/proxy/src/adapters/opencode-go.ts index 8be7d87..85f57f2 100644 --- a/proxy/src/adapters/opencode-go.ts +++ b/proxy/src/adapters/opencode-go.ts @@ -1,22 +1,130 @@ /** * OpenCode Go adapter * - * Optimizations for the OpenCode Go API gateway: - * - Handles OpenCode Go-specific model routing - * - OpenCode Go-specific API headers and URL formatting - * - Handles reasoning field extraction for routed models - * - Captures and re-sends session_id for context cache discounts + * Endpoint-aware routing for the OpenCode Go gateway: the mapped MODEL is + * never switched — instead the request goes to the endpoint that model + * requires (see https://opencode.ai/docs/go/#endpoints): + * - /chat/completions — OpenAI-compatible (default) + * - /responses — Responses API (grok, gpt-luna, muse-spark contributors) + * - /messages — Anthropic Messages API (minimax, qwen families) + * + * Always runs maximum reasoning effort on every path, and captures + + * re-sends session_id for context cache discounts. */ import { OpenAICompatAdapter } from './openai.js'; -import { getEffortForModel } from '../reasoning-effort.js'; +import { randomUUID } from 'crypto'; +import { getGoEndpoint } from '../opencode-endpoints.js'; +import { GatewayMessagesAdapter } from './gateway-messages.js'; +import { GatewayResponsesAdapter } from './gateway-responses.js'; import type { OpenAIMessage } from '../mapper.js'; +import type { StreamChunk } from './types.js'; + +const THINKING_ERROR_PATTERNS = [/\[1210\]/, /always engages in thinking/i, /cannot be disabled/i]; + +// Effort values accepted by thinking-mandated Go models (per gateway error text). +const VALID_THINKING_EFFORTS = new Set(['low', 'high', 'max']); + +/** True when the gateway rejected the request because the model mandates thinking. */ +export function isThinkingRequiredError(err: unknown): boolean { + const msg = err instanceof Error ? err.message : String(err ?? ''); + return THINKING_ERROR_PATTERNS.some((r) => r.test(msg)); +} + +/** + * Run a Go request, transparently retrying ONCE with reasoning_effort=low + * when the gateway reports the model mandates thinking ([1210]). Safety net + * for non-max configurations; normally max effort is already sent. + * Never retries after data was already yielded (avoids duplicating content). + */ +export async function* streamWithThinkingRetry( + config: Record | undefined, + run: (cfg: Record | undefined) => AsyncGenerator, + signal?: AbortSignal, +): AsyncGenerator { + let yielded = false; + try { + for await (const chunk of run(config)) { + yielded = true; + yield chunk; + } + return; + } catch (err) { + if (signal?.aborted || yielded || !isThinkingRequiredError(err)) throw err; + const current = (config as any)?.providerOptions?.openai?.reasoningEffort; + if (VALID_THINKING_EFFORTS.has(String(current))) throw err; + const retryConfig: Record = { + ...(config || {}), + providerOptions: { + ...((config as any)?.providerOptions || {}), + openai: { ...((config as any)?.providerOptions?.openai || {}), reasoningEffort: 'low' }, + }, + }; + for await (const chunk of run(retryConfig)) { + yield chunk; + } + } +} + +/** Which gateway protocol handles this Go model. Pure — unit-tested. */ +export function selectGoHandler(model: string): 'chat' | 'responses' | 'messages' { + const endpoint = getGoEndpoint(model); + if (endpoint === 'responses') return 'responses'; + if (endpoint === 'messages') return 'messages'; + return 'chat'; +} export class OpencodeGoAdapter extends OpenAICompatAdapter { constructor(provider: string, baseUrl: string, apiKey: string) { super(provider, baseUrl, apiKey); } + /** + * OpenCode Go requires a stable `x-opencode-session` header per + * conversation for routing and prompt caching (see + * https://opencode.ai/docs/go/#where-can-i-use-it). Requests without it + * are rejected with `MissingSessionID`. The session id is threaded through + * `config.providerOptions.sessionId` by index.ts (stable per convId); + * if absent we generate a one-off UUID so the request is still routable. + * A distinct User-Agent (not a generic SDK name) is also required. + */ + protected buildHeaders(config?: Record): Record { + const sessionId = (config as any)?.providerOptions?.sessionId || randomUUID(); + return { + ...super.buildHeaders(config), + 'x-opencode-session': String(sessionId), + 'User-Agent': 'antigravity-proxy/1.0.7', + }; + } + + async *stream( + model: string, + messages: OpenAIMessage[], + tools?: Record, + config?: Record, + signal?: AbortSignal, + system?: string, + ): AsyncGenerator { + // Same model, applicable endpoint: only the URL path switches. + // Sub-adapters are built per call so hot-reloaded keys always apply. + const handler = selectGoHandler(model); + if (handler === 'responses') { + const adapter = new GatewayResponsesAdapter(this.baseUrl, this.apiKey, this.provider); + yield* adapter.stream(model, messages, tools, config, signal, system); + return; + } + if (handler === 'messages') { + const adapter = new GatewayMessagesAdapter(this.baseUrl, this.apiKey, this.provider); + yield* adapter.stream(model, messages, tools, config, signal, system); + return; + } + yield* streamWithThinkingRetry( + config, + (cfg) => super.stream(model, messages, tools, cfg, signal, system), + signal, + ); + } + protected buildRequest( model: string, messages: OpenAIMessage[], @@ -46,11 +154,8 @@ export class OpencodeGoAdapter extends OpenAICompatAdapter { if (config?.topP != null) body.top_p = config.topP; if ((config as any)?.stopSequences?.length) body.stop = (config as any).stopSequences; - // Reasoning effort - const explicitEffort = (config as any)?.providerOptions?.openai?.reasoningEffort; - const perModelEffort = getEffortForModel(model); - const effort = explicitEffort || (perModelEffort && perModelEffort !== 'default' ? perModelEffort : null); - if (effort) body.reasoning_effort = effort; + // Always maximum reasoning effort, regardless of per-model config. + body.reasoning_effort = 'max'; return body; } diff --git a/proxy/src/adapters/zen.ts b/proxy/src/adapters/zen.ts index 831cc1b..4b5a899 100644 --- a/proxy/src/adapters/zen.ts +++ b/proxy/src/adapters/zen.ts @@ -1,15 +1,34 @@ /** * Zen (OpenCode) adapter * - * Optimizations for the OpenCode / Zen API gateway: - * - Handles OpenCode-specific model routing - * - OpenCode-specific API headers and URL formatting - * - Handles reasoning field extraction for routed models + * Endpoint-aware routing for the OpenCode Zen gateway: the mapped MODEL is + * never switched — instead the request goes to the endpoint that model + * requires (see https://opencode.ai/docs/zen/#endpoints): + * - /chat/completions — OpenAI-compatible (default) + * - /responses — Responses API (GPT, Grok, Muse Spark) + * - /messages — Anthropic Messages API (Claude, Qwen) + * Zen Google-native model endpoints (/v1/models/) are not supported and + * fail fast with an actionable error. + * + * Always runs maximum reasoning effort on every path. */ import { OpenAICompatAdapter } from './openai.js'; -import { getEffortForModel } from '../reasoning-effort.js'; +import { randomUUID } from 'crypto'; +import { getZenEndpoint, endpointError, type GatewayEndpoint } from '../opencode-endpoints.js'; +import { GatewayMessagesAdapter } from './gateway-messages.js'; +import { GatewayResponsesAdapter } from './gateway-responses.js'; import type { OpenAIMessage } from '../mapper.js'; +import type { StreamChunk } from './types.js'; + +/** Which gateway protocol handles this Zen model. Pure — unit-tested. */ +export function selectZenHandler(model: string): 'chat' | 'responses' | 'messages' | 'google-native' { + const endpoint: GatewayEndpoint = getZenEndpoint(model); + if (endpoint === 'responses') return 'responses'; + if (endpoint === 'messages') return 'messages'; + if (endpoint === 'google-native') return 'google-native'; + return 'chat'; +} export class ZenAdapter extends OpenAICompatAdapter { constructor(provider: string, baseUrl: string, apiKey: string) { @@ -17,10 +36,48 @@ export class ZenAdapter extends OpenAICompatAdapter { } /** - * Override buildRequest to include OpenCode/Zen specific parameters: - * - Reasoning effort via provider-specific parameter name - * - Proper model name handling for the gateway + * Zen is an OpenCode gateway like Go: send a stable `x-opencode-session` + * header per conversation (the gateway answers `MissingSessionID` without + * it — including the "free tier can only be used in OpenCode" variant) + * plus a distinct User-Agent. Session id flows via + * `config.providerOptions.sessionId`; fall back to a generated UUID so the + * header is never absent. */ + protected buildHeaders(config?: Record): Record { + const sessionId = (config as any)?.providerOptions?.sessionId || randomUUID(); + return { + ...super.buildHeaders(config), + 'x-opencode-session': String(sessionId), + 'User-Agent': 'antigravity-proxy/1.0.7', + }; + } + + async *stream( + model: string, + messages: OpenAIMessage[], + tools?: Record, + config?: Record, + signal?: AbortSignal, + system?: string, + ): AsyncGenerator { + // Same model, applicable endpoint: only the URL path switches. + const handler = selectZenHandler(model); + if (handler === 'responses') { + const adapter = new GatewayResponsesAdapter(this.baseUrl, this.apiKey, this.provider); + yield* adapter.stream(model, messages, tools, config, signal, system); + return; + } + if (handler === 'messages') { + const adapter = new GatewayMessagesAdapter(this.baseUrl, this.apiKey, this.provider); + yield* adapter.stream(model, messages, tools, config, signal, system); + return; + } + if (handler === 'google-native') { + throw endpointError('zen', model, `models/${model}`); + } + yield* super.stream(model, messages, tools, config, signal, system); + } + protected buildRequest( model: string, messages: OpenAIMessage[], @@ -44,12 +101,9 @@ export class ZenAdapter extends OpenAICompatAdapter { if (config?.topP != null) body.top_p = config.topP; if ((config as any)?.stopSequences?.length) body.stop = (config as any).stopSequences; - // Reasoning effort: check explicit providerOptions first, then per-model config. - // Zen/OpenCode gateway forwards reasoning_effort to the underlying model. - const explicitEffort = (config as any)?.providerOptions?.openai?.reasoningEffort; - const perModelEffort = getEffortForModel(model); - const effort = explicitEffort || (perModelEffort && perModelEffort !== 'default' ? perModelEffort : null); - if (effort) body.reasoning_effort = effort; + // Always maximum reasoning effort, regardless of per-model config. + // The Zen gateway forwards reasoning_effort to the underlying model. + body.reasoning_effort = 'max'; return body; } diff --git a/proxy/src/engine.ts b/proxy/src/engine.ts index 4eb62f5..aa53bde 100644 --- a/proxy/src/engine.ts +++ b/proxy/src/engine.ts @@ -2,6 +2,7 @@ import { config } from './config.js'; import { logger } from './logger.js'; import { Router } from './router.js'; import { modelResolver } from './models.js'; +import { reload as reloadReasoningEffort } from './reasoning-effort.js'; import { ANTIGRAVITY_CONTEXT } from './antigravity-context.js'; import { registerBuiltinPlugins } from './plugins/builtin-plugins.js'; import { toolCapabilityRegistry } from './tool-capabilities.js'; @@ -111,6 +112,7 @@ export function reloadRouter(): void { router.updateProviders(config.providers, { retries: config.retries, backoffMs: config.backoffMs }); } modelResolver.reload(); + reloadReasoningEffort(); logger.info('[engine] Router, config, and model maps reloaded'); } diff --git a/proxy/src/index.ts b/proxy/src/index.ts index b1541eb..3ca54c3 100644 --- a/proxy/src/index.ts +++ b/proxy/src/index.ts @@ -3,6 +3,7 @@ import path from 'path'; import http from 'http'; import https from 'https'; import http2 from 'http2'; +import { randomUUID } from 'crypto'; import { fileURLToPath } from 'url'; import { config } from './config.js'; import { logger } from './logger.js'; @@ -21,6 +22,8 @@ import { installAgentContext } from './install-context.js'; import { USER_CERT_FILE, USER_KEY_FILE } from './data-paths.js'; import { getWorkspaceContextEnvelope, wrapToolResultForContextFile, isWorkspaceContextFile } from './workspace-context.js'; import { getSessionId, setSessionId } from './session-store.js'; +import { modelResolver } from './models.js'; +import { findEndpointMismatches } from './opencode-endpoints.js'; import { safeWrite } from './utils/safe-write.js'; import { formatErrorResponse } from './utils/error-response.js'; import { injectContext } from './context-injector.js'; @@ -435,13 +438,21 @@ async function handleStreamGenerate(req: http2.Http2ServerRequest, res: http2.Ht const convId = extractConvId(request.requestId); injectReasoning(mapped.messages, convId); - // Inject stored session_id for OpenCode Go context cache discounts - const storedSessionId = getSessionId(convId); - if (storedSessionId) { - if (!mapped.providerOptions) mapped.providerOptions = {}; - (mapped.providerOptions as any).sessionId = storedSessionId; + // OpenCode gateways require a stable x-opencode-session per conversation + // (MissingSessionID 400 otherwise). Reuse the stored session id when we + // have one; otherwise mint one now so even the FIRST request of a + // conversation carries the header. Later responses may return a + // server-side session_id which replaces this value for cache discounts. + let storedSessionId = getSessionId(convId); + if (!storedSessionId) { + storedSessionId = randomUUID(); + setSessionId(convId, storedSessionId); + logger.info(` Session minted: ${storedSessionId.substring(0, 12)}...`); + } else { logger.info(` Session cache hit: ${storedSessionId.substring(0, 12)}...`); } + if (!mapped.providerOptions) mapped.providerOptions = {}; + (mapped.providerOptions as any).sessionId = storedSessionId; logger.info(` Provider priority: ${config.providerPriority.join(', ')}`); @@ -849,6 +860,12 @@ async function main(): Promise { logger.info(`${config.provider}: ${config.baseUrl}`); + // Validate gateway mappings against per-model endpoint requirements + // (chat/completions vs responses vs messages vs google-native). + for (const w of findEndpointMismatches(modelResolver.getProviderMap())) { + logger.warn(`[gateway] ${w}`); + } + scanLocalProviders().then(async (results) => { const online = results.filter(p => p.online); if (online.length > 0) { diff --git a/proxy/src/opencode-endpoints.ts b/proxy/src/opencode-endpoints.ts new file mode 100644 index 0000000..33c8fdf --- /dev/null +++ b/proxy/src/opencode-endpoints.ts @@ -0,0 +1,141 @@ +/** + * OpenCode Go endpoint requirements. + * + * Go serves different models over different protocols — they are NOT all + * OpenAI-compatible chat/completions (see https://opencode.ai/docs/go/#endpoints): + * - `/chat/completions` — OpenAI-compatible (what this proxy speaks) + * - `/responses` — OpenAI Responses API (different request/response shape) + * - `/messages` — Anthropic Messages API (different request/response shape) + * + * Sending a responses/messages model to /chat/completions fails at the + * gateway, so mappings for such models must be caught with an actionable + * error instead of burning retries on an obscure 4xx. Unknown model ids are + * allowed through (docs lag behind the live model list). + */ + +export type GoEndpoint = 'chat/completions' | 'responses' | 'messages' | 'unknown'; + +export type GatewayEndpoint = 'chat' | 'responses' | 'messages' | 'google-native' | 'unknown'; + +const RESPONSES_MODELS = new Set([ + 'grok-4.6', + 'gpt-5.6-luna', + 'muse-spark-1.3-contributor', + 'muse-spark-1.2-contributor', +]); + +const MESSAGES_MODELS = new Set([ + 'minimax-m3', + 'minimax-m2.7', + 'minimax-m2.5', + 'qwen3.8-max', + 'qwen3.8-flash', + 'qwen3.7-max', + 'qwen3.7-plus', + 'qwen3.6-plus', +]); + +export function getGoEndpoint(modelId: string): GoEndpoint { + const id = modelId.replace(/^models\//, '').toLowerCase(); + if (RESPONSES_MODELS.has(id)) return 'responses'; + if (MESSAGES_MODELS.has(id)) return 'messages'; + return 'unknown'; +} + +/** True when the model can be served via POST /chat/completions. */ +export function isGoChatCompatible(modelId: string): boolean { + return getGoEndpoint(modelId) !== 'responses' && getGoEndpoint(modelId) !== 'messages'; +} + +// ─── Zen (https://opencode.ai/docs/zen/#endpoints) ─────────────────────── +// Same three protocols as Go, plus Google-native model endpoints +// (/v1/models/) which this proxy cannot speak. + +const ZEN_RESPONSES_MODELS = new Set([ + 'gpt-6-astra', + 'gpt-5.6-sol', 'gpt-5.6-terra', 'gpt-5.6-luna', + 'gpt-5.5', 'gpt-5.5-pro', + 'gpt-5.4', 'gpt-5.4-pro', 'gpt-5.4-mini', 'gpt-5.4-nano', + 'gpt-5.3-codex', 'gpt-5.3-codex-spark', + 'gpt-5.2', 'gpt-5.2-codex', + 'gpt-5.1', 'gpt-5.1-codex', 'gpt-5.1-codex-max', 'gpt-5.1-codex-mini', + 'gpt-5', 'gpt-5-codex', 'gpt-5-nano', + 'grok-4.6', 'grok-4.5', 'grok-build-0.1', + 'muse-spark-1.3', 'muse-spark-1.2', +]); + +const ZEN_MESSAGES_MODELS = new Set([ + 'claude-fable-5-1', 'claude-fable-5', + 'claude-opus-5', 'claude-opus-4-8', 'claude-opus-4-7', 'claude-opus-4-6', 'claude-opus-4-5', + 'claude-sonnet-5', 'claude-sonnet-4-6', 'claude-sonnet-4-5', + 'claude-haiku-4-5', + 'qwen3.7-max', 'qwen3.7-plus', 'qwen3.6-plus', 'qwen3.5-plus', +]); + +const ZEN_GOOGLE_NATIVE_MODELS = new Set([ + 'gemini-3.8-flash', 'gemini-3.7-flash', 'gemini-3.6-flash', + 'gemini-3.5-flash', 'gemini-3.5-flash-lite', + 'gemini-3.1-pro', 'gemini-3-flash', +]); + +export function getZenEndpoint(modelId: string): GatewayEndpoint { + const id = modelId.replace(/^models\//, '').toLowerCase(); + if (ZEN_RESPONSES_MODELS.has(id)) return 'responses'; + if (ZEN_MESSAGES_MODELS.has(id)) return 'messages'; + if (ZEN_GOOGLE_NATIVE_MODELS.has(id)) return 'google-native'; + return 'unknown'; +} + +/** + * Actionable error for a model whose gateway endpoint this proxy cannot + * speak (currently only Zen Google-native models). Matches the router's + * deterministic-failure pattern so it fails over without retrying. + */ +export function endpointError(gateway: 'opencode-go' | 'zen', model: string, endpoint: string): Error { + return new Error( + `[${gateway}] Model ${model} requires the ${gateway === 'zen' ? 'Zen' : 'Go'} /${endpoint} endpoint, ` + + `which this proxy does not support. ` + + `Remap it in models.json to a supported model.`, + ); +} + +/** + * Scan a provider map (`_provider_models`-shaped, values may be strings or + * fallback arrays) for mappings this proxy cannot serve. chat/responses/ + * messages are all handled — only Zen Google-native models are flagged. + * Returns human-readable warnings. + */ +export function findEndpointMismatches( + providerMap: Record>, +): string[] { + const warnings: string[] = []; + const check = (alias: string, model: unknown, gateway: 'opencode-go' | 'zen') => { + const models = Array.isArray(model) ? model : [model]; + for (const m of models) { + if (typeof m !== 'string') continue; + const endpoint = gateway === 'zen' ? getZenEndpoint(m) : getGoEndpoint(m); + if (endpoint === 'google-native') { + warnings.push( + `${alias} → ${gateway}:${m} requires a Google-native endpoint, ` + + `which this proxy does not support. Remap to a chat/responses/messages model.`, + ); + } + } + }; + for (const [alias, perProvider] of Object.entries(providerMap)) { + if (alias === 'default') continue; + check(alias, perProvider['opencode-go'], 'opencode-go'); + check(alias, perProvider['zen'], 'zen'); + } + const defaults = providerMap['default'] || {}; + check('default', defaults['opencode-go'], 'opencode-go'); + check('default', defaults['zen'], 'zen'); + return warnings; +} + +/** @deprecated Use findEndpointMismatches — responses/messages are now served. */ +export function findGoEndpointMismatches( + providerMap: Record>, +): string[] { + return findEndpointMismatches(providerMap); +} diff --git a/proxy/src/router.ts b/proxy/src/router.ts index 684c8e8..6c44b5a 100644 --- a/proxy/src/router.ts +++ b/proxy/src/router.ts @@ -21,6 +21,31 @@ export interface RouterOptions { const DEFAULT_OPTIONS: RouterOptions = { retries: 10, backoffMs: 1000 }; +export type ProviderErrorClass = 'retry' | 'failover'; + +// Matched first: these are always worth backing off and retrying. +const RETRYABLE_PATTERNS = [/429/, /rate_limit/i, /413/, /Request too large/i, /API error 5\d\d/, /timeout/i, /timed out/i, /fetch failed/i, /ECONN/i, /socket hang up/i, /Too Many Requests/]; + +// Only reached when nothing retryable matched: the same request can never +// succeed, so fail over immediately instead of burning latency/quota/money. +const DETERMINISTIC_PATTERNS = [/API error 400/, /API error 401/, /API error 403/, /API error 404/, /API error 410/, /which this proxy does not support/, /FreeUsageLimitError/]; + +/** + * Classify a provider failure as retryable (429/5xx/network — back off and + * retry) or deterministic (400/401/403/404/410 … — fail over without retrying). + */ +export function classifyProviderError(message: string): ProviderErrorClass { + const msg = message || ''; + if (RETRYABLE_PATTERNS.some((r) => r.test(msg))) return 'retry'; + if (DETERMINISTIC_PATTERNS.some((r) => r.test(msg))) return 'failover'; + return 'retry'; +} + +/** Account/billing failures need human action in the provider console. */ +export function isAccountFailure(message: string): boolean { + return /CreditsError|Insufficient balance|billing|RegionError|opt in/i.test(message || ''); +} + export class Router { private adapters = new Map(); private options: RouterOptions; @@ -185,9 +210,26 @@ export class Router { logger.error(`[router] ${providerId} model ${resolvedModel} failed mid-stream — cannot retry, failing over: ${err.message}`); fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', ...(isFallbackModel ? { modelFallback: true } : {}) }; + lastError = err.message; break; // Break out of retry loop, try next fallback model or next provider } + // Deterministic failures (400/401/403/404/410 …) can never succeed + // on retry. Account-wide failures skip the provider's remaining + // models too; model-specific ones advance to the next fallback model. + if (classifyProviderError(err.message) === 'failover') { + if (isAccountFailure(err.message)) { + logger.error(`[router] ${providerId} account/billing failure (check console billing or usage limits) — failing over without retry: ${err.message}`); + providerFullyFailed = true; + } else { + logger.warn(`[router] ${providerId} model ${resolvedModel} deterministic failure — failing over without retry: ${err.message}`); + } + fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); + yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', ...(isFallbackModel ? { modelFallback: true } : {}) }; + lastError = err.message; + break; + } + const isLastAttempt = attempt >= perProviderRetries; const isLastFallbackModel = modelIdx === fallbackModels.length - 1; const isLastProvider = candidates.indexOf(providerId) === candidates.length - 1; @@ -209,6 +251,7 @@ export class Router { logger.warn(`[router] ${providerId} model ${resolvedModel} exhausted: ${err.message}`); fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', ...(isFallbackModel ? { modelFallback: true } : {}) }; + lastError = err.message; } } } @@ -310,6 +353,22 @@ export class Router { logger.error(`[router] ${providerId} (fallback) model ${resolvedModel} failed mid-stream — failing over: ${err.message}`); fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', fallback: true, ...(isFallbackModel ? { modelFallback: true } : {}) }; + lastError = err.message; + break; + } + // Deterministic failures never succeed on retry — advance to the + // next fallback model (or provider). Account-wide failures skip + // the provider's remaining models too. + if (classifyProviderError(err.message) === 'failover') { + if (isAccountFailure(err.message)) { + logger.error(`[router] ${providerId} (fallback) account/billing failure (check console billing or usage limits) — trying next without retry: ${err.message}`); + providerFullyFailed = true; + } else { + logger.warn(`[router] ${providerId} model ${resolvedModel} (fallback) deterministic failure — trying next without retry: ${err.message}`); + } + fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); + yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', fallback: true, ...(isFallbackModel ? { modelFallback: true } : {}) }; + lastError = err.message; break; } const isLastAttempt = attempt >= fallbackRetries; @@ -332,6 +391,7 @@ export class Router { logger.warn(`[router] ${providerId} model ${resolvedModel} (fallback) exhausted: ${err.message}`); fireFailoverWebhook(providerId, resolvedModel, err.message, 'failover'); yield { type: 'attempt', provider: providerId, resolvedModel, attempt: attempt + 1, status: 'failover', fallback: true }; + lastError = err.message; } } } diff --git a/proxy/test/gateway-protocols.test.ts b/proxy/test/gateway-protocols.test.ts new file mode 100644 index 0000000..210b761 --- /dev/null +++ b/proxy/test/gateway-protocols.test.ts @@ -0,0 +1,157 @@ +/** + * Unit tests for gateway protocol handlers (Responses API + Anthropic + * Messages over Go/Zen base URLs) and Zen endpoint classification. + */ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { + buildResponsesRequest, + handleResponsesEvent, + responsesObjectToChunks, + toResponsesInput, +} from '../src/adapters/gateway-responses.js'; +import { GatewayMessagesAdapter } from '../src/adapters/gateway-messages.js'; +import { AnthropicAdapter } from '../src/adapters/anthropic.js'; +import { getZenEndpoint } from '../src/opencode-endpoints.js'; +import { selectZenHandler } from '../src/adapters/zen.js'; +import { selectGoHandler } from '../src/adapters/opencode-go.js'; + +// ─── Responses request building ───────────────────────────────────────── + +test('R1: buildResponsesRequest maps model/input/tools/reasoning', () => { + const body = buildResponsesRequest( + 'muse-spark-1.3-contributor', + [ + { role: 'user', content: 'hi' }, + { role: 'assistant', content: null, tool_calls: [{ id: 'c1', type: 'function', function: { name: 'run_command', arguments: '{"a":1}' } }] }, + { role: 'tool', tool_call_id: 'c1', content: 'done' }, + ], + { run_command: { description: 'Run', parameters: { type: 'object' } } }, + { maxTokens: 500 }, + 'be helpful', + ) as any; + assert.equal(body.model, 'muse-spark-1.3-contributor'); + assert.equal(body.stream, true); + assert.deepEqual(body.reasoning, { effort: 'high' }); + assert.equal(body.instructions, 'be helpful'); + assert.equal(body.max_output_tokens, 500); + assert.equal(body.temperature, undefined, 'temperature must be omitted for reasoning models'); + assert.equal(body.tools[0].type, 'function'); + assert.equal(body.tools[0].name, 'run_command'); + const types = body.input.map((i: any) => i.type); + assert.deepEqual(types, ['message', 'function_call', 'function_call_output']); + assert.equal(body.input[2].call_id, 'c1'); +}); + +test('R2: toResponsesInput folds history system messages into instructions', () => { + const { input, instructions } = toResponsesInput( + [ + { role: 'system', content: 'sys-a' }, + { role: 'user', content: 'hi' }, + ], + 'sys-b', + ); + assert.equal(instructions, 'sys-b\nsys-a'); + assert.equal(input.length, 1); + assert.equal(input[0].role, 'user'); +}); + +// ─── Responses event handling ─────────────────────────────────────────── + +test('R3: handleResponsesEvent maps deltas and tool calls', () => { + assert.deepEqual( + handleResponsesEvent({ type: 'response.output_text.delta', delta: 'hello' }), + [{ type: 'text', content: 'hello' }], + ); + assert.deepEqual( + handleResponsesEvent({ type: 'response.reasoning_summary_text.delta', delta: 'thinking…' }), + [{ type: 'thought', content: 'thinking…' }], + ); + assert.deepEqual(handleResponsesEvent({ type: 'response.created' }), []); + assert.deepEqual(handleResponsesEvent(null), []); + const tool = handleResponsesEvent({ + type: 'response.output_item.done', + item: { type: 'function_call', name: 'run_command', arguments: '{"CommandLine":"ls"}' }, + }); + assert.equal(tool.length, 1); + assert.equal(tool[0].type, 'tool-call'); + assert.equal(tool[0].name, 'run_command'); + assert.deepEqual(tool[0].args, { CommandLine: 'ls' }); +}); + +test('R4: handleResponsesEvent throws on terminal errors', () => { + assert.throws( + () => handleResponsesEvent({ type: 'response.failed', response: { error: { message: 'boom' } } }), + /boom/, + ); + assert.throws(() => handleResponsesEvent({ type: 'error', message: 'bad' }), /bad/); +}); + +test('R5: responsesObjectToChunks walks non-streaming output', () => { + const chunks = responsesObjectToChunks({ + output: [ + { type: 'message', content: [{ type: 'output_text', text: 'answer' }] }, + { type: 'function_call', name: 'x', arguments: '{}' }, + { type: 'reasoning', summary: [] }, + ], + }); + assert.equal(chunks.length, 2); + assert.equal(chunks[0].type, 'text'); + assert.equal(chunks[1].type, 'tool-call'); +}); + +// ─── Gateway messages adapter ─────────────────────────────────────────── + +test('M1: GatewayMessagesAdapter sends gateway headers (Bearer + session + UA)', () => { + const adapter = new GatewayMessagesAdapter('https://opencode.ai/zen/go/v1', 'sek', 'opencode-go'); + const headers = (adapter as any).buildHeaders({ providerOptions: { sessionId: 's-1' } }); + assert.equal(headers['Authorization'], 'Bearer sek'); + assert.equal(headers['x-opencode-session'], 's-1'); + assert.match(String(headers['User-Agent']), /antigravity/i); + assert.equal(headers['x-api-key'], undefined); +}); + +test('M2: GatewayMessagesAdapter always enables max thinking and strips temp', () => { + const adapter = new GatewayMessagesAdapter('https://opencode.ai/zen/go/v1', 'sek', 'opencode-go'); + const body = (adapter as any).buildRequest( + 'minimax-m3', + [{ role: 'user', content: 'hi' }], + undefined, + { maxTokens: 4096, temperature: 0.7, topP: 0.9 }, + ); + assert.equal(body.thinking?.type, 'enabled'); + assert.ok(body.thinking.budget_tokens >= 1024 && body.thinking.budget_tokens < 4096); + assert.equal(body.temperature, undefined, 'temp conflicts with thinking'); + assert.equal(body.top_p, undefined, 'top_p conflicts with thinking'); +}); + +test('M3: AnthropicAdapter unchanged — no thinking without effort, temp kept', () => { + const adapter = new AnthropicAdapter('https://api.anthropic.com/v1', 'k'); + const plain = (adapter as any).buildRequest('claude-x', [{ role: 'user', content: 'hi' }], undefined, { temperature: 0.5 }); + assert.equal(plain.thinking, undefined); + assert.equal(plain.temperature, 0.5); + const effort = (adapter as any).buildRequest('claude-x', [{ role: 'user', content: 'hi' }], undefined, { providerOptions: { openai: { reasoningEffort: 'high' } } }); + assert.equal(effort.thinking?.type, 'enabled'); + assert.equal(effort.temperature, undefined); +}); + +// ─── Zen endpoint classification ──────────────────────────────────────── + +test('Z1: getZenEndpoint classifies Zen models per docs', () => { + for (const m of ['gpt-5.6-luna', 'grok-4.6', 'muse-spark-1.3']) assert.equal(getZenEndpoint(m), 'responses', m); + for (const m of ['claude-sonnet-4-6', 'claude-opus-4-5', 'qwen3.7-max']) assert.equal(getZenEndpoint(m), 'messages', m); + for (const m of ['gemini-3.5-flash', 'gemini-3.1-pro', 'gemini-3-flash']) assert.equal(getZenEndpoint(m), 'google-native', m); + for (const m of ['mimo-v2.5-free', 'big-pickle', 'deepseek-v4-flash', 'some-future-model']) { + assert.equal(getZenEndpoint(m), 'unknown', `${m} falls through to chat`); + } +}); + +test('Z2: handler selection never switches the model, only the protocol', () => { + assert.equal(selectZenHandler('mimo-v2.5-free'), 'chat'); + assert.equal(selectZenHandler('claude-haiku-4-5'), 'messages'); + assert.equal(selectZenHandler('gpt-5-nano'), 'responses'); + assert.equal(selectZenHandler('gemini-3.1-pro'), 'google-native'); + assert.equal(selectGoHandler('muse-spark-1.3-contributor'), 'responses'); + assert.equal(selectGoHandler('qwen3.6-plus'), 'messages'); + assert.equal(selectGoHandler('omen-alpha'), 'chat'); +}); diff --git a/proxy/test/opencode-go-session.test.ts b/proxy/test/opencode-go-session.test.ts new file mode 100644 index 0000000..07bc248 --- /dev/null +++ b/proxy/test/opencode-go-session.test.ts @@ -0,0 +1,187 @@ +/** + * Regression tests for the opencode-go / zen / nvidia router failures: + * + * 1. opencode-go 400 MissingSessionID — the Go gateway requires a stable + * `x-opencode-session` header on EVERY request (including the first of a + * conversation) plus a non-generic User-Agent. + * See https://opencode.ai/docs/go/#where-can-i-use-it + * 2. zen 401 "Model omen-alpha is not supported" — the Go-only default model + * must never be sent to Zen; each priority provider needs its own default. + * 3. nvidia 404 — same root cause as (2): a foreign model id sent to NVIDIA. + * 4. opencode-go 400 [1210] thinking-required (glm-5.3-flash) — the model + * always thinks; requests without reasoning_effort low/high/max are + * rejected. Fixed via reasoning-effort.json entry. + * 5. nvidia 410 Gone — stepfun-ai/step-3.7-flash EOL 2026-08-28; replaced + * with live minimaxai/minimax-m3. + * 6. zen 400 MissingSessionID "free tier can only be used in OpenCode" — + * Zen gateway also requires x-opencode-session; ZenAdapter now sends it. + */ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import fs from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { OpencodeGoAdapter, selectGoHandler } from '../src/adapters/opencode-go.js'; +import { ZenAdapter, selectZenHandler } from '../src/adapters/zen.js'; +import { getGoEndpoint, isGoChatCompatible, findEndpointMismatches, endpointError } from '../src/opencode-endpoints.js'; +import { classifyProviderError } from '../src/router.js'; +import { modelResolver } from '../src/models.js'; + +// ─── x-opencode-session header ────────────────────────────────────────── + +test('G1: OpencodeGoAdapter sends x-opencode-session from providerOptions.sessionId', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'k'); + const headers = (adapter as any).buildHeaders({ providerOptions: { sessionId: 'sess-abc-123' } }); + assert.equal(headers['x-opencode-session'], 'sess-abc-123'); +}); + +test('G2: OpencodeGoAdapter still sends a session header when no sessionId configured (never MissingSessionID)', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'k'); + for (const cfg of [undefined, {}, { providerOptions: {} }]) { + const headers = (adapter as any).buildHeaders(cfg); + assert.ok(headers['x-opencode-session'], `expected fallback session header for config ${JSON.stringify(cfg)}`); + assert.ok(String(headers['x-opencode-session']).length >= 8, 'fallback session id should be non-trivial'); + } +}); + +test('G3: OpencodeGoAdapter identifies itself with its own User-Agent (not a generic SDK name)', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'k'); + const headers = (adapter as any).buildHeaders({ providerOptions: { sessionId: 's' } }); + assert.ok(headers['User-Agent'], 'expected User-Agent header'); + assert.match(String(headers['User-Agent']), /antigravity/i); +}); + +test('G4: OpencodeGoAdapter keeps Authorization + Content-Type headers', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'secret-key'); + const headers = (adapter as any).buildHeaders({ providerOptions: { sessionId: 's' } }); + assert.equal(headers['Authorization'], 'Bearer secret-key'); + assert.equal(headers['Content-Type'], 'application/json'); +}); + +test('G5: OpencodeGoAdapter buildRequest still puts session_id in body for cache discounts', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'k'); + const body = (adapter as any).buildRequest( + 'omen-alpha', + [{ role: 'user', content: 'hi' }], + undefined, + { providerOptions: { sessionId: 'sess-abc-123' } }, + ); + assert.equal(body.session_id, 'sess-abc-123'); +}); + +// ─── Per-provider model routing ───────────────────────────────────────── +// NOTE: assertions below are config-shape checks, not local-config checks, +// so they hold for any models.json (including upstream defaults). + +test('G6: provider defaults never leak a foreign-only model id', () => { + modelResolver.reload(); + for (const provider of ['opencode-go', 'zen', 'nvidia']) { + const def = modelResolver.getDefaultModel(provider); + if (def) { + const mismatches = findEndpointMismatches({ default: { [provider]: def } }); + assert.deepEqual(mismatches, [], `default ${provider} model unsupported: ${def}`); + } + } +}); + +// ─── Zen session header (free-tier MissingSessionID) ──────────────────── + +test('G9: ZenAdapter sends x-opencode-session from providerOptions.sessionId', () => { + const adapter = new ZenAdapter('zen', 'https://opencode.ai/zen/v1', 'k'); + const headers = (adapter as any).buildHeaders({ providerOptions: { sessionId: 'sess-zen-1' } }); + assert.equal(headers['x-opencode-session'], 'sess-zen-1'); + assert.match(String(headers['User-Agent']), /antigravity/i); +}); + +test('G10: ZenAdapter still sends a session header when none configured', () => { + const adapter = new ZenAdapter('zen', 'https://opencode.ai/zen/v1', 'k'); + const headers = (adapter as any).buildHeaders({}); + assert.ok(headers['x-opencode-session'], 'zen must never omit x-opencode-session'); +}); + +// ─── Always-maximum effort on Go/Zen chat path ───────────────────────── + +test('G11: OpencodeGoAdapter always sends reasoning_effort=max', () => { + const adapter = new OpencodeGoAdapter('opencode-go', 'https://opencode.ai/zen/go/v1', 'k'); + for (const m of ['omen-alpha', 'glm-5.3-flash', 'mimo-v2.5', 'kimi-k2.7-code']) { + const body = (adapter as any).buildRequest( + m, + [{ role: 'user', content: 'hi' }], + undefined, + { providerOptions: { sessionId: 's', openai: { reasoningEffort: 'low' } } }, + ); + assert.equal(body.reasoning_effort, 'max', `${m} must use max effort even when low requested`); + } +}); + +test('G12: ZenAdapter always sends reasoning_effort=max', () => { + const adapter = new ZenAdapter('zen', 'https://opencode.ai/zen/v1', 'k'); + const body = (adapter as any).buildRequest( + 'mimo-v2.5-free', + [{ role: 'user', content: 'hi' }], + undefined, + {}, + ); + assert.equal(body.reasoning_effort, 'max'); +}); + +// ─── Retired NVIDIA model must not be referenced ──────────────────────── + +test('G13: models.json contains no references to EOL stepfun-ai/step-3.7-flash', () => { + const __dirname = path.dirname(fileURLToPath(import.meta.url)); + const raw = fs.readFileSync(path.resolve(__dirname, '..', 'models.json'), 'utf-8'); + assert.ok(!raw.includes('step-3.7-flash'), 'EOL model still referenced in models.json'); + modelResolver.reload(); + assert.equal(modelResolver.getDefaultModel('nvidia'), 'minimaxai/minimax-m3'); +}); + +// ─── Go per-model endpoint requirements ─────────────────────────────── + +test('G14: documented Go endpoints are classified correctly', () => { + // chat/completions (OpenAI-compatible) + for (const m of ['omen-alpha', 'deepseek-v4-flash', 'glm-5.3-flash', 'kimi-k2.7-code', 'mimo-v2.5']) { + assert.equal(getGoEndpoint(m), 'unknown', `${m} should not be flagged (chat-compatible or unlisted)`); + assert.ok(isGoChatCompatible(m), `${m} should be chat-compatible`); + } + // Responses API models + for (const m of ['muse-spark-1.3-contributor', 'grok-4.6', 'gpt-5.6-luna']) { + assert.equal(getGoEndpoint(m), 'responses'); + assert.ok(!isGoChatCompatible(m)); + } + // Anthropic Messages API models + for (const m of ['minimax-m3', 'qwen3.7-max', 'qwen3.6-plus']) { + assert.equal(getGoEndpoint(m), 'messages'); + assert.ok(!isGoChatCompatible(m)); + } +}); + +test('G15: validator passes current mappings (only Zen google-native is flagged)', () => { + modelResolver.reload(); + const mismatches = findEndpointMismatches(modelResolver.getProviderMap()); + assert.deepEqual(mismatches, [], `unexpected mismatches: ${mismatches.join('; ')}`); +}); + +test('G16: Go/Zen handlers route by model endpoint (model never switched)', () => { + assert.equal(selectGoHandler('omen-alpha'), 'chat'); + assert.equal(selectGoHandler('muse-spark-1.3-contributor'), 'responses'); + assert.equal(selectGoHandler('minimax-m3'), 'messages'); + assert.equal(selectZenHandler('mimo-v2.5-free'), 'chat'); + assert.equal(selectZenHandler('claude-sonnet-4-6'), 'messages'); + assert.equal(selectZenHandler('gpt-5.6-luna'), 'responses'); + assert.equal(selectZenHandler('gemini-3.5-flash'), 'google-native'); +}); + +test('G17: validator flags Zen google-native mappings (unit)', () => { + const bad = findEndpointMismatches({ + 'some-alias': { 'opencode-go': 'minimax-m3', 'zen': 'gemini-3.5-flash' }, + 'default': { 'opencode-go': 'omen-alpha' }, + }); + assert.equal(bad.length, 1); + assert.match(bad[0], /some-alias.*zen:gemini-3\.5-flash/); +}); + +test('G18: endpoint guard error matches the router deterministic pattern', () => { + const err = endpointError('zen', 'gemini-3.5-flash', 'models/gemini-3.5-flash'); + assert.match(err.message, /which this proxy does not support/); + assert.equal(classifyProviderError(`[zen] API error 400: ${err.message}`), 'failover'); +}); diff --git a/proxy/test/provider-adapters.test.ts b/proxy/test/provider-adapters.test.ts index 233d9c5..6093e5e 100644 --- a/proxy/test/provider-adapters.test.ts +++ b/proxy/test/provider-adapters.test.ts @@ -105,7 +105,7 @@ test('A1: GroqAdapter passes standard params correctly', () => { // ─── ZenAdapter tests ──────────────────────────────────────────────────── -test('A2: ZenAdapter forwards reasoning_effort from providerOptions', () => { +test('A2: ZenAdapter always sends reasoning_effort=max', () => { const adapter = new ZenAdapter('zen', 'https://opencode.ai/zen/v1', 'test-key'); const body = (adapter as any).buildRequest( 'deepseek-r1', @@ -113,10 +113,10 @@ test('A2: ZenAdapter forwards reasoning_effort from providerOptions', () => { undefined, { providerOptions: { openai: { reasoningEffort: 'high' } } }, ) as any; - assert.equal(body.reasoning_effort, 'high', 'should forward reasoning_effort from providerOptions'); + assert.equal(body.reasoning_effort, 'max', 'must use max effort even when high requested'); }); -test('A2: ZenAdapter does not set reasoning_effort when not configured', () => { +test('A2: ZenAdapter sends reasoning_effort=max when not configured', () => { const adapter = new ZenAdapter('zen', 'https://opencode.ai/zen/v1', 'test-key'); const body = (adapter as any).buildRequest( 'deepseek-r1', @@ -124,7 +124,7 @@ test('A2: ZenAdapter does not set reasoning_effort when not configured', () => { undefined, {}, ) as any; - assert.equal(body.reasoning_effort, undefined, 'should not set reasoning_effort when not configured'); + assert.equal(body.reasoning_effort, 'max', 'must always use max effort'); }); test('A2: ZenAdapter passes standard params correctly', () => { diff --git a/proxy/test/router-classify.test.ts b/proxy/test/router-classify.test.ts new file mode 100644 index 0000000..db4b595 --- /dev/null +++ b/proxy/test/router-classify.test.ts @@ -0,0 +1,174 @@ +/** + * Unit tests for router error classification and Go thinking auto-retry. + * + * Root causes (from proxy session logs): + * - Deterministic gateway errors (400 [1210], 401 CreditsError/ModelError, + * 404, 410 Gone) were retried 4x with backoff before failover — burning + * latency, quota, and money on attempts that could never succeed. + * - Thinking-mandated Go models ([1210]) needed per-model config to work. + * - Terminal errors could surface as "All providers failed: unknown" because + * failover break-paths never recorded lastError. + */ +import { test } from 'node:test'; +import assert from 'node:assert/strict'; +import { classifyProviderError, isAccountFailure } from '../src/router.js'; +import { isThinkingRequiredError, streamWithThinkingRetry } from '../src/adapters/opencode-go.js'; +import type { StreamChunk } from '../src/adapters/types.js'; + +// ─── classifyProviderError ───────────────────────────────────────────── + +test('C1: deterministic failures fail over without retry', () => { + const cases = [ + '[opencode-go] API error 400: {"error":{"type":"server_error","message":"Upstream request failed: [1210] cannot be disabled"}}', + '[zen] API error 401: {"type":"error","error":{"type":"ModelError","message":"Model X is not supported"}}', + '[opencode-go] API error 401: {"type":"error","error":{"type":"CreditsError","message":"Insufficient balance"}}', + '[opencode-go] API error 400: {"type":"error","error":{"type":"MissingSessionID","message":"..."}}', + '[nvidia] API error 404: 404 page not found', + "[nvidia] API error 410: {\"title\":\"Gone\",\"detail\":\"model reached end of life\"}", + '[groq] API error 403: forbidden', + '[opencode-go] API error 403: {"type":"error","error":{"type":"RegionError","message":"only available hosted in China, requires opt in"}}', + '[opencode-go] Model muse-spark-1.3-contributor requires the Go /responses endpoint, which this proxy does not support (chat/completions only). Remap it in models.json.', + '[opencode-go] Model minimax-m3 requires the Go /messages endpoint, which this proxy does not support.', + ]; + for (const msg of cases) { + assert.equal(classifyProviderError(msg), 'failover', `expected failover for: ${msg.slice(0, 60)}`); + } +}); + +test('C1b: free-tier quota errors without a status code fail over; 429s stay retryable', () => { + assert.equal(classifyProviderError('FreeUsageLimitError: quota exhausted'), 'failover'); + assert.equal( + classifyProviderError('[zen] API error 429: {"type":"error","error":{"type":"FreeUsageLimitError","message":"Rate limit exceeded"}}'), + 'retry', + '429 status keeps retry-with-backoff semantics', + ); +}); + +test('C2: transient failures stay retryable', () => { + const cases = [ + '[zen] API error 500: {"type":"error","message":"Internal server error"}', + '[nvidia] API error 429: {"status":429,"title":"Too Many Requests"}', + '[zen] API error 429: {"type":"error","message":"Rate limit exceeded"}', + 'upstream timed out for POST /v1/chat/completions', + '[proxy] upstream error: fetch failed', + '[nvidia] API error 502: Bad Gateway', + '[nvidia] API error 503: Service Unavailable', + ]; + for (const msg of cases) { + assert.equal(classifyProviderError(msg), 'retry', `expected retry for: ${msg.slice(0, 60)}`); + } +}); + +test('C3: unknown errors default to retry (safe)', () => { + assert.equal(classifyProviderError('something weird happened'), 'retry'); + assert.equal(classifyProviderError(''), 'retry'); +}); + +test('C4: isAccountFailure flags billing errors only', () => { + assert.ok(isAccountFailure('[opencode-go] API error 401: {"type":"error","error":{"type":"CreditsError","message":"Insufficient balance"}}')); + assert.ok(isAccountFailure('[opencode-go] API error 403: {"type":"error","error":{"type":"RegionError","message":"requires explicit opt in"}}')); + assert.ok(!isAccountFailure('[zen] API error 500: Internal server error')); + assert.ok(!isAccountFailure('[nvidia] API error 429: Too Many Requests')); +}); + +// ─── isThinkingRequiredError ─────────────────────────────────────────── + +test('C5: thinking-required errors detected', () => { + assert.ok(isThinkingRequiredError(new Error('[1210] This model always engages in thinking and cannot be disabled; please use low, high, or max'))); + assert.ok(isThinkingRequiredError('[opencode-go] API error 400: cannot be disabled')); + assert.ok(!isThinkingRequiredError(new Error('Internal server error'))); + assert.ok(!isThinkingRequiredError(new Error('CreditsError: Insufficient balance'))); +}); + +// ─── streamWithThinkingRetry ─────────────────────────────────────────── + +function textRun(text: string): (cfg: Record | undefined) => AsyncGenerator { + return async function* () { + yield { type: 'text' as const, content: text }; + }; +} + +function failingRun(message: string): (cfg: Record | undefined) => AsyncGenerator { + return async function* () { + throw new Error(message); + }; +} + +const THINKING_MSG = '[opencode-go] API error 400: [1210] cannot be disabled; please use low, high, or max'; + +test('C6: retries once with reasoning_effort=low on [1210]', async () => { + const calls: Array | undefined> = []; + const run = async function* (cfg: Record | undefined) { + calls.push(cfg); + if (calls.length === 1) throw new Error(THINKING_MSG); + yield { type: 'text' as const, content: 'recovered' }; + }; + const out: string[] = []; + for await (const chunk of streamWithThinkingRetry(undefined, run)) { + if (chunk.type === 'text') out.push(chunk.content || ''); + } + assert.deepEqual(out, ['recovered']); + assert.equal(calls.length, 2); + assert.equal((calls[1] as any)?.providerOptions?.openai?.reasoningEffort, 'low'); +}); + +test('C7: non-thinking errors rethrow without retry', async () => { + let runs = 0; + const run = async function* (_cfg: Record | undefined) { + runs++; + throw new Error('CreditsError: Insufficient balance'); + }; + await assert.rejects( + async () => { for await (const _ of streamWithThinkingRetry(undefined, run)) { /* drain */ } }, + /CreditsError/, + ); + assert.equal(runs, 1); +}); + +test('C8: no retry when a valid effort was already sent', async () => { + let runs = 0; + const run = async function* (_cfg: Record | undefined) { + runs++; + throw new Error(THINKING_MSG); + }; + const cfg = { providerOptions: { openai: { reasoningEffort: 'max' } } }; + await assert.rejects( + async () => { for await (const _ of streamWithThinkingRetry(cfg, run)) { /* drain */ } }, + /1210/, + ); + assert.equal(runs, 1); +}); + +test('C9: no retry after data was already yielded (avoids duplicates)', async () => { + let runs = 0; + const run = async function* (_cfg: Record | undefined) { + runs++; + yield { type: 'text' as const, content: 'partial' }; + throw new Error(THINKING_MSG); + }; + const seen: string[] = []; + await assert.rejects( + async () => { + for await (const chunk of streamWithThinkingRetry(undefined, run)) { + if (chunk.type === 'text') seen.push(chunk.content || ''); + } + }, + /1210/, + ); + assert.deepEqual(seen, ['partial']); + assert.equal(runs, 1); +}); + +test('C10: first-try success runs once', async () => { + let runs = 0; + const run = async function* (_cfg: Record | undefined) { + runs++; + yield* textRun('ok')(_cfg); + }; + const out: string[] = []; + for await (const chunk of streamWithThinkingRetry({ a: 1 }, run)) { + if (chunk.type === 'text') out.push(chunk.content || ''); + } + assert.deepEqual(out, ['ok']); + assert.equal(runs, 1); +}); From 35d61befd33b525244165a5cbd512923f3e2b3e1 Mon Sep 17 00:00:00 2001 From: imvir Date: Mon, 7 Sep 2026 18:54:48 +0530 Subject: [PATCH 2/2] fix(gateway): normalize null message content, debug-log rejected bodies, flag DataPolicy - serializeMessages maps null content to '' (strict gateways reject null with 'messages illegal'-class 400s). - Adapters log the exact request body at debug level on non-OK responses so the next 4xx is diagnosable in seconds (nothing at default levels). - DataPolicyError counts as an account failure (console opt-in hint). --- proxy/src/adapters/anthropic.ts | 3 +++ proxy/src/adapters/gateway-responses.ts | 4 ++++ proxy/src/adapters/openai.ts | 11 ++++++++++- proxy/src/router.ts | 2 +- proxy/test/provider-adapters.test.ts | 11 ++++++++++- proxy/test/router-classify.test.ts | 1 + 6 files changed, 29 insertions(+), 3 deletions(-) diff --git a/proxy/src/adapters/anthropic.ts b/proxy/src/adapters/anthropic.ts index 06f0562..2e32bbf 100644 --- a/proxy/src/adapters/anthropic.ts +++ b/proxy/src/adapters/anthropic.ts @@ -235,6 +235,9 @@ export class AnthropicAdapter implements ModelAdapter { }); if (!response.ok) { const err = await response.text().catch(() => 'unknown'); + logger.debug(`[${this.provider}] rejected body`, { + body: JSON.stringify(body).substring(0, 4000), + }); throw new Error(`[${this.provider}] API error ${response.status}: ${err}`); } return response; diff --git a/proxy/src/adapters/gateway-responses.ts b/proxy/src/adapters/gateway-responses.ts index 510bede..9e5fca8 100644 --- a/proxy/src/adapters/gateway-responses.ts +++ b/proxy/src/adapters/gateway-responses.ts @@ -10,6 +10,7 @@ import { randomUUID } from 'crypto'; import type { OpenAIMessage } from '../mapper.js'; import type { StreamChunk, ModelAdapter } from './types.js'; import { poolFetch } from '../http-pool.js'; +import { logger } from '../logger.js'; import { parseToolArgs } from '../utils/parse-tool-args.js'; /** Convert OpenAI-style messages to Responses `input` items. */ @@ -188,6 +189,9 @@ export class GatewayResponsesAdapter implements ModelAdapter { }); if (!response.ok) { const err = await response.text().catch(() => 'unknown'); + logger.debug(`[${this.provider}] rejected body for ${model}`, { + body: JSON.stringify(body).substring(0, 4000), + }); throw new Error(`[${this.provider}] API error ${response.status}: ${err}`); } diff --git a/proxy/src/adapters/openai.ts b/proxy/src/adapters/openai.ts index 4555b87..ed4b350 100644 --- a/proxy/src/adapters/openai.ts +++ b/proxy/src/adapters/openai.ts @@ -2,6 +2,7 @@ import type { OpenAIMessage } from '../mapper.js'; import type { StreamChunk, ModelAdapter } from './types.js'; import { poolFetch } from '../http-pool.js'; import { getEffortForModel } from '../reasoning-effort.js'; +import { logger } from '../logger.js'; import { parseToolArgs } from '../utils/parse-tool-args.js'; /** @@ -247,7 +248,10 @@ export class OpenAICompatAdapter implements ModelAdapter { }).filter(Boolean); out.content = cleaned.length === 0 ? '' : cleaned; } else { - out.content = m.content; + // Null content (emitted by the mapper for tool-call-only assistant + // turns) is spec-legal but rejected as "illegal" by strict gateways — + // normalize to '' which is accepted everywhere null is. + out.content = m.content ?? ''; } if (m.tool_calls) out.tool_calls = m.tool_calls; if (m.tool_call_id) out.tool_call_id = m.tool_call_id; @@ -272,6 +276,11 @@ export class OpenAICompatAdapter implements ModelAdapter { }); if (!response.ok) { const err = await response.text().catch(() => 'unknown'); + // Debug-level exact body: 400s like "[1214] messages illegal" are only + // diagnosable with the precise payload. Never logged at default levels. + logger.debug(`[${this.provider}] rejected body for ${body['model']}`, { + body: JSON.stringify(body).substring(0, 4000), + }); throw new Error(`[${this.provider}] API error ${response.status}: ${err}`); } return response; diff --git a/proxy/src/router.ts b/proxy/src/router.ts index 6c44b5a..273a740 100644 --- a/proxy/src/router.ts +++ b/proxy/src/router.ts @@ -43,7 +43,7 @@ export function classifyProviderError(message: string): ProviderErrorClass { /** Account/billing failures need human action in the provider console. */ export function isAccountFailure(message: string): boolean { - return /CreditsError|Insufficient balance|billing|RegionError|opt in/i.test(message || ''); + return /CreditsError|Insufficient balance|billing|RegionError|DataPolicyError|opt in/i.test(message || ''); } export class Router { diff --git a/proxy/test/provider-adapters.test.ts b/proxy/test/provider-adapters.test.ts index 6093e5e..64bf937 100644 --- a/proxy/test/provider-adapters.test.ts +++ b/proxy/test/provider-adapters.test.ts @@ -13,6 +13,7 @@ import assert from 'node:assert/strict'; import { GroqAdapter } from '../src/adapters/groq.js'; import { ZenAdapter } from '../src/adapters/zen.js'; import { NvidiaAdapter } from '../src/adapters/nvidia.js'; +import { OpenAICompatAdapter } from '../src/adapters/openai.js'; import type { OpenAIMessage } from '../src/mapper.js'; // ─── GroqAdapter tests ─────────────────────────────────────────────────── @@ -226,7 +227,15 @@ test('A3: NvidiaAdapter serializes tools correctly', () => { assert.equal(body.tools[0].function.name, 'search'); }); -// ─── Cross-adapter consistency tests ───────────────────────────────────── +test('A4: serializeMessages normalizes null content to empty string', () => { + const adapter = new OpenAICompatAdapter('go', 'http://go', 'k'); + const out = (adapter as any).serializeMessages([ + { role: 'assistant', content: null, tool_calls: [{ id: 'c1', type: 'function', function: { name: 'x', arguments: '{}' } }] }, + { role: 'user', content: 'hi' }, + ]); + assert.equal(out[0].content, '', 'null content must become empty string (strict gateways reject null)'); + assert.equal(out[1].content, 'hi'); +}); test('A4: All provider adapters set model and stream:true', () => { const groq = new GroqAdapter('groq', 'http://groq', 'k'); diff --git a/proxy/test/router-classify.test.ts b/proxy/test/router-classify.test.ts index db4b595..851ecc9 100644 --- a/proxy/test/router-classify.test.ts +++ b/proxy/test/router-classify.test.ts @@ -67,6 +67,7 @@ test('C3: unknown errors default to retry (safe)', () => { test('C4: isAccountFailure flags billing errors only', () => { assert.ok(isAccountFailure('[opencode-go] API error 401: {"type":"error","error":{"type":"CreditsError","message":"Insufficient balance"}}')); assert.ok(isAccountFailure('[opencode-go] API error 403: {"type":"error","error":{"type":"RegionError","message":"requires explicit opt in"}}')); + assert.ok(isAccountFailure('[opencode-go] API error 403: {"type":"error","error":{"type":"DataPolicyError","message":"requires explicit opt in"}}')); assert.ok(!isAccountFailure('[zen] API error 500: Internal server error')); assert.ok(!isAccountFailure('[nvidia] API error 429: Too Many Requests')); });