diff --git a/CHANGELOG.md b/CHANGELOG.md index 320b16d..6ae2024 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added +- **The OpenAI-compatible endpoint backend sets `reasoning_effort`.** A typed + `reasoning_effort` (`none`, `minimal`, `low`, `medium`, `high` or `xhigh`) + on the endpoint config goes into every `/chat/completions` request body, + and leaving it unset sends nothing, as before. Python takes + `OpenAiEndpointBackend(..., reasoning_effort="none")`, typed + `TReasoningEffort`; Go takes `OpenAIOpts{ReasoningEffort: + ReasoningEffortNone}`; Rust takes `OpenAiEndpoint { reasoning_effort: + Some(ReasoningEffort::None), .. }`. Cerebras' `qwen-3.8-27b` reasons by + default. For one 850-token rewrite prompt it spent a median 6,300 + reasoning tokens to produce 520 output tokens, 4.4s at p50 and 7.4s at p95 + over 12 calls. With `reasoning_effort: "none"` the same calls took 0.73s at + p50 and 1.15s at p95. The Rust `OpenAiEndpoint` struct gains a public + field, so a struct literal that names every field must add it. + ## [0.13.4] - 2026-09-17 ### Security diff --git a/conformance/schema/run_spec.schema.json b/conformance/schema/run_spec.schema.json index f00ae73..44859e5 100644 --- a/conformance/schema/run_spec.schema.json +++ b/conformance/schema/run_spec.schema.json @@ -51,6 +51,17 @@ }, "model": { "type": "string" + }, + "reasoning_effort": { + "enum": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + null + ] } }, "type": [ diff --git a/conformance/vectors/plan/openai-endpoint-plain.json b/conformance/vectors/plan/openai-endpoint-plain.json index 41d7545..b07550e 100644 --- a/conformance/vectors/plan/openai-endpoint-plain.json +++ b/conformance/vectors/plan/openai-endpoint-plain.json @@ -19,7 +19,8 @@ "openai_endpoint": { "api_key": "sk-test", "base_url": "http://local.test/v1", - "model": "qwen3" + "model": "qwen3", + "reasoning_effort": null } }, "host": { diff --git a/conformance/vectors/plan/openai-endpoint-reasoning-effort.json b/conformance/vectors/plan/openai-endpoint-reasoning-effort.json new file mode 100644 index 0000000..2c808bb --- /dev/null +++ b/conformance/vectors/plan/openai-endpoint-reasoning-effort.json @@ -0,0 +1,48 @@ +{ + "name": "openai-endpoint-reasoning-effort", + "op": "plan", + "input": { + "provider": "openai_endpoint", + "spec": { + "prompt": "ping", + "model": "qwen3", + "agent": false, + "isolated": true, + "timeout": 180, + "max_attempts": 5, + "api_auth": false, + "schema": null, + "apple": null, + "claude": null, + "codex": null, + "gemini": null, + "openai_endpoint": { + "api_key": "sk-test", + "base_url": "http://local.test/v1", + "model": "qwen3", + "reasoning_effort": "none" + } + }, + "host": { + "platform": "darwin" + } + }, + "expected": { + "kind": "http", + "method": "POST", + "url": "http://local.test/v1/chat/completions", + "headers": { + "Authorization": "Bearer sk-test" + }, + "body": { + "model": "qwen3", + "messages": [ + { + "role": "user", + "content": "ping" + } + ], + "reasoning_effort": "none" + } + } +} diff --git a/conformance/vectors/plan/openai-endpoint-schema.json b/conformance/vectors/plan/openai-endpoint-schema.json index f43756f..2b21faf 100644 --- a/conformance/vectors/plan/openai-endpoint-schema.json +++ b/conformance/vectors/plan/openai-endpoint-schema.json @@ -29,7 +29,8 @@ "openai_endpoint": { "api_key": "sk-test", "base_url": "http://local.test/v1", - "model": "qwen3" + "model": "qwen3", + "reasoning_effort": null } }, "host": { diff --git a/go/backend.go b/go/backend.go index f21e5b0..a9d935f 100644 --- a/go/backend.go +++ b/go/backend.go @@ -122,11 +122,27 @@ func AntigravityBackend() Backend { return &cliBackend{provider: ProviderAntigra // passing it explicitly. func AppleBackend() Backend { return &cliBackend{provider: ProviderApple} } +// ReasoningEffort is the reasoning_effort an OpenAIEndpoint backend sends with +// every request. The empty string sends none, leaving the server default. +type ReasoningEffort string + +// The reasoning efforts an OpenAI-compatible /chat/completions server accepts. +const ( + ReasoningEffortNone ReasoningEffort = "none" + ReasoningEffortMinimal ReasoningEffort = "minimal" + ReasoningEffortLow ReasoningEffort = "low" + ReasoningEffortMedium ReasoningEffort = "medium" + ReasoningEffortHigh ReasoningEffort = "high" + ReasoningEffortXhigh ReasoningEffort = "xhigh" +) + // OpenAIOpts configures an OpenAIEndpoint backend. APIKey "" becomes "local"; -// Client nil uses http.DefaultClient. +// Client nil uses http.DefaultClient; ReasoningEffort "" sends no +// reasoning_effort. type OpenAIOpts struct { - APIKey string - Client *http.Client + APIKey string + Client *http.Client + ReasoningEffort ReasoningEffort } // OpenAIEndpoint returns a backend that POSTs to an OpenAI-compatible @@ -140,7 +156,7 @@ func OpenAIEndpoint(baseURL, model string, opts OpenAIOpts) Backend { if client == nil { client = http.DefaultClient } - return &openaiBackend{baseURL: baseURL, model: model, apiKey: apiKey, client: client} + return &openaiBackend{baseURL: baseURL, model: model, apiKey: apiKey, reasoningEffort: opts.ReasoningEffort, client: client} } func backendForProvider(p Provider) Backend { diff --git a/go/httpbackend.go b/go/httpbackend.go index f9cd0e1..b5180dd 100644 --- a/go/httpbackend.go +++ b/go/httpbackend.go @@ -9,10 +9,11 @@ import ( ) type openaiBackend struct { - baseURL string - model string - apiKey string - client *http.Client + baseURL string + model string + apiKey string + reasoningEffort ReasoningEffort + client *http.Client } func (b *openaiBackend) Provider() Provider { return ProviderOpenAIEndpoint } @@ -24,7 +25,11 @@ func (b *openaiBackend) CheckStatus(_ context.Context) BackendStatus { func (b *openaiBackend) execute(ctx context.Context, spec RunSpec, wantsValue bool) (*attempt, error) { cs := spec.core() cs.Model = b.model - cs.OpenAIEndpoint = &coreOpenAI{APIKey: b.apiKey, BaseURL: b.baseURL, Model: b.model} + var effort *ReasoningEffort + if b.reasoningEffort != "" { + effort = &b.reasoningEffort + } + cs.OpenAIEndpoint = &coreOpenAI{APIKey: b.apiKey, BaseURL: b.baseURL, Model: b.model, ReasoningEffort: effort} kind, _, plan, err := corePlan(ProviderOpenAIEndpoint, cs) if err != nil { return nil, err diff --git a/go/httpbackend_test.go b/go/httpbackend_test.go index 9dc4a99..78e0941 100644 --- a/go/httpbackend_test.go +++ b/go/httpbackend_test.go @@ -2,6 +2,7 @@ package spawnllm import ( "context" + "encoding/json" "errors" "io" "net" @@ -37,6 +38,36 @@ func TestOpenAIEndpointSuccess(t *testing.T) { } } +func TestOpenAIEndpointSendsReasoningEffortOnlyWhenSet(t *testing.T) { + for _, tc := range []struct { + name string + effort ReasoningEffort + want any + }{ + {"unset", "", nil}, + {"none", ReasoningEffortNone, "none"}, + } { + t.Run(tc.name, func(t *testing.T) { + var body map[string]any + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if err := json.NewDecoder(r.Body).Decode(&body); err != nil { + t.Errorf("decode body: %v", err) + } + _, _ = io.WriteString(w, `{"choices":[{"message":{"content":"pong"}}]}`) + })) + defer srv.Close() + + b := OpenAIEndpoint(srv.URL, "qwen3", OpenAIOpts{ReasoningEffort: tc.effort}) + if _, err := RunOn(context.Background(), b, RunSpec{Prompt: "ping"}); err != nil { + t.Fatalf("RunOn: %v", err) + } + if got, ok := body["reasoning_effort"]; got != tc.want || ok != (tc.want != nil) { + t.Fatalf("reasoning_effort = %v (present %v), want %v", got, ok, tc.want) + } + }) + } +} + func TestOpenAIEndpointErrorBody(t *testing.T) { srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { _, _ = io.WriteString(w, `{"error":{"message":"model is on fire"}}`) diff --git a/go/spec.go b/go/spec.go index 28655ce..04e036e 100644 --- a/go/spec.go +++ b/go/spec.go @@ -229,9 +229,10 @@ type coreGemini struct { } type coreOpenAI struct { - APIKey string `json:"api_key"` - BaseURL string `json:"base_url"` - Model string `json:"model"` + APIKey string `json:"api_key"` + BaseURL string `json:"base_url"` + Model string `json:"model"` + ReasoningEffort *ReasoningEffort `json:"reasoning_effort"` } func optString(s string) *string { diff --git a/rust/conformance-gen/fixtures/schema/run_spec.schema.json b/rust/conformance-gen/fixtures/schema/run_spec.schema.json index f00ae73..44859e5 100644 --- a/rust/conformance-gen/fixtures/schema/run_spec.schema.json +++ b/rust/conformance-gen/fixtures/schema/run_spec.schema.json @@ -51,6 +51,17 @@ }, "model": { "type": "string" + }, + "reasoning_effort": { + "enum": [ + "none", + "minimal", + "low", + "medium", + "high", + "xhigh", + null + ] } }, "type": [ diff --git a/rust/conformance-gen/src/cases.rs b/rust/conformance-gen/src/cases.rs index 5aa34e4..5a8d295 100644 --- a/rust/conformance-gen/src/cases.rs +++ b/rust/conformance-gen/src/cases.rs @@ -1,7 +1,7 @@ use serde_json::{Value, json}; use spawnllm_core::wire::{ AppleConfig, AppleGuardrails, AppleSampling, AppleUseCase, ClaudeConfig, CodexConfig, - GeminiConfig, OpenAiEndpoint, RunSpec, + GeminiConfig, OpenAiEndpoint, ReasoningEffort, RunSpec, }; pub struct Case { @@ -131,7 +131,11 @@ fn gemini_cfg_case(name: &str, cfg: GeminiConfig) -> Case { ) } -fn endpoint_case(name: &str, schema: Option) -> Case { +fn endpoint_case( + name: &str, + schema: Option, + reasoning_effort: Option, +) -> Case { plan_case( name, "openai_endpoint", @@ -141,6 +145,7 @@ fn endpoint_case(name: &str, schema: Option) -> Case { api_key: "sk-test".to_owned(), base_url: "http://local.test/v1".to_owned(), model: "qwen3".to_owned(), + reasoning_effort, }), ..spec("ping", "qwen3") }, @@ -561,8 +566,13 @@ fn plan_cases() -> Vec { ..AppleConfig::default() }, ), - endpoint_case("openai-endpoint-plain", None), - endpoint_case("openai-endpoint-schema", Some(schema_value())), + endpoint_case("openai-endpoint-plain", None, None), + endpoint_case("openai-endpoint-schema", Some(schema_value()), None), + endpoint_case( + "openai-endpoint-reasoning-effort", + None, + Some(ReasoningEffort::None), + ), ] } diff --git a/rust/spawnllm-core/src/plan/openai.rs b/rust/spawnllm-core/src/plan/openai.rs index 418f1d5..f88ba6c 100644 --- a/rust/spawnllm-core/src/plan/openai.rs +++ b/rust/spawnllm-core/src/plan/openai.rs @@ -16,6 +16,9 @@ pub(super) fn plan(spec: &RunSpec) -> InvocationPlan { json!([{"role": "user", "content": spec.prompt}]), ), ]); + if let Some(effort) = endpoint.reasoning_effort { + body.insert("reasoning_effort".to_string(), json!(effort)); + } if let Some(schema) = &spec.schema { body.insert( "response_format".to_string(), @@ -49,7 +52,7 @@ mod tests { use serde_json::{Value, json}; use super::*; - use crate::wire::OpenAiEndpoint; + use crate::wire::{OpenAiEndpoint, ReasoningEffort}; fn spec(schema: Option) -> RunSpec { RunSpec { @@ -69,6 +72,7 @@ mod tests { api_key: "sk-test".to_string(), base_url: "http://local.test/v1".to_string(), model: "qwen3".to_string(), + reasoning_effort: None, }), } } @@ -124,4 +128,25 @@ mod tests { }) ); } + + #[test] + fn reasoning_effort_plan_matches_vector() { + let mut spec = spec(None); + spec.openai_endpoint.as_mut().unwrap().reasoning_effort = Some(ReasoningEffort::None); + + assert_eq!( + serde_json::to_value(plan(&spec)).unwrap(), + json!({ + "kind": "http", + "method": "POST", + "url": "http://local.test/v1/chat/completions", + "headers": {"Authorization": "Bearer sk-test"}, + "body": { + "model": "qwen3", + "messages": [{"role": "user", "content": "ping"}], + "reasoning_effort": "none", + }, + }) + ); + } } diff --git a/rust/spawnllm-core/src/wire.rs b/rust/spawnllm-core/src/wire.rs index 2c3e4b8..a0e7bc1 100644 --- a/rust/spawnllm-core/src/wire.rs +++ b/rust/spawnllm-core/src/wire.rs @@ -96,11 +96,23 @@ pub struct GeminiConfig { pub extensions: Option>, } +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ReasoningEffort { + None, + Minimal, + Low, + Medium, + High, + Xhigh, +} + #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct OpenAiEndpoint { pub api_key: String, pub base_url: String, pub model: String, + pub reasoning_effort: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] diff --git a/rust/spawnllm/src/backend.rs b/rust/spawnllm/src/backend.rs index 48922dc..ae54e1a 100644 --- a/rust/spawnllm/src/backend.rs +++ b/rust/spawnllm/src/backend.rs @@ -14,6 +14,8 @@ use crate::spec::{ModelTier, Specialty}; #[cfg(feature = "openai")] use serde_json::Map; +#[cfg(feature = "openai")] +use spawnllm_core::wire::ReasoningEffort; /// An OpenAI-compatible `/chat/completions` endpoint. #[cfg(feature = "openai")] @@ -22,6 +24,7 @@ pub struct OpenAiEndpoint { pub api_key: String, pub base_url: String, pub model: String, + pub reasoning_effort: Option, } /// A concrete LLM backend: one of the five CLIs, or an OpenAI-compatible endpoint. @@ -98,6 +101,10 @@ impl Backend { ("api_key".to_owned(), json!(endpoint.api_key)), ("base_url".to_owned(), json!(endpoint.base_url)), ("model".to_owned(), json!(endpoint.model)), + ( + "reasoning_effort".to_owned(), + json!(endpoint.reasoning_effort), + ), ]))), _ => None, } diff --git a/rust/spawnllm/src/lib.rs b/rust/spawnllm/src/lib.rs index 13554e5..4e85355 100644 --- a/rust/spawnllm/src/lib.rs +++ b/rust/spawnllm/src/lib.rs @@ -28,6 +28,8 @@ mod http; pub use backend::OpenAiEndpoint; pub use backend::{Backend, BackendStatus, select_backend}; pub use error::{Error, RunError}; +#[cfg(feature = "openai")] +pub use spawnllm_core::wire::ReasoningEffort; pub use spec::{ AppleConfig, AppleGuardrails, AppleSampling, AppleUseCase, CallOpts, ClaudeConfig, CodexConfig, DiscardedAttempt, GeminiConfig, ModelTier, Response, RunResult, RunSpec, Specialty, diff --git a/spawnllm/__init__.py b/spawnllm/__init__.py index 7b7dd72..7c8a85f 100644 --- a/spawnllm/__init__.py +++ b/spawnllm/__init__.py @@ -34,7 +34,7 @@ from spawnllm.response import DiscardedAttempt, Error, Output, Response, Result from spawnllm.run import run, run_sync from spawnllm.spec import AppleConfig, ClaudeConfig, CodexConfig, GeminiConfig, RunSpec -from spawnllm.types import ProviderName, TModel, TSpecialty +from spawnllm.types import ProviderName, TModel, TReasoningEffort, TSpecialty __all__ = [ "AntigravityCliBackend", @@ -66,6 +66,7 @@ "Result", "RunSpec", "TModel", + "TReasoningEffort", "TSpecialty", "call", "call_sync", diff --git a/spawnllm/backends/openai_endpoint.py b/spawnllm/backends/openai_endpoint.py index d65c79b..96e8317 100644 --- a/spawnllm/backends/openai_endpoint.py +++ b/spawnllm/backends/openai_endpoint.py @@ -12,7 +12,7 @@ from spawnllm.backends.base import BackendStatus from spawnllm.response import Response from spawnllm.spec import RunSpec - from spawnllm.types import ProviderName, TModel + from spawnllm.types import ProviderName, TModel, TReasoningEffort class OpenAiEndpointBackend(LlmBackend): @@ -30,6 +30,9 @@ class OpenAiEndpointBackend(LlmBackend): model: The literal model id sent in every request body. api_key: Bearer token for the `Authorization` header; defaults to `"local"` for self-hosted servers that ignore it. + reasoning_effort: The `reasoning_effort` sent in every request body; + `None` omits it and leaves the server's default, which on a + reasoning model can spend most of a call's latency thinking. transport: Async transport injected into the `httpx.AsyncClient` used by `aexecute` — e.g. a record/replay caching transport; `None` uses httpx's default transport. The synchronous `execute` path always uses @@ -44,17 +47,29 @@ class OpenAiEndpointBackend(LlmBackend): schema_dialect: ClassVar[str | None] = "openai" def __init__( - self, base_url: str, model: str, *, api_key: str = "local", transport: httpx.AsyncBaseTransport | None = None + self, + base_url: str, + model: str, + *, + api_key: str = "local", + reasoning_effort: TReasoningEffort | None = None, + transport: httpx.AsyncBaseTransport | None = None, ) -> None: self.base_url = base_url.rstrip("/") self.model = model self.api_key = api_key + self.reasoning_effort = reasoning_effort self.transport = transport self.models: dict[TModel, str] = {"small": model, "medium": model, "large": model} def openai_section(self) -> dict[str, Any]: """Return the `openai_endpoint` wire section the core turns into the HTTP request.""" - return {"api_key": self.api_key, "base_url": self.base_url, "model": self.model} + return { + "api_key": self.api_key, + "base_url": self.base_url, + "model": self.model, + "reasoning_effort": self.reasoning_effort, + } def resolve(self, resp: httpx.Response, spec: RunSpec) -> Response: return self.to_response( diff --git a/spawnllm/types.py b/spawnllm/types.py index 82d3b55..dd95d14 100644 --- a/spawnllm/types.py +++ b/spawnllm/types.py @@ -4,7 +4,7 @@ from typing import Literal -__all__ = ["ProviderName", "TModel", "TSettingSource", "TSpecialty"] +__all__ = ["ProviderName", "TModel", "TReasoningEffort", "TSettingSource", "TSpecialty"] TSpecialty = Literal["debugging", "review", "general"] """Task specialty; `LlmBackends.for_specialty` maps each to its registered backend.""" @@ -15,5 +15,8 @@ TModel = Literal["small", "medium", "large"] """Abstract model tier; each backend maps it to a provider-specific model name.""" +TReasoningEffort = Literal["none", "minimal", "low", "medium", "high", "xhigh"] +"""A `reasoning_effort` an OpenAI-compatible endpoint accepts; `OpenAiEndpointBackend` sends it with every request.""" + ProviderName = Literal["claude", "claude-sdk", "codex", "gemini", "antigravity", "mlx", "apple", "openai_endpoint"] """Backend provider identifier; keys the per-backend `provider_configs` on a `RunSpec`.""" diff --git a/tests/test_backends.py b/tests/test_backends.py index d55e63c..e6073c0 100644 --- a/tests/test_backends.py +++ b/tests/test_backends.py @@ -136,7 +136,12 @@ def test_raw_string_schema_parsed_to_object(self) -> None: def test_openai_endpoint_section_set_only_by_owning_backend(self) -> None: assert ClaudeCliBackend().wire_spec(RunSpec(prompt="hi", model="haiku"))["openai_endpoint"] is None endpoint = OpenAiEndpointBackend(ENDPOINT, "q", api_key="sk").wire_spec(RunSpec(prompt="hi", model="q")) - assert endpoint["openai_endpoint"] == {"api_key": "sk", "base_url": ENDPOINT, "model": "q"} + assert endpoint["openai_endpoint"] == { + "api_key": "sk", + "base_url": ENDPOINT, + "model": "q", + "reasoning_effort": None, + } def test_schema_and_response_model_together_raise(self) -> None: with pytest.raises(ValueError, match="either response_model or schema"): @@ -648,6 +653,21 @@ def handler(request: httpx.Request) -> httpx.Response: assert resp.result.raw == "pong" assert resp.result.parsed is None + def test_execute_sends_reasoning_effort(self, monkeypatch: pytest.MonkeyPatch) -> None: + seen: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + seen["body"] = json.loads(request.content) + return httpx.Response(200, json=completion("pong")) + + mock_transport(monkeypatch, handler) + OpenAiEndpointBackend(ENDPOINT, "qwen3", reasoning_effort="none").execute(RunSpec(prompt="ping", model="q")) + assert seen["body"] == { + "model": "qwen3", + "messages": [{"role": "user", "content": "ping"}], + "reasoning_effort": "none", + } + async def test_aexecute_posts_and_reads_content(self, monkeypatch: pytest.MonkeyPatch) -> None: mock_transport(monkeypatch, lambda _request: httpx.Response(200, json=completion("pong"))) resp = await OpenAiEndpointBackend(ENDPOINT, "qwen3").aexecute(RunSpec(prompt="ping", model="q"))