Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- **The OpenAI-compatible endpoint backend sets `reasoning_effort`.** A typed
`reasoning_effort` (`none`, `minimal`, `low`, `medium`, `high` or `xhigh`)
on the endpoint config goes into every `/chat/completions` request body,
and leaving it unset sends nothing, as before. Python takes
`OpenAiEndpointBackend(..., reasoning_effort="none")`, typed
`TReasoningEffort`; Go takes `OpenAIOpts{ReasoningEffort:
ReasoningEffortNone}`; Rust takes `OpenAiEndpoint { reasoning_effort:
Some(ReasoningEffort::None), .. }`. Cerebras' `qwen-3.8-27b` reasons by
default. For one 850-token rewrite prompt it spent a median 6,300
reasoning tokens to produce 520 output tokens, 4.4s at p50 and 7.4s at p95
over 12 calls. With `reasoning_effort: "none"` the same calls took 0.73s at
p50 and 1.15s at p95. The Rust `OpenAiEndpoint` struct gains a public
field, so a struct literal that names every field must add it.

## [0.13.4] - 2026-09-17

### Security
Expand Down
11 changes: 11 additions & 0 deletions conformance/schema/run_spec.schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,17 @@
},
"model": {
"type": "string"
},
"reasoning_effort": {
"enum": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
null
]
}
},
"type": [
Expand Down
3 changes: 2 additions & 1 deletion conformance/vectors/plan/openai-endpoint-plain.json
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,8 @@
"openai_endpoint": {
"api_key": "sk-test",
"base_url": "http://local.test/v1",
"model": "qwen3"
"model": "qwen3",
"reasoning_effort": null
}
},
"host": {
Expand Down
48 changes: 48 additions & 0 deletions conformance/vectors/plan/openai-endpoint-reasoning-effort.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
{
"name": "openai-endpoint-reasoning-effort",
"op": "plan",
"input": {
"provider": "openai_endpoint",
"spec": {
"prompt": "ping",
"model": "qwen3",
"agent": false,
"isolated": true,
"timeout": 180,
"max_attempts": 5,
"api_auth": false,
"schema": null,
"apple": null,
"claude": null,
"codex": null,
"gemini": null,
"openai_endpoint": {
"api_key": "sk-test",
"base_url": "http://local.test/v1",
"model": "qwen3",
"reasoning_effort": "none"
}
},
"host": {
"platform": "darwin"
}
},
"expected": {
"kind": "http",
"method": "POST",
"url": "http://local.test/v1/chat/completions",
"headers": {
"Authorization": "Bearer sk-test"
},
"body": {
"model": "qwen3",
"messages": [
{
"role": "user",
"content": "ping"
}
],
"reasoning_effort": "none"
}
}
}
3 changes: 2 additions & 1 deletion conformance/vectors/plan/openai-endpoint-schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,8 @@
"openai_endpoint": {
"api_key": "sk-test",
"base_url": "http://local.test/v1",
"model": "qwen3"
"model": "qwen3",
"reasoning_effort": null
}
},
"host": {
Expand Down
24 changes: 20 additions & 4 deletions go/backend.go
Original file line number Diff line number Diff line change
Expand Up @@ -122,11 +122,27 @@ func AntigravityBackend() Backend { return &cliBackend{provider: ProviderAntigra
// passing it explicitly.
func AppleBackend() Backend { return &cliBackend{provider: ProviderApple} }

// ReasoningEffort is the reasoning_effort an OpenAIEndpoint backend sends with
// every request. The empty string sends none, leaving the server default.
type ReasoningEffort string

// The reasoning efforts an OpenAI-compatible /chat/completions server accepts.
const (
ReasoningEffortNone ReasoningEffort = "none"
ReasoningEffortMinimal ReasoningEffort = "minimal"
ReasoningEffortLow ReasoningEffort = "low"
ReasoningEffortMedium ReasoningEffort = "medium"
ReasoningEffortHigh ReasoningEffort = "high"
ReasoningEffortXhigh ReasoningEffort = "xhigh"
)

// OpenAIOpts configures an OpenAIEndpoint backend. APIKey "" becomes "local";
// Client nil uses http.DefaultClient.
// Client nil uses http.DefaultClient; ReasoningEffort "" sends no
// reasoning_effort.
type OpenAIOpts struct {
APIKey string
Client *http.Client
APIKey string
Client *http.Client
ReasoningEffort ReasoningEffort
}

// OpenAIEndpoint returns a backend that POSTs to an OpenAI-compatible
Expand All @@ -140,7 +156,7 @@ func OpenAIEndpoint(baseURL, model string, opts OpenAIOpts) Backend {
if client == nil {
client = http.DefaultClient
}
return &openaiBackend{baseURL: baseURL, model: model, apiKey: apiKey, client: client}
return &openaiBackend{baseURL: baseURL, model: model, apiKey: apiKey, reasoningEffort: opts.ReasoningEffort, client: client}
}

func backendForProvider(p Provider) Backend {
Expand Down
15 changes: 10 additions & 5 deletions go/httpbackend.go
Original file line number Diff line number Diff line change
Expand Up @@ -9,10 +9,11 @@ import (
)

type openaiBackend struct {
baseURL string
model string
apiKey string
client *http.Client
baseURL string
model string
apiKey string
reasoningEffort ReasoningEffort
client *http.Client
}

func (b *openaiBackend) Provider() Provider { return ProviderOpenAIEndpoint }
Expand All @@ -24,7 +25,11 @@ func (b *openaiBackend) CheckStatus(_ context.Context) BackendStatus {
func (b *openaiBackend) execute(ctx context.Context, spec RunSpec, wantsValue bool) (*attempt, error) {
cs := spec.core()
cs.Model = b.model
cs.OpenAIEndpoint = &coreOpenAI{APIKey: b.apiKey, BaseURL: b.baseURL, Model: b.model}
var effort *ReasoningEffort
if b.reasoningEffort != "" {
effort = &b.reasoningEffort
}
cs.OpenAIEndpoint = &coreOpenAI{APIKey: b.apiKey, BaseURL: b.baseURL, Model: b.model, ReasoningEffort: effort}
kind, _, plan, err := corePlan(ProviderOpenAIEndpoint, cs)
if err != nil {
return nil, err
Expand Down
31 changes: 31 additions & 0 deletions go/httpbackend_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ package spawnllm

import (
"context"
"encoding/json"
"errors"
"io"
"net"
Expand Down Expand Up @@ -37,6 +38,36 @@ func TestOpenAIEndpointSuccess(t *testing.T) {
}
}

func TestOpenAIEndpointSendsReasoningEffortOnlyWhenSet(t *testing.T) {
for _, tc := range []struct {
name string
effort ReasoningEffort
want any
}{
{"unset", "", nil},
{"none", ReasoningEffortNone, "none"},
} {
t.Run(tc.name, func(t *testing.T) {
var body map[string]any
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if err := json.NewDecoder(r.Body).Decode(&body); err != nil {
t.Errorf("decode body: %v", err)
}
_, _ = io.WriteString(w, `{"choices":[{"message":{"content":"pong"}}]}`)
}))
defer srv.Close()

b := OpenAIEndpoint(srv.URL, "qwen3", OpenAIOpts{ReasoningEffort: tc.effort})
if _, err := RunOn(context.Background(), b, RunSpec{Prompt: "ping"}); err != nil {
t.Fatalf("RunOn: %v", err)
}
if got, ok := body["reasoning_effort"]; got != tc.want || ok != (tc.want != nil) {
t.Fatalf("reasoning_effort = %v (present %v), want %v", got, ok, tc.want)
}
})
}
}

func TestOpenAIEndpointErrorBody(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
_, _ = io.WriteString(w, `{"error":{"message":"model is on fire"}}`)
Expand Down
7 changes: 4 additions & 3 deletions go/spec.go
Original file line number Diff line number Diff line change
Expand Up @@ -229,9 +229,10 @@ type coreGemini struct {
}

type coreOpenAI struct {
APIKey string `json:"api_key"`
BaseURL string `json:"base_url"`
Model string `json:"model"`
APIKey string `json:"api_key"`
BaseURL string `json:"base_url"`
Model string `json:"model"`
ReasoningEffort *ReasoningEffort `json:"reasoning_effort"`
}

func optString(s string) *string {
Expand Down
11 changes: 11 additions & 0 deletions rust/conformance-gen/fixtures/schema/run_spec.schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,17 @@
},
"model": {
"type": "string"
},
"reasoning_effort": {
"enum": [
"none",
"minimal",
"low",
"medium",
"high",
"xhigh",
null
]
}
},
"type": [
Expand Down
18 changes: 14 additions & 4 deletions rust/conformance-gen/src/cases.rs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
use serde_json::{Value, json};
use spawnllm_core::wire::{
AppleConfig, AppleGuardrails, AppleSampling, AppleUseCase, ClaudeConfig, CodexConfig,
GeminiConfig, OpenAiEndpoint, RunSpec,
GeminiConfig, OpenAiEndpoint, ReasoningEffort, RunSpec,
};

pub struct Case {
Expand Down Expand Up @@ -131,7 +131,11 @@ fn gemini_cfg_case(name: &str, cfg: GeminiConfig) -> Case {
)
}

fn endpoint_case(name: &str, schema: Option<Value>) -> Case {
fn endpoint_case(
name: &str,
schema: Option<Value>,
reasoning_effort: Option<ReasoningEffort>,
) -> Case {
plan_case(
name,
"openai_endpoint",
Expand All @@ -141,6 +145,7 @@ fn endpoint_case(name: &str, schema: Option<Value>) -> Case {
api_key: "sk-test".to_owned(),
base_url: "http://local.test/v1".to_owned(),
model: "qwen3".to_owned(),
reasoning_effort,
}),
..spec("ping", "qwen3")
},
Expand Down Expand Up @@ -561,8 +566,13 @@ fn plan_cases() -> Vec<Case> {
..AppleConfig::default()
},
),
endpoint_case("openai-endpoint-plain", None),
endpoint_case("openai-endpoint-schema", Some(schema_value())),
endpoint_case("openai-endpoint-plain", None, None),
endpoint_case("openai-endpoint-schema", Some(schema_value()), None),
endpoint_case(
"openai-endpoint-reasoning-effort",
None,
Some(ReasoningEffort::None),
),
]
}

Expand Down
27 changes: 26 additions & 1 deletion rust/spawnllm-core/src/plan/openai.rs
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,9 @@ pub(super) fn plan(spec: &RunSpec) -> InvocationPlan {
json!([{"role": "user", "content": spec.prompt}]),
),
]);
if let Some(effort) = endpoint.reasoning_effort {
body.insert("reasoning_effort".to_string(), json!(effort));
}
if let Some(schema) = &spec.schema {
body.insert(
"response_format".to_string(),
Expand Down Expand Up @@ -49,7 +52,7 @@ mod tests {
use serde_json::{Value, json};

use super::*;
use crate::wire::OpenAiEndpoint;
use crate::wire::{OpenAiEndpoint, ReasoningEffort};

fn spec(schema: Option<Value>) -> RunSpec {
RunSpec {
Expand All @@ -69,6 +72,7 @@ mod tests {
api_key: "sk-test".to_string(),
base_url: "http://local.test/v1".to_string(),
model: "qwen3".to_string(),
reasoning_effort: None,
}),
}
}
Expand Down Expand Up @@ -124,4 +128,25 @@ mod tests {
})
);
}

#[test]
fn reasoning_effort_plan_matches_vector() {
let mut spec = spec(None);
spec.openai_endpoint.as_mut().unwrap().reasoning_effort = Some(ReasoningEffort::None);

assert_eq!(
serde_json::to_value(plan(&spec)).unwrap(),
json!({
"kind": "http",
"method": "POST",
"url": "http://local.test/v1/chat/completions",
"headers": {"Authorization": "Bearer sk-test"},
"body": {
"model": "qwen3",
"messages": [{"role": "user", "content": "ping"}],
"reasoning_effort": "none",
},
})
);
}
}
12 changes: 12 additions & 0 deletions rust/spawnllm-core/src/wire.rs
Original file line number Diff line number Diff line change
Expand Up @@ -96,11 +96,23 @@ pub struct GeminiConfig {
pub extensions: Option<Vec<String>>,
}

#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ReasoningEffort {
None,
Minimal,
Low,
Medium,
High,
Xhigh,
}

#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
pub struct OpenAiEndpoint {
pub api_key: String,
pub base_url: String,
pub model: String,
pub reasoning_effort: Option<ReasoningEffort>,
}

#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
Expand Down
Loading
Loading