From 3be0a69f5ae36f11b92ca4de0ea442ffd436d836 Mon Sep 17 00:00:00 2001 From: Jonathan Ellis Date: Sat, 12 Sep 2026 17:01:08 -0500 Subject: [PATCH 1/3] Enforce DeepSeek structured inference through Responses API --- AGENTS.md | 4 + Cargo.lock | 8 +- Cargo.toml | 8 +- crates/anvil-client-python/Cargo.toml | 4 +- crates/anvil-client/Cargo.toml | 2 +- crates/anvil-client/src/deepseek_client.rs | 328 +++++++++++++++++++++ crates/anvil-client/src/infer.rs | 23 +- crates/anvil-client/src/lib.rs | 1 + crates/anvil-client/src/llm_client.rs | 6 + crates/anvil-client/src/responses_api.rs | 7 + crates/anvil-minimizer/Cargo.toml | 2 +- docs/src/content/docs/python-client.md | 8 + python/brokk_anvil/__init__.py | 2 +- 13 files changed, 376 insertions(+), 27 deletions(-) create mode 100644 crates/anvil-client/src/deepseek_client.rs diff --git a/AGENTS.md b/AGENTS.md index 4206fcc..3523693 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -37,6 +37,10 @@ live in `anvil-client`; CLI and Python use `HostedClient`. Keep package and path dependency versions in lockstep. Python task cancellation must cancel native work, and client close must cancel outstanding calls. Build wheels and run installed-wheel tests before publishing. The existing `python/` package remains the CLI launcher. +DeepSeek structured inference uses the stateless Responses backend; agent chat +keeps its Chat Completions backend. Native schema-enforced inference must not +prepend dynamic schema text ahead of caller messages. Reject incomplete Responses +output even when its text parses as valid JSON. ## Release workflow diff --git a/Cargo.lock b/Cargo.lock index 0f1373e..36cf67d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -882,7 +882,7 @@ dependencies = [ [[package]] name = "brokk-anvil" -version = "0.28.3" +version = "0.28.4" dependencies = [ "agent-client-protocol", "anyhow", @@ -933,7 +933,7 @@ dependencies = [ [[package]] name = "brokk-anvil-client" -version = "0.28.3" +version = "0.28.4" dependencies = [ "anyhow", "aws-config", @@ -968,7 +968,7 @@ dependencies = [ [[package]] name = "brokk-anvil-client-python" -version = "0.28.3" +version = "0.28.4" dependencies = [ "brokk-anvil-client", "pyo3", @@ -981,7 +981,7 @@ dependencies = [ [[package]] name = "brokk-anvil-minimizer" -version = "0.28.3" +version = "0.28.4" dependencies = [ "brush-parser", "parking_lot", diff --git a/Cargo.toml b/Cargo.toml index 432dc50..c0622e6 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,7 +7,7 @@ default-members = ["."] [package] name = "brokk-anvil" -version = "0.28.3" +version = "0.28.4" edition = "2024" description = "Anvil: Rust ACP server with first-run setup for Codex, Ollama, and OpenRouter" license = "LGPL-3.0-only" @@ -36,8 +36,8 @@ axum = { version = "0.8", optional = true } # the published crates are namespaced. Both are versioned in lockstep with this # crate: bump the requirement, the sub-crate's version, and the root version # together on every release (CI enforces the match). -anvil-minimizer = { package = "brokk-anvil-minimizer", version = "0.28.3", path = "crates/anvil-minimizer" } -anvil-client = { package = "brokk-anvil-client", version = "0.28.3", path = "crates/anvil-client", features = ["bedrock-credits"] } +anvil-minimizer = { package = "brokk-anvil-minimizer", version = "0.28.4", path = "crates/anvil-minimizer" } +anvil-client = { package = "brokk-anvil-client", version = "0.28.4", path = "crates/anvil-client", features = ["bedrock-credits"] } anyhow = "1" clap = { version = "4", features = ["derive", "env"] } # Pure-parsing library, also built as a wasm32-wasip2 binary that @@ -114,7 +114,7 @@ tempfile = "3" # Enables anvil-client's `test-support` feature so Anvil's tests can use # `openrouter_auth::test_support` (env-var guard/scope helpers). Feature # unification means only test builds see it; published binaries never do. -anvil-client = { package = "brokk-anvil-client", version = "0.28.3", path = "crates/anvil-client", features = ["bedrock-credits", "test-support"] } +anvil-client = { package = "brokk-anvil-client", version = "0.28.4", path = "crates/anvil-client", features = ["bedrock-credits", "test-support"] } [build-dependencies] serde = { version = "1", features = ["derive"] } diff --git a/crates/anvil-client-python/Cargo.toml b/crates/anvil-client-python/Cargo.toml index cbd23e2..6fb4104 100644 --- a/crates/anvil-client-python/Cargo.toml +++ b/crates/anvil-client-python/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "brokk-anvil-client-python" -version = "0.28.3" +version = "0.28.4" edition = "2024" license = "LGPL-3.0-only" description = "Native asynchronous Python bindings for Anvil's LLM client" @@ -12,7 +12,7 @@ name = "_native" crate-type = ["cdylib"] [dependencies] -anvil-client = { package = "brokk-anvil-client", path = "../anvil-client", version = "0.28.3" } +anvil-client = { package = "brokk-anvil-client", path = "../anvil-client", version = "0.28.4" } pyo3 = { version = "0.29", features = ["abi3-py311"] } pyo3-async-runtimes = { version = "0.29", features = ["tokio-runtime"] } serde = { version = "1", features = ["derive"] } diff --git a/crates/anvil-client/Cargo.toml b/crates/anvil-client/Cargo.toml index 237f11d..cae71fa 100644 --- a/crates/anvil-client/Cargo.toml +++ b/crates/anvil-client/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "brokk-anvil-client" -version = "0.28.3" +version = "0.28.4" edition = "2024" description = "Standalone LLM client extracted from Anvil: OpenAI-compatible, Bedrock, Codex, Meta/Muse, Grok, Ollama, OpenRouter, DeepSeek, and Kimi backends with auth, discovery, usage accounting, and ChatGPT-subscription transcription" license = "LGPL-3.0-only" diff --git a/crates/anvil-client/src/deepseek_client.rs b/crates/anvil-client/src/deepseek_client.rs new file mode 100644 index 0000000..1ad8984 --- /dev/null +++ b/crates/anvil-client/src/deepseek_client.rs @@ -0,0 +1,328 @@ +//! DeepSeek's stateless Responses API for native structured inference. +//! +//! Agent chat continues to use the Chat Completions backend. This client is +//! deliberately routed only by HostedClient, where JSON Schema enforcement is +//! required and partial completions must never be accepted as valid output. + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{Context, Result, bail}; +use futures::{StreamExt, future::BoxFuture}; + +use crate::llm_client::{LlmBackend, LlmResponse, StreamChatRequest}; +use crate::responses_api::{build_responses_request, drive_responses_sse_stream}; + +pub struct DeepSeekClient { + http: reqwest::Client, + api_key: String, + base_url: String, +} + +impl DeepSeekClient { + /// Env credentials take precedence over the existing consolidated store. + pub fn load() -> Result>> { + let key = match std::env::var(crate::discovery::DEEPSEEK_API_KEY_ENV) + .ok() + .filter(|key| !key.trim().is_empty()) + { + Some(key) => Some(key), + None => crate::deepseek_auth::read()?.map(|auth| auth.api_key), + }; + key.filter(|key| !key.trim().is_empty()) + .map(|key| { + Self::new(crate::discovery::DEEPSEEK_BASE_URL, key) + .map(|client| Arc::new(client) as Arc) + }) + .transpose() + } + + /// Explicit endpoint construction also supports local wire-level tests. + pub fn new(base_url: impl Into, api_key: impl Into) -> Result { + Ok(Self { + http: reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .connect_timeout(Duration::from_secs(20)) + .build() + .context("building DeepSeek Responses client")?, + api_key: api_key.into().trim().to_string(), + base_url: base_url.into().trim_end_matches('/').to_string(), + }) + } + + async fn invoke(&self, request: StreamChatRequest) -> Result { + let StreamChatRequest { + model, + messages, + tools, + reasoning_effort, + structured_output, + on_token, + on_thought, + cancel, + idle_timeouts, + .. + } = request; + let effort = reasoning_effort.as_deref().map(|effort| { + match effort.trim().to_ascii_lowercase().as_str() { + "none" => "none", + "minimal" | "low" => "low", + "max" => "max", + _ => "high", + } + }); + let body = build_responses_request( + &model, + &messages, + tools.as_deref(), + effort, + structured_output.as_ref(), + false, + None, + ); + let response = crate::http_retry::send_with_retries( + "posting DeepSeek Responses request", + || { + self.http + .post(format!("{}/responses", self.base_url)) + .bearer_auth(&self.api_key) + .header("Accept", "text/event-stream") + .json(&body) + }, + Some(&cancel), + Some(idle_timeouts.first_progress), + ) + .await?; + let status = response.status(); + if !status.is_success() { + // Read only enough to classify retryable/provider errors. Never + // expose an error body that might echo credentials or prompt data. + let read = async { + let mut stream = response.bytes_stream(); + let mut bytes = Vec::new(); + while let Some(Ok(chunk)) = stream.next().await { + let remaining = 64 * 1024 - bytes.len(); + bytes.extend_from_slice(&chunk[..chunk.len().min(remaining)]); + if bytes.len() == 64 * 1024 { + break; + } + } + String::from_utf8_lossy(&bytes).into_owned() + }; + let body = tokio::select! { + _ = cancel.cancelled() => bail!("DeepSeek Responses request cancelled"), + body = tokio::time::timeout(Duration::from_secs(3), read) => body.unwrap_or_default(), + }; + return Err(crate::http_retry::retryable_llm_error_for_status_and_body( + format!("DeepSeek Responses API failed (HTTP {status})"), + status, + &body, + )); + } + let stream = response + .bytes_stream() + .map(|chunk| chunk.map(|b| b.to_vec()).map_err(anyhow::Error::from)); + let outcome = + drive_responses_sse_stream(stream, on_token, on_thought, cancel.clone(), idle_timeouts) + .await?; + if cancel.is_cancelled() { + bail!("DeepSeek Responses request cancelled"); + } + if outcome.incomplete { + bail!("DeepSeek Responses output was incomplete; structured output cannot be accepted"); + } + Ok(outcome.response) + } +} + +impl LlmBackend for DeepSeekClient { + fn enforces_structured_output(&self) -> bool { + true + } + + fn list_models(&self) -> BoxFuture<'_, Result>> { + Box::pin(async move { + crate::llm_client::OpenAiClient::new(self.base_url.clone(), Some(self.api_key.clone())) + .list_models() + .await + }) + } + + fn stream_chat(&self, request: StreamChatRequest) -> BoxFuture<'_, Result> { + Box::pin(self.invoke(request)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::infer::{InferMessage, InferOptions, StructuredInferRequest, infer_structured}; + use crate::llm_client::{ChatMessage, IdleTimeouts}; + use crate::structured_output::StructuredOutputRequest; + use serde_json::json; + use tokio_util::sync::CancellationToken; + use wiremock::matchers::{header, method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + fn schema() -> serde_json::Value { + json!({"type":"object", "properties":{"slot0":{"type":"boolean"}}, + "required":["slot0"], "additionalProperties":false}) + } + + fn request(cancel: CancellationToken) -> StreamChatRequest { + StreamChatRequest { + model: "deepseek-v4-flash".to_string(), + messages: vec![ + ChatMessage::system("stable rules"), + ChatMessage::user("stable articles then candidates"), + ], + tools: None, + reasoning_effort: Some("max".to_string()), + service_tier: None, + temperature: None, + structured_output: Some(StructuredOutputRequest { + schema_name: "coverage".to_string(), + schema: schema(), + allow_coercion: false, + prefer_json_object: false, + }), + on_token: Box::new(|_| {}), + on_thought: Box::new(|_| {}), + cancel, + idle_timeouts: IdleTimeouts::uniform(Duration::from_secs(2)), + } + } + + fn completed() -> String { + format!( + "data: {}\n\ndata: {}\n\n", + json!({"type":"response.output_text.delta","delta":"{\"slot0\":true}"}), + json!({"type":"response.completed","response":{"id":"resp_test","usage":{"input_tokens":100,"output_tokens":15, + "input_tokens_details":{"cached_tokens":80},"output_tokens_details":{"reasoning_tokens":10}}}}) + ) + } + + #[tokio::test] + async fn structured_wire_preserves_prefix_and_enforces_native_schema() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/responses")) + .and(header("Authorization", "Bearer test-key")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("content-type", "text/event-stream") + .set_body_string(completed()), + ) + .mount(&server) + .await; + let client = DeepSeekClient::new(server.uri(), "test-key").unwrap(); + let result = infer_structured( + &client, + "deepseek-v4-flash", + StructuredInferRequest { + messages: vec![ + InferMessage::system("stable rules"), + InferMessage::user("stable articles then candidates"), + ], + schema_name: "coverage".to_string(), + schema: schema(), + }, + InferOptions { + reasoning_effort: Some("low".to_string()), + ..InferOptions::default() + }, + CancellationToken::new(), + ) + .await + .unwrap(); + assert_eq!(result.output, json!({"slot0":true})); + assert_eq!(result.usage.input_tokens, 20); + assert_eq!(result.usage.cached_read_tokens, 80); + assert_eq!(result.usage.thought_tokens, 10); + assert_eq!(result.usage.output_tokens, 5); + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = serde_json::from_slice(&requests[0].body).unwrap(); + assert_eq!(body["instructions"], "stable rules"); + assert_eq!(body["input"].as_array().unwrap().len(), 1); + assert_eq!( + body["input"][0]["content"][0]["text"], + "stable articles then candidates" + ); + assert_eq!(body["text"]["format"]["type"], "json_schema"); + assert_eq!(body["text"]["format"]["schema"], schema()); + assert_eq!(body["reasoning"]["effort"], "low"); + assert_eq!(body["store"], false); + assert!(body.get("previous_response_id").is_none()); + } + + #[tokio::test] + async fn incomplete_response_rejected_even_when_partial_text_is_valid_json() { + let server = MockServer::start().await; + let body = format!( + "data: {}\n\ndata: {}\n\n", + json!({"type":"response.output_text.delta","delta":"{\"slot0\":true}"}), + json!({"type":"response.incomplete","response":{"incomplete_details":{"reason":"max_output_tokens"}}}) + ); + Mock::given(path("/responses")) + .respond_with(ResponseTemplate::new(200).set_body_string(body)) + .mount(&server) + .await; + let client = DeepSeekClient::new(server.uri(), "test").unwrap(); + let error = client + .stream_chat(request(CancellationToken::new())) + .await + .unwrap_err(); + assert!(error.to_string().contains("incomplete")); + } + + #[tokio::test] + async fn cancellation_interrupts_pending_response() { + let server = MockServer::start().await; + Mock::given(path("/responses")) + .respond_with( + ResponseTemplate::new(200) + .set_delay(Duration::from_secs(10)) + .set_body_string(completed()), + ) + .mount(&server) + .await; + let client = DeepSeekClient::new(server.uri(), "test").unwrap(); + let cancel = CancellationToken::new(); + let task = client.stream_chat(request(cancel.clone())); + let trigger = async { + tokio::time::sleep(Duration::from_millis(30)).await; + cancel.cancel(); + }; + let (result, ()) = tokio::time::timeout(Duration::from_secs(1), async { + tokio::join!(task, trigger) + }) + .await + .unwrap(); + assert!(result.is_err()); + } + + #[tokio::test] + async fn reasoning_modes_follow_responses_dialect() { + let server = MockServer::start().await; + Mock::given(path("/responses")) + .respond_with(ResponseTemplate::new(200).set_body_string(completed())) + .mount(&server) + .await; + let client = DeepSeekClient::new(server.uri(), "test").unwrap(); + for (requested, expected) in [ + ("none", "none"), + ("minimal", "low"), + ("medium", "high"), + ("xhigh", "high"), + ("max", "max"), + ] { + let mut req = request(CancellationToken::new()); + req.reasoning_effort = Some(requested.to_string()); + client.stream_chat(req).await.unwrap(); + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = + serde_json::from_slice(&requests.last().unwrap().body).unwrap(); + assert_eq!(body["reasoning"]["effort"], expected); + } + } +} diff --git a/crates/anvil-client/src/infer.rs b/crates/anvil-client/src/infer.rs index 9b16519..a818b0b 100644 --- a/crates/anvil-client/src/infer.rs +++ b/crates/anvil-client/src/infer.rs @@ -181,7 +181,8 @@ impl HostedClient { "codex" => Some(Arc::new(crate::codex_client::CodexClient::new())), "meta" => crate::meta_client::MetaClient::load() .map_err(|e| InferError::new(InferErrorKind::Authentication, e))?, - "deepseek" => crate::hosted::build_deepseek_backend(), + "deepseek" => crate::deepseek_client::DeepSeekClient::load() + .map_err(|e| InferError::new(InferErrorKind::Authentication, e))?, "kimi" => crate::hosted::build_kimi_backend(), "grok" => crate::hosted::build_grok_backend(), _ => { @@ -274,16 +275,14 @@ pub async fn infer_structured( prefer_json_object: false, }; let mut messages = messages; - // Some providers downgrade json_schema to json_object. They require the - // word JSON in the prompt and otherwise never receive the actual schema. - // Supply it in-band as well; local validation remains authoritative. - messages.insert( - 0, - ChatMessage::system(format!( + // JSON-object fallbacks need an in-band schema, but dynamic schemas must + // follow the caller's stable prompt/article prefix rather than displace it. + if !backend.enforces_structured_output() { + messages.push(ChatMessage::user(format!( "Return only JSON matching this JSON Schema: {}", structured_output.schema, - )), - ); + ))); + } let mut total_usage = TokenUsage::default(); let mut validation_attempt = 0; let output = loop { @@ -497,11 +496,7 @@ mod tests { observed.lock().unwrap().as_slice(), [ObservedRequest { model: "utility-model".to_string(), - roles: vec![ - "system".to_string(), - "system".to_string(), - "user".to_string() - ], + roles: vec!["system".to_string(), "user".to_string(), "user".to_string()], tools_are_none: true, has_structured_output: true, reasoning_effort: Some("low".to_string()), diff --git a/crates/anvil-client/src/lib.rs b/crates/anvil-client/src/lib.rs index d1521c8..a728ff7 100644 --- a/crates/anvil-client/src/lib.rs +++ b/crates/anvil-client/src/lib.rs @@ -38,6 +38,7 @@ pub mod codex_client; pub mod codex_credits; pub mod deepseek_auth; pub mod deepseek_balance; +pub mod deepseek_client; pub mod discovery; pub mod grok_auth; pub mod grok_client; diff --git a/crates/anvil-client/src/llm_client.rs b/crates/anvil-client/src/llm_client.rs index e5c8263..8479d0f 100644 --- a/crates/anvil-client/src/llm_client.rs +++ b/crates/anvil-client/src/llm_client.rs @@ -987,6 +987,12 @@ impl ModelMetadata { // --------------------------------------------------------------------------- pub trait LlmBackend: Send + Sync { + /// True when structured inference sends the schema through an enforced + /// native output format and does not need an in-band schema instruction. + fn enforces_structured_output(&self) -> bool { + false + } + fn list_models(&self) -> BoxFuture<'_, Result>>; fn resolve_model_info(&self, configured_model: &str) -> ResolvedModelInfo { diff --git a/crates/anvil-client/src/responses_api.rs b/crates/anvil-client/src/responses_api.rs index 682672e..c2405b0 100644 --- a/crates/anvil-client/src/responses_api.rs +++ b/crates/anvil-client/src/responses_api.rs @@ -345,6 +345,8 @@ enum OutputItemContent { pub(crate) struct ResponsesStreamOutcome { pub(crate) response: LlmResponse, pub(crate) response_id: Option, + /// A token-limited partial response, which structured callers must reject. + pub(crate) incomplete: bool, } pub(crate) async fn drive_responses_sse_stream( @@ -363,6 +365,7 @@ where let mut deadline = tokio::time::Instant::now() + idle.first_progress; let mut saw_progress = false; let mut completed = false; + let mut incomplete = false; let mut failure: Option = None; let mut usage = TokenUsage::default(); let mut deltas_received = false; @@ -521,6 +524,7 @@ where break; } "response.incomplete" => { + incomplete = true; if let Some(final_body) = event.response { if let Some(u) = final_body.usage { usage = u.into_usage(); @@ -591,6 +595,7 @@ where codex_reasoning: None, }, response_id, + incomplete, }); } if !completed { @@ -608,6 +613,7 @@ where codex_reasoning: None, }, response_id, + incomplete, }) } else { Ok(ResponsesStreamOutcome { @@ -619,6 +625,7 @@ where codex_reasoning: None, }, response_id, + incomplete, }) } } diff --git a/crates/anvil-minimizer/Cargo.toml b/crates/anvil-minimizer/Cargo.toml index 891be17..edd9cdd 100644 --- a/crates/anvil-minimizer/Cargo.toml +++ b/crates/anvil-minimizer/Cargo.toml @@ -2,7 +2,7 @@ name = "brokk-anvil-minimizer" # Versioned in lockstep with the root brokk-anvil crate (CI enforces the # match), so the published minimizer always corresponds to the release tag. -version = "0.28.3" +version = "0.28.4" edition = "2024" license = "MIT" repository = "https://github.com/BrokkAi/anvil" diff --git a/docs/src/content/docs/python-client.md b/docs/src/content/docs/python-client.md index 29048b7..f689cc9 100644 --- a/docs/src/content/docs/python-client.md +++ b/docs/src/content/docs/python-client.md @@ -27,6 +27,14 @@ The client accepts the same explicit hosted-provider routes and credentials as Grok, and DeepSeek. It reuses native authentication, retries, schema validation, and usage accounting. No tools, agent sessions, or project instructions run. +DeepSeek structured inference uses its stateless Responses API with native +`text.format: json_schema` enforcement. Truncated responses are rejected even +when their partial text happens to be valid JSON. The schema is sent out of band, +so changing it does not prepend instructions ahead of your stable message prefix. +Other providers that require JSON-mode fallback receive the schema after your +messages; local validation still checks every provider's output. DeepSeek's normal +agent chat continues to use Chat Completions. + Keep one client for repeated calls to reuse provider connections. Use it as an async context manager or call `close()`; closing cancels outstanding requests. Asyncio task cancellation cancels native inference. Credentials are resolved diff --git a/python/brokk_anvil/__init__.py b/python/brokk_anvil/__init__.py index 1b11753..7144657 100644 --- a/python/brokk_anvil/__init__.py +++ b/python/brokk_anvil/__init__.py @@ -1,3 +1,3 @@ """Python distribution metadata for the native Anvil launcher.""" -__version__ = "0.28.3" +__version__ = "0.28.4" From 9ec9edf660f33172c3a120c05999cb8f188e22ff Mon Sep 17 00:00:00 2001 From: Jonathan Ellis Date: Sat, 12 Sep 2026 17:02:37 -0500 Subject: [PATCH 2/3] Refresh release license notices for version 0.28.4 --- licenses/THIRD_PARTY_LICENSES.html | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/licenses/THIRD_PARTY_LICENSES.html b/licenses/THIRD_PARTY_LICENSES.html index 2e7737a..68a7961 100644 --- a/licenses/THIRD_PARTY_LICENSES.html +++ b/licenses/THIRD_PARTY_LICENSES.html @@ -8782,9 +8782,9 @@

Used by:

GNU Lesser General Public License v3.0 only

Used by:

@@ -10124,7 +10124,7 @@

Used by:

MIT License

Used by:

MIT License
 

From 84c354f6e044421db6e0429477fdf6b7d5e1b88a Mon Sep 17 00:00:00 2001
From: Jonathan Ellis 
Date: Sat, 12 Sep 2026 17:11:29 -0500
Subject: [PATCH 3/3] Keep local validation authoritative for native schema
 formats

---
 AGENTS.md                                     |  6 ++-
 .../anvil-client-python/tests/test_client.py  |  5 ++-
 crates/anvil-client/src/deepseek_client.rs    | 39 +++++++++++++++++--
 crates/anvil-client/src/infer.rs              |  2 +-
 crates/anvil-client/src/llm_client.rs         |  7 ++--
 docs/src/content/docs/python-client.md        |  7 +++-
 6 files changed, 53 insertions(+), 13 deletions(-)

diff --git a/AGENTS.md b/AGENTS.md
index 3523693..f88cde0 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -38,9 +38,11 @@ dependency versions in lockstep. Python task cancellation must cancel native wor
 and client close must cancel outstanding calls. Build wheels and run installed-wheel
 tests before publishing. The existing `python/` package remains the CLI launcher.
 DeepSeek structured inference uses the stateless Responses backend; agent chat
-keeps its Chat Completions backend. Native schema-enforced inference must not
+keeps its Chat Completions backend. Native schema requests must not
 prepend dynamic schema text ahead of caller messages. Reject incomplete Responses
-output even when its text parses as valid JSON.
+output even when its text parses as valid JSON. DeepSeek Responses and beta
+strict tool calls have returned schema violations in live probes: native format
+support is not an enforcement guarantee, and local validation must remain.
 
 ## Release workflow
 
diff --git a/crates/anvil-client-python/tests/test_client.py b/crates/anvil-client-python/tests/test_client.py
index adb7a52..afd7ce8 100644
--- a/crates/anvil-client-python/tests/test_client.py
+++ b/crates/anvil-client-python/tests/test_client.py
@@ -21,7 +21,10 @@ def log_message(self, *args):
     def do_POST(self):
         request = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
         self.server.requests.append(request)
-        prompt = request["messages"][-1]["content"]
+        # The native client may append an in-band schema for JSON-mode
+        # providers. Dispatch mock behavior from the original caller input.
+        prompt = next(message["content"] for message in request["messages"]
+                      if message["role"] == "user")
         if "slow" in prompt:
             time.sleep(0.5)
         output = {"wrong": True} if "invalid" in prompt else {"ok": True}
diff --git a/crates/anvil-client/src/deepseek_client.rs b/crates/anvil-client/src/deepseek_client.rs
index 1ad8984..99cec40 100644
--- a/crates/anvil-client/src/deepseek_client.rs
+++ b/crates/anvil-client/src/deepseek_client.rs
@@ -1,8 +1,10 @@
 //! DeepSeek's stateless Responses API for native structured inference.
 //!
 //! Agent chat continues to use the Chat Completions backend. This client is
-//! deliberately routed only by HostedClient, where JSON Schema enforcement is
-//! required and partial completions must never be accepted as valid output.
+//! deliberately routed only by HostedClient, where a native JSON Schema request is
+//! useful and partial completions must never be accepted as valid output.
+//! DeepSeek may violate its requested schema even with strict=true, so the
+//! shared inference layer must still validate every result locally.
 
 use std::sync::Arc;
 use std::time::Duration;
@@ -136,7 +138,7 @@ impl DeepSeekClient {
 }
 
 impl LlmBackend for DeepSeekClient {
-    fn enforces_structured_output(&self) -> bool {
+    fn supports_native_structured_output(&self) -> bool {
         true
     }
 
@@ -203,7 +205,7 @@ mod tests {
     }
 
     #[tokio::test]
-    async fn structured_wire_preserves_prefix_and_enforces_native_schema() {
+    async fn structured_wire_preserves_prefix_and_requests_native_schema() {
         let server = MockServer::start().await;
         Mock::given(method("POST"))
             .and(path("/responses"))
@@ -255,6 +257,35 @@ mod tests {
         assert!(body.get("previous_response_id").is_none());
     }
 
+    #[tokio::test]
+    async fn native_format_still_rejects_provider_schema_violations_locally() {
+        let server = MockServer::start().await;
+        let body = completed().replace(r#"\"slot0\":true"#, r#"\"slot0\":true,\"no_match\":true"#);
+        Mock::given(path("/responses"))
+            .respond_with(ResponseTemplate::new(200).set_body_string(body))
+            .mount(&server)
+            .await;
+        let client = DeepSeekClient::new(server.uri(), "test").unwrap();
+        let error = infer_structured(
+            &client,
+            "deepseek-v4-flash",
+            StructuredInferRequest {
+                messages: vec![InferMessage::user("Classify this")],
+                schema_name: "coverage".to_string(),
+                schema: schema(),
+            },
+            InferOptions {
+                validation_retries: 0,
+                ..InferOptions::default()
+            },
+            CancellationToken::new(),
+        )
+        .await
+        .unwrap_err();
+        assert_eq!(error.kind(), crate::infer::InferErrorKind::StructuredOutput);
+        assert!(error.to_string().contains("no_match"));
+    }
+
     #[tokio::test]
     async fn incomplete_response_rejected_even_when_partial_text_is_valid_json() {
         let server = MockServer::start().await;
diff --git a/crates/anvil-client/src/infer.rs b/crates/anvil-client/src/infer.rs
index a818b0b..54c9993 100644
--- a/crates/anvil-client/src/infer.rs
+++ b/crates/anvil-client/src/infer.rs
@@ -277,7 +277,7 @@ pub async fn infer_structured(
     let mut messages = messages;
     // JSON-object fallbacks need an in-band schema, but dynamic schemas must
     // follow the caller's stable prompt/article prefix rather than displace it.
-    if !backend.enforces_structured_output() {
+    if !backend.supports_native_structured_output() {
         messages.push(ChatMessage::user(format!(
             "Return only JSON matching this JSON Schema: {}",
             structured_output.schema,
diff --git a/crates/anvil-client/src/llm_client.rs b/crates/anvil-client/src/llm_client.rs
index 8479d0f..49a837b 100644
--- a/crates/anvil-client/src/llm_client.rs
+++ b/crates/anvil-client/src/llm_client.rs
@@ -987,9 +987,10 @@ impl ModelMetadata {
 // ---------------------------------------------------------------------------
 
 pub trait LlmBackend: Send + Sync {
-    /// True when structured inference sends the schema through an enforced
-    /// native output format and does not need an in-band schema instruction.
-    fn enforces_structured_output(&self) -> bool {
+    /// True when structured inference sends the schema through a native output
+    /// format and does not need an in-band schema instruction. This is not a
+    /// guarantee of provider enforcement: local validation remains mandatory.
+    fn supports_native_structured_output(&self) -> bool {
         false
     }
 
diff --git a/docs/src/content/docs/python-client.md b/docs/src/content/docs/python-client.md
index f689cc9..1c746cc 100644
--- a/docs/src/content/docs/python-client.md
+++ b/docs/src/content/docs/python-client.md
@@ -28,11 +28,14 @@ Grok, and DeepSeek. It reuses native authentication, retries, schema validation,
 and usage accounting. No tools, agent sessions, or project instructions run.
 
 DeepSeek structured inference uses its stateless Responses API with native
-`text.format: json_schema` enforcement. Truncated responses are rejected even
+`text.format: json_schema` requests. DeepSeek can still return schema violations
+even with `strict: true`; the client validates locally and uses bounded repair
+retries before returning an error. Its beta strict tool-call mode also cannot
+be relied on to enforce nested schemas. Truncated responses are rejected even
 when their partial text happens to be valid JSON. The schema is sent out of band,
 so changing it does not prepend instructions ahead of your stable message prefix.
 Other providers that require JSON-mode fallback receive the schema after your
-messages; local validation still checks every provider's output. DeepSeek's normal
+messages; local validation checks every provider's output. DeepSeek's normal
 agent chat continues to use Chat Completions.
 
 Keep one client for repeated calls to reuse provider connections. Use it as an