diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c4824a4..433b2cb 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -162,7 +162,11 @@ jobs: targets: aarch64-linux-android,wasm32-wasip2 - name: Install Android SDK + NDK - uses: android-actions/setup-android@v3 + uses: android-actions/setup-android@v4 + with: + # The legacy `tools` SDK package was removed from Google's repository. + # This job only needs sdkmanager (installed by the action) plus the NDK. + packages: '' - name: Install Android NDK shell: bash diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index af5e602..e69a4d9 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -130,7 +130,11 @@ jobs: targets: aarch64-linux-android,wasm32-wasip2 - name: Install Android SDK + NDK - uses: android-actions/setup-android@v3 + uses: android-actions/setup-android@v4 + with: + # The legacy `tools` SDK package was removed from Google's repository. + # This job only needs sdkmanager (installed by the action) plus the NDK. + packages: '' - name: Install Android NDK shell: bash diff --git a/crates/anvil-client/src/bedrock_client.rs b/crates/anvil-client/src/bedrock_client.rs index c1960ba..6776824 100644 --- a/crates/anvil-client/src/bedrock_client.rs +++ b/crates/anvil-client/src/bedrock_client.rs @@ -9,7 +9,9 @@ use crate::llm_client::{ LlmResponse, ModelMetadata, ModelsResponse, OpenAiClient, OutputBudgetExhaustedError, ReasoningLevelPreset, StreamChatRequest, TokenSink, TokenUsage, ToolCall, ToolDefinition, }; -use crate::responses_api::{build_responses_request, drive_responses_sse_stream}; +use crate::responses_api::{ + ResponsesRequestOptions, build_responses_request, drive_responses_sse_stream, +}; use crate::responses_chain::{ RESPONSES_CHAIN_CACHE_CAP, ResponsesChainCache, find_responses_continuation, hash_responses_context, looks_like_expired_previous_response_id, @@ -1023,8 +1025,10 @@ impl BedrockClient { tools.as_deref(), reasoning_effort.as_deref(), structured_output.as_ref(), - true, - None, + ResponsesRequestOptions { + store: true, + ..Default::default() + }, ) }; @@ -1037,8 +1041,11 @@ impl BedrockClient { tools.as_deref(), reasoning_effort.as_deref(), structured_output.as_ref(), - true, - Some(previous_response_id.as_str()), + ResponsesRequestOptions { + store: true, + previous_response_id: Some(previous_response_id.as_str()), + ..Default::default() + }, ); (body, Some(hash_responses_context(&messages[..*boundary]))) } @@ -2983,8 +2990,10 @@ mod tests { None, Some("medium"), None, - true, - None, + ResponsesRequestOptions { + store: true, + ..Default::default() + }, ) }; let req_with = serde_json::to_value(build(&messages_with)).unwrap(); diff --git a/crates/anvil-client/src/deepseek_client.rs b/crates/anvil-client/src/deepseek_client.rs index 99cec40..2fd1198 100644 --- a/crates/anvil-client/src/deepseek_client.rs +++ b/crates/anvil-client/src/deepseek_client.rs @@ -13,7 +13,9 @@ use anyhow::{Context, Result, bail}; use futures::{StreamExt, future::BoxFuture}; use crate::llm_client::{LlmBackend, LlmResponse, StreamChatRequest}; -use crate::responses_api::{build_responses_request, drive_responses_sse_stream}; +use crate::responses_api::{ + ResponsesRequestOptions, build_responses_request, drive_responses_sse_stream, +}; pub struct DeepSeekClient { http: reqwest::Client, @@ -79,8 +81,7 @@ impl DeepSeekClient { tools.as_deref(), effort, structured_output.as_ref(), - false, - None, + ResponsesRequestOptions::default(), ); let response = crate::http_retry::send_with_retries( "posting DeepSeek Responses request", diff --git a/crates/anvil-client/src/discovery.rs b/crates/anvil-client/src/discovery.rs index 0f51bba..a6bac68 100644 --- a/crates/anvil-client/src/discovery.rs +++ b/crates/anvil-client/src/discovery.rs @@ -3,7 +3,8 @@ //! (`http://localhost:11434/v1/models`), a local ds4-server //! (antirez/ds4, an OpenAI-compatible DeepSeek V4 inference engine), and //! Kimi Code, Grok Build OAuth, generic OpenAI-compatible profiles from `providers.json`, -//! and OpenRouter (`https://openrouter.ai/api/v1/models`, gated on the +//! Xiaomi MiMo Token Plan and pay-as-you-go APIs, and OpenRouter +//! (`https://openrouter.ai/api/v1/models`, gated on the //! `OPENROUTER_API_KEY` env var), and hosted DeepSeek //! (`https://api.deepseek.com/v1/models`, gated on `DEEPSEEK_API_KEY`). //! @@ -16,6 +17,9 @@ //! the official Grok Build OAuth credential file. Generic //! OpenAI-compatible profiles are enabled only when `providers.json` //! configures them. +//! Xiaomi MiMo's plans are enabled by dedicated environment variables (or +//! prefix-disambiguated `MIMO_API_KEY`) and deliberately target separate +//! subscription and pay-as-you-go endpoints. //! //! ds4 is the one source whose port is *not* fixed: `ds4-server` has no //! standard port, so instead of probing a constant we detect a running @@ -29,8 +33,9 @@ //! (`MultiBackend`) can pick the right HTTP client at request time. The //! catalog is presented to ACP clients as `::` wire ids, e.g. //! `codex::gpt-5-codex`, `ollama::llama3:latest`, `deepseek::deepseek-v4-pro`, -//! `kimi::k3`, `openai::deca/model-id`, and -//! `openrouter::anthropic/claude-3.5-sonnet`. The double-colon separator +//! `kimi::k3`, `openai::deca/model-id`, `mimo::mimo-v2.6-pro`, +//! `mimo-payg::mimo-v2.6-pro`, and `openrouter::anthropic/claude-3.5-sonnet`. +//! The double-colon separator //! avoids collision with Ollama tags (`model:tag`) and with OpenRouter //! ids (`vendor/model`). //! @@ -60,6 +65,8 @@ impl ModelSource { pub const DS4: &'static str = "ds4"; pub const GROK: &'static str = "grok"; pub const KIMI: &'static str = "kimi"; + pub const MIMO: &'static str = "mimo"; + pub const MIMO_PAYG: &'static str = "mimo-payg"; pub const OLLAMA: &'static str = "ollama"; pub const OPENAI: &'static str = "openai"; pub const OPENROUTER: &'static str = "openrouter"; diff --git a/crates/anvil-client/src/grok_client.rs b/crates/anvil-client/src/grok_client.rs index 8ab1203..2fd3581 100644 --- a/crates/anvil-client/src/grok_client.rs +++ b/crates/anvil-client/src/grok_client.rs @@ -12,7 +12,9 @@ use crate::llm_client::{ LlmBackend, LlmResponse, ModelMetadata, ReasoningLevelPreset, ResolvedModelInfo, StreamChatRequest, }; -use crate::responses_api::{ReasoningConfig, build_responses_request, drive_responses_sse_stream}; +use crate::responses_api::{ + ReasoningConfig, ResponsesRequestOptions, build_responses_request, drive_responses_sse_stream, +}; const GROK_API_BASE_URL: &str = "https://cli-chat-proxy.grok.com/v1"; @@ -183,8 +185,7 @@ impl GrokClient { tools.as_deref(), reasoning_effort.as_deref(), structured_output.as_ref(), - false, - None, + ResponsesRequestOptions::default(), ); match body.reasoning.as_mut() { Some(reasoning) => reasoning.summary = Some("concise".to_string()), diff --git a/crates/anvil-client/src/hosted.rs b/crates/anvil-client/src/hosted.rs index 4810771..4c82bc8 100644 --- a/crates/anvil-client/src/hosted.rs +++ b/crates/anvil-client/src/hosted.rs @@ -1,7 +1,7 @@ //! Shared hosted-provider construction for Anvil and language bindings. use crate::llm_client::LlmBackend; -use crate::{deepseek_auth, discovery, grok_client, kimi_auth, llm_client}; +use crate::{deepseek_auth, discovery, grok_client, kimi_auth, llm_client, mimo_client}; use std::sync::Arc; /// Build a hosted DeepSeek chat backend from a raw API key. DeepSeek's API @@ -103,3 +103,30 @@ pub fn build_grok_backend() -> Option> { } } } + +/// Build the Xiaomi MiMo Token Plan backend. The generic `MIMO_API_KEY` is +/// accepted when its `tp-` (individual) or `ttp-` (team) prefix identifies +/// a Token Plan credential. +pub fn build_mimo_token_plan_backend() -> Option> { + match mimo_client::MimoClient::load(mimo_client::MimoPlan::TokenPlan) { + Ok(backend) => backend, + Err(error) => { + tracing::warn!("failed to configure Xiaomi MiMo Token Plan authentication: {error:#}"); + None + } + } +} + +/// Build Xiaomi MiMo's billed pay-as-you-go backend. The generic +/// `MIMO_API_KEY` is accepted only when its `sk-` prefix identifies that plan. +pub fn build_mimo_pay_as_you_go_backend() -> Option> { + match mimo_client::MimoClient::load(mimo_client::MimoPlan::PayAsYouGo) { + Ok(backend) => backend, + Err(error) => { + tracing::warn!( + "failed to configure Xiaomi MiMo pay-as-you-go authentication: {error:#}" + ); + None + } + } +} diff --git a/crates/anvil-client/src/infer.rs b/crates/anvil-client/src/infer.rs index 54c9993..8f0648f 100644 --- a/crates/anvil-client/src/infer.rs +++ b/crates/anvil-client/src/infer.rs @@ -183,13 +183,15 @@ impl HostedClient { .map_err(|e| InferError::new(InferErrorKind::Authentication, e))?, "deepseek" => crate::deepseek_client::DeepSeekClient::load() .map_err(|e| InferError::new(InferErrorKind::Authentication, e))?, + "mimo" => crate::hosted::build_mimo_token_plan_backend(), + "mimo-payg" => crate::hosted::build_mimo_pay_as_you_go_backend(), "kimi" => crate::hosted::build_kimi_backend(), "grok" => crate::hosted::build_grok_backend(), _ => { return Err(InferError::new( InferErrorKind::InvalidRequest, anyhow!( - "unsupported inference provider {source:?}; expected codex, meta, kimi, grok, or deepseek" + "unsupported inference provider {source:?}; expected codex, meta, kimi, grok, mimo, mimo-payg, or deepseek" ), )); } @@ -278,10 +280,9 @@ pub async fn infer_structured( // JSON-object fallbacks need an in-band schema, but dynamic schemas must // follow the caller's stable prompt/article prefix rather than displace it. if !backend.supports_native_structured_output() { - messages.push(ChatMessage::user(format!( - "Return only JSON matching this JSON Schema: {}", - structured_output.schema, - ))); + messages.push(ChatMessage::user( + crate::structured_output::json_schema_instruction(&structured_output), + )); } let mut total_usage = TokenUsage::default(); let mut validation_attempt = 0; diff --git a/crates/anvil-client/src/lib.rs b/crates/anvil-client/src/lib.rs index a728ff7..09b1bde 100644 --- a/crates/anvil-client/src/lib.rs +++ b/crates/anvil-client/src/lib.rs @@ -15,7 +15,8 @@ //! - [`infer`]: one-shot, tool-free, schema-constrained inference for callers //! that do not need an ACP session or Anvil's agent loop. //! - Provider backends: [`bedrock_client`], [`codex_client`], [`grok_client`], -//! and user-configured OpenAI-compatible endpoints via [`openai_providers`]. +//! Xiaomi MiMo Token Plan and pay-as-you-go APIs via [`mimo_client`], and +//! user-configured OpenAI-compatible endpoints via [`openai_providers`]. //! - Auth and credential storage: [`secrets`], [`codex_auth`], [`bedrock_auth`], //! [`grok_auth`], [`kimi_auth`], [`openrouter_auth`], [`deepseek_auth`]. //! - Credit/balance probes: [`bedrock_credits`], [`codex_credits`], @@ -48,6 +49,7 @@ pub mod infer; pub mod kimi_auth; pub mod llm_client; pub mod meta_client; +pub mod mimo_client; pub mod multi_backend; pub mod openai_providers; pub mod openrouter_auth; diff --git a/crates/anvil-client/src/meta_client.rs b/crates/anvil-client/src/meta_client.rs index 8839745..cee42ae 100644 --- a/crates/anvil-client/src/meta_client.rs +++ b/crates/anvil-client/src/meta_client.rs @@ -21,7 +21,9 @@ use tokio_util::sync::CancellationToken; use crate::llm_client::{ LlmBackend, LlmResponse, ModelMetadata, ResolvedModelInfo, StreamChatRequest, }; -use crate::responses_api::{build_responses_request, drive_responses_sse_stream}; +use crate::responses_api::{ + ResponsesRequestOptions, build_responses_request, drive_responses_sse_stream, +}; const META_API_BASE_URL: &str = "https://api.meta.ai/v1"; const META_MINT_BASE_URL: &str = "https://api.meta.ai"; @@ -298,8 +300,7 @@ impl MetaClient { tools.as_deref(), reasoning_effort.as_deref(), structured_output.as_ref(), - false, - None, + ResponsesRequestOptions::default(), ); let api_key = self.api_key_for(&cancel).await?; let response = self diff --git a/crates/anvil-client/src/mimo_client.rs b/crates/anvil-client/src/mimo_client.rs new file mode 100644 index 0000000..eefa001 --- /dev/null +++ b/crates/anvil-client/src/mimo_client.rs @@ -0,0 +1,837 @@ +//! Xiaomi MiMo's OpenAI-compatible Responses APIs. +//! +//! Token Plan and pay-as-you-go deliberately use different base URLs and key +//! prefixes. Keeping them as explicit plan routes prevents subscription traffic +//! from accidentally hitting the billed endpoint (or vice versa). + +use std::sync::Arc; +use std::time::Duration; + +use anyhow::{Context, Result, bail}; +use futures::{StreamExt, future::BoxFuture}; + +use crate::llm_client::{ChatMessage, LlmBackend, LlmResponse, ModelMetadata, StreamChatRequest}; +use crate::responses_api::{ + ResponsesRequestOptions, ResponsesTextConfig, ResponsesTextFormat, build_responses_request, + drive_responses_sse_stream, +}; + +pub const MIMO_TOKEN_PLAN_API_KEY_ENV: &str = "MIMO_TOKEN_PLAN_API_KEY"; +/// Xiaomi's integration examples use this name for both MiMo plans. Anvil +/// accepts it as a fallback, but the dedicated variable above makes plan +/// selection unambiguous when both credentials exist in the same shell. +pub const MIMO_API_KEY_ENV: &str = "MIMO_API_KEY"; +pub const MIMO_TOKEN_PLAN_BASE_URL_ENV: &str = "MIMO_TOKEN_PLAN_BASE_URL"; +pub const MIMO_TOKEN_PLAN_BASE_URL: &str = "https://token-plan-cn.xiaomimimo.com/v1"; +pub const MIMO_PAY_AS_YOU_GO_API_KEY_ENV: &str = "MIMO_PAY_AS_YOU_GO_API_KEY"; +pub const MIMO_PAY_AS_YOU_GO_BASE_URL_ENV: &str = "MIMO_PAY_AS_YOU_GO_BASE_URL"; +pub const MIMO_PAY_AS_YOU_GO_BASE_URL: &str = "https://api.xiaomimimo.com/v1"; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum MimoPlan { + TokenPlan, + PayAsYouGo, +} + +impl MimoPlan { + fn label(self) -> &'static str { + match self { + Self::TokenPlan => "Xiaomi MiMo Token Plan", + Self::PayAsYouGo => "Xiaomi MiMo pay-as-you-go", + } + } + + fn default_base_url(self) -> &'static str { + match self { + Self::TokenPlan => MIMO_TOKEN_PLAN_BASE_URL, + Self::PayAsYouGo => MIMO_PAY_AS_YOU_GO_BASE_URL, + } + } + + fn base_url_env(self) -> &'static str { + match self { + Self::TokenPlan => MIMO_TOKEN_PLAN_BASE_URL_ENV, + Self::PayAsYouGo => MIMO_PAY_AS_YOU_GO_BASE_URL_ENV, + } + } + + fn dedicated_api_key_env(self) -> &'static str { + match self { + Self::TokenPlan => MIMO_TOKEN_PLAN_API_KEY_ENV, + Self::PayAsYouGo => MIMO_PAY_AS_YOU_GO_API_KEY_ENV, + } + } + + fn generic_api_key_matches(self, key: &str) -> bool { + match self { + Self::TokenPlan => key.starts_with("tp-") || key.starts_with("ttp-"), + Self::PayAsYouGo => key.starts_with("sk-"), + } + } +} + +pub struct MimoClient { + plan: MimoPlan, + http: reqwest::Client, + api_key: String, + base_url: String, +} + +impl MimoClient { + /// Load a dedicated plan variable first. Xiaomi's generic `MIMO_API_KEY` + /// is accepted only when its documented key prefix identifies the plan. + pub fn load(plan: MimoPlan) -> Result>> { + let key = select_api_key(plan, |name| std::env::var(name).ok()); + let Some(key) = key else { + return Ok(None); + }; + + let base_url = match std::env::var(plan.base_url_env()) { + Ok(value) => { + let value = value.trim().to_string(); + if value.is_empty() { + bail!("{} is set but empty", plan.base_url_env()); + } + value + } + Err(_) => plan.default_base_url().to_string(), + }; + Ok(Some( + Arc::new(Self::new(plan, base_url, key)?) as Arc + )) + } + + /// Explicit endpoint construction also supports local wire-level tests. + pub fn new( + plan: MimoPlan, + base_url: impl Into, + api_key: impl Into, + ) -> Result { + Ok(Self { + plan, + http: reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .connect_timeout(Duration::from_secs(20)) + .build() + .context("building Xiaomi MiMo Responses client")?, + api_key: api_key.into().trim().to_string(), + base_url: base_url.into().trim_end_matches('/').to_string(), + }) + } + + fn responses_url(&self) -> String { + if self.base_url.ends_with("/v1") { + format!("{}/responses", self.base_url) + } else { + format!("{}/v1/responses", self.base_url) + } + } + + async fn invoke(&self, request: StreamChatRequest) -> Result { + let StreamChatRequest { + model, + mut messages, + tools, + reasoning_effort, + structured_output, + on_token, + on_thought, + cancel, + idle_timeouts, + .. + } = request; + // MiMo supports JSON-object mode, not native JSON Schema. Include + // the schema after the stable caller prefix, including direct ACP + // callers that do not go through infer_structured's fallback. + if let Some(output) = &structured_output { + let instruction = crate::structured_output::json_schema_instruction(output); + if !messages + .iter() + .any(|message| message.role == "user" && message.content_text() == instruction) + { + messages.push(ChatMessage::user(instruction)); + } + } + let effort = reasoning_effort.as_deref().map(mimo_reasoning_effort); + let mut body = build_responses_request( + &model, + &messages, + tools.as_deref(), + effort, + None, + ResponsesRequestOptions { + replay_reasoning: true, + ..Default::default() + }, + ); + body.text = structured_output.as_ref().map(|_| ResponsesTextConfig { + format: ResponsesTextFormat::JsonObject, + }); + // Xiaomi's model catalog explicitly marks parallel tool calls + // unsupported. Keep the wire request sequential even when the agent + // loop is willing to execute a returned batch concurrently. + body.parallel_tool_calls = false; + + let response = crate::http_retry::send_with_retries( + "posting Xiaomi MiMo Responses request", + || { + self.http + .post(self.responses_url()) + .bearer_auth(&self.api_key) + .header("Accept", "text/event-stream") + .json(&body) + }, + Some(&cancel), + Some(idle_timeouts.first_progress), + ) + .await?; + let status = response.status(); + if !status.is_success() { + let body_text = tokio::select! { + biased; + _ = cancel.cancelled() => bail!("{} request cancelled", self.plan.label()), + body = tokio::time::timeout( + Duration::from_secs(3).min(idle_timeouts.inter_chunk), + read_limited_error_body(response), + ) => body.unwrap_or_default(), + }; + return Err(crate::http_retry::retryable_llm_error_for_status_and_body( + format!("{} Responses API failed (HTTP {status})", self.plan.label()), + status, + &body_text, + )); + } + + let stream = response + .bytes_stream() + .map(|chunk| chunk.map(|b| b.to_vec()).map_err(anyhow::Error::from)); + let outcome = + drive_responses_sse_stream(stream, on_token, on_thought, cancel.clone(), idle_timeouts) + .await?; + if cancel.is_cancelled() { + bail!("{} request cancelled", self.plan.label()); + } + if outcome.incomplete { + bail!("{} output was incomplete", self.plan.label()); + } + Ok(outcome.response) + } +} + +fn select_api_key(plan: MimoPlan, lookup: impl Fn(&str) -> Option) -> Option { + if let Some(key) = lookup(plan.dedicated_api_key_env()) + .map(|key| key.trim().to_string()) + .filter(|key| !key.is_empty()) + { + return Some(key); + } + + let generic = lookup(MIMO_API_KEY_ENV) + .map(|key| key.trim().to_string()) + .filter(|key| !key.is_empty())?; + if plan.generic_api_key_matches(&generic) { + Some(generic) + } else { + tracing::info!( + "{} is set but its key prefix does not identify {}; backend skipped", + MIMO_API_KEY_ENV, + plan.label() + ); + None + } +} + +async fn read_limited_error_body(response: reqwest::Response) -> String { + let mut stream = response.bytes_stream(); + let mut bytes = Vec::new(); + while let Some(Ok(chunk)) = stream.next().await { + let remaining = 64 * 1024 - bytes.len(); + bytes.extend_from_slice(&chunk[..chunk.len().min(remaining)]); + if bytes.len() == 64 * 1024 { + break; + } + } + String::from_utf8_lossy(&bytes).into_owned() +} + +/// MiMo advertises `none`, `low`, `medium`, and `high`. Map the larger +/// Anvil effort vocabulary onto that set before the first request. +fn mimo_reasoning_effort(effort: &str) -> &'static str { + match effort.trim().to_ascii_lowercase().as_str() { + "none" => "none", + "minimal" | "low" => "low", + "medium" => "medium", + _ => "high", + } +} + +const MIMO_REASONING_PRESETS: &[(&str, &str)] = &[ + ("none", "No extra reasoning for faster responses"), + ("low", "Fast responses with lighter reasoning"), + ( + "medium", + "Balances speed and reasoning depth for everyday tasks", + ), + ("high", "Greater reasoning depth for complex problems"), +]; + +fn mimo_reasoning_presets() -> Vec { + MIMO_REASONING_PRESETS + .iter() + .map( + |(effort, description)| crate::llm_client::ReasoningLevelPreset { + effort: (*effort).to_string(), + description: (*description).to_string(), + }, + ) + .collect() +} + +/// Xiaomi's `/models` response is OpenAI-shaped and may not expose all of +/// the capabilities published in its Codex model catalog. Enrich known IDs +/// while preserving newly introduced models from the live endpoint. +fn enrich_mimo_metadata(mut metadata: ModelMetadata, plan: MimoPlan) -> ModelMetadata { + let known = matches!( + metadata.id.as_str(), + "mimo-v2.6-pro" + | "mimo-v2.6-flash" + | "mimo-v2.6-pro-ultraspeed" + | "mimo-v2.5-pro" + | "mimo-v2.5" + ); + if known { + metadata.default_reasoning_level = Some("low".to_string()); + metadata.supported_reasoning_levels = mimo_reasoning_presets(); + metadata.context_length = Some(1_048_576); + if plan == MimoPlan::TokenPlan { + metadata.pricing = None; + } + if metadata.id != "mimo-v2.5-pro" { + metadata.supports_images = Some(true); + } else { + metadata.supports_images = Some(false); + } + } + metadata +} + +impl LlmBackend for MimoClient { + fn list_models(&self) -> BoxFuture<'_, Result>> { + Box::pin(async move { + Ok(self + .list_model_metadata() + .await? + .into_iter() + .map(|model| model.id) + .collect()) + }) + } + + fn list_model_metadata(&self) -> BoxFuture<'_, Result>> { + Box::pin(async move { + let models = crate::llm_client::OpenAiClient::new( + self.base_url.clone(), + Some(self.api_key.clone()), + ) + .list_model_metadata() + .await?; + Ok(models + .into_iter() + .filter(|metadata| !is_audio_only_model(&metadata.id)) + .map(|metadata| enrich_mimo_metadata(metadata, self.plan)) + .collect()) + }) + } + + fn stream_chat(&self, request: StreamChatRequest) -> BoxFuture<'_, Result> { + Box::pin(self.invoke(request)) + } +} + +fn is_audio_only_model(id: &str) -> bool { + matches!( + id, + "mimo-v2.5-asr" + | "mimo-v2.5-tts" + | "mimo-v2.5-tts-voiceclone" + | "mimo-v2.5-tts-voicedesign" + ) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::infer::{ + InferErrorKind, InferMessage, InferOptions, StructuredInferRequest, infer_structured, + }; + use crate::llm_client::IdleTimeouts; + use crate::structured_output::StructuredOutputRequest; + use serde_json::json; + use tokio_util::sync::CancellationToken; + use wiremock::matchers::{header, method, path}; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + fn request(cancel: CancellationToken) -> StreamChatRequest { + StreamChatRequest { + model: "mimo-v2.6-pro".to_string(), + messages: vec![ + ChatMessage::system("stable rules"), + ChatMessage::user("summarize the issue"), + ], + tools: None, + reasoning_effort: Some("max".to_string()), + service_tier: None, + temperature: None, + structured_output: None, + on_token: Box::new(|_| {}), + on_thought: Box::new(|_| {}), + cancel, + idle_timeouts: IdleTimeouts::uniform(Duration::from_secs(2)), + } + } + + fn completed() -> String { + format!( + "data: {}\n\ndata: {}\n\n", + json!({"type":"response.output_text.delta","delta":"done"}), + json!({"type":"response.completed","response":{"id":"resp_test","usage":{"input_tokens":20,"output_tokens":5}}}) + ) + } + + #[test] + fn mimo_efforts_are_clamped_to_documented_levels() { + for (requested, expected) in [ + ("none", "none"), + ("minimal", "low"), + ("medium", "medium"), + ("xhigh", "high"), + ("max", "high"), + ] { + assert_eq!(mimo_reasoning_effort(requested), expected); + } + } + + #[test] + fn plan_urls_stay_separate() { + let token_plan = + MimoClient::new(MimoPlan::TokenPlan, MIMO_TOKEN_PLAN_BASE_URL, "tp-test").unwrap(); + assert_eq!( + token_plan.responses_url(), + "https://token-plan-cn.xiaomimimo.com/v1/responses" + ); + + let pay_as_you_go = + MimoClient::new(MimoPlan::PayAsYouGo, MIMO_PAY_AS_YOU_GO_BASE_URL, "sk-test").unwrap(); + assert_eq!( + pay_as_you_go.responses_url(), + "https://api.xiaomimimo.com/v1/responses" + ); + + let custom_origin = MimoClient::new( + MimoPlan::TokenPlan, + "https://token-plan.example.test", + "tp-test", + ) + .unwrap(); + assert_eq!( + custom_origin.responses_url(), + "https://token-plan.example.test/v1/responses" + ); + } + + #[test] + fn generic_credentials_select_only_their_documented_plan() { + let lookup = |name: &str| match name { + MIMO_TOKEN_PLAN_API_KEY_ENV => None, + MIMO_PAY_AS_YOU_GO_API_KEY_ENV => None, + MIMO_API_KEY_ENV => Some("tp-token-plan".to_string()), + _ => None, + }; + assert_eq!( + select_api_key(MimoPlan::TokenPlan, lookup), + Some("tp-token-plan".to_string()) + ); + assert_eq!(select_api_key(MimoPlan::PayAsYouGo, lookup), None); + + let lookup = |name: &str| match name { + MIMO_TOKEN_PLAN_API_KEY_ENV => None, + MIMO_PAY_AS_YOU_GO_API_KEY_ENV => None, + MIMO_API_KEY_ENV => Some("sk-payg".to_string()), + _ => None, + }; + assert_eq!(select_api_key(MimoPlan::TokenPlan, lookup), None); + assert_eq!( + select_api_key(MimoPlan::PayAsYouGo, lookup), + Some("sk-payg".to_string()) + ); + } + + #[test] + fn dedicated_credentials_take_precedence_over_generic_keys() { + let lookup = |name: &str| match name { + MIMO_TOKEN_PLAN_API_KEY_ENV => Some("tp-dedicated".to_string()), + MIMO_PAY_AS_YOU_GO_API_KEY_ENV => Some("sk-dedicated".to_string()), + MIMO_API_KEY_ENV => Some("sk-generic".to_string()), + _ => None, + }; + assert_eq!( + select_api_key(MimoPlan::TokenPlan, lookup), + Some("tp-dedicated".to_string()) + ); + assert_eq!( + select_api_key(MimoPlan::PayAsYouGo, lookup), + Some("sk-dedicated".to_string()) + ); + } + + #[test] + fn shared_team_credentials_select_token_plan_only() { + let lookup = |name: &str| (name == MIMO_API_KEY_ENV).then(|| " ttp-team-key ".into()); + assert_eq!( + select_api_key(MimoPlan::TokenPlan, lookup), + Some("ttp-team-key".into()) + ); + assert_eq!(select_api_key(MimoPlan::PayAsYouGo, lookup), None); + } + + fn output_schema() -> serde_json::Value { + json!({"type":"object", "properties":{"ok":{"type":"boolean"}}, + "required":["ok"], "additionalProperties":false}) + } + + #[tokio::test] + async fn structured_inference_uses_json_object_and_validates_locally() { + for (output, valid) in [(r#"{"ok":true}"#, true), (r#"{"ok":"wrong type"}"#, false)] { + let server = MockServer::start().await; + let body = format!( + "data: {}\n\ndata: {}\n\n", + json!({"type":"response.output_text.delta","delta":output}), + json!({"type":"response.completed","response":{"id":"resp_test"}}) + ); + Mock::given(path("/v1/responses")) + .respond_with(ResponseTemplate::new(200).set_body_string(body)) + .mount(&server) + .await; + let client = MimoClient::new(MimoPlan::PayAsYouGo, server.uri(), "sk-test").unwrap(); + let result = infer_structured( + &client, + "mimo-v2.6-pro", + StructuredInferRequest { + messages: vec![ + InferMessage::system("stable rules"), + InferMessage::user("stable input"), + ], + schema_name: "result".into(), + schema: output_schema(), + }, + InferOptions { + validation_retries: 0, + ..Default::default() + }, + CancellationToken::new(), + ) + .await; + if valid { + assert_eq!(result.unwrap().output, json!({"ok":true})); + } else { + assert_eq!(result.unwrap_err().kind(), InferErrorKind::StructuredOutput); + } + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = serde_json::from_slice(&requests[0].body).unwrap(); + assert_eq!(body["text"]["format"], json!({"type":"json_object"})); + assert_eq!(body["instructions"], "stable rules"); + let input = body["input"].as_array().unwrap(); + assert_eq!(input.len(), 2, "schema must be included exactly once"); + assert_eq!(input[0]["content"][0]["text"], "stable input"); + assert_eq!( + input[1]["content"][0]["text"], + format!( + "Return only JSON matching this JSON Schema: {}", + output_schema() + ) + ); + } + } + + #[tokio::test] + async fn direct_structured_chat_also_includes_schema_after_caller_input() { + let server = MockServer::start().await; + Mock::given(path("/v1/responses")) + .respond_with(ResponseTemplate::new(200).set_body_string(completed())) + .mount(&server) + .await; + let client = MimoClient::new(MimoPlan::TokenPlan, server.uri(), "tp-test").unwrap(); + let mut req = request(CancellationToken::new()); + req.structured_output = Some(StructuredOutputRequest { + schema_name: "result".into(), + schema: output_schema(), + allow_coercion: false, + prefer_json_object: false, + }); + client.stream_chat(req).await.unwrap(); + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = serde_json::from_slice(&requests[0].body).unwrap(); + assert_eq!(body["text"]["format"], json!({"type":"json_object"})); + assert_eq!(body["instructions"], "stable rules"); + assert_eq!( + body["input"][0]["content"][0]["text"], + "summarize the issue" + ); + assert_eq!( + body["input"][1]["content"][0]["text"], + format!( + "Return only JSON matching this JSON Schema: {}", + output_schema() + ) + ); + } + + #[tokio::test] + async fn error_body_obeys_cancellation_and_deadline() { + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + for cancel_request in [true, false] { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap(); + let endpoint = format!("http://{}", listener.local_addr().unwrap()); + let (headers_sent, headers_received) = tokio::sync::oneshot::channel(); + let server = tokio::spawn(async move { + let (mut socket, _) = listener.accept().await.unwrap(); + let mut buffer = [0; 4096]; + let mut request_headers = Vec::new(); + while !request_headers.windows(4).any(|part| part == b"\r\n\r\n") { + let received = socket.read(&mut buffer).await.unwrap(); + assert_ne!(received, 0, "request ended before its headers"); + request_headers.extend_from_slice(&buffer[..received]); + } + socket + .write_all(b"HTTP/1.1 400 Bad Request\r\nContent-Length: 100\r\n\r\n") + .await + .unwrap(); + headers_sent.send(()).unwrap(); + std::future::pending::<()>().await; + }); + let client = MimoClient::new(MimoPlan::TokenPlan, endpoint, "tp-test").unwrap(); + let cancel = CancellationToken::new(); + let mut req = request(cancel.clone()); + if !cancel_request { + req.idle_timeouts.inter_chunk = Duration::from_millis(100); + } + let mut call = tokio::spawn(async move { client.stream_chat(req).await }); + headers_received.await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + if cancel_request { + cancel.cancel(); + } + let result = tokio::time::timeout(Duration::from_secs(1), &mut call).await; + call.abort(); + server.abort(); + let error = result + .expect("stalled error body must terminate promptly") + .unwrap() + .unwrap_err(); + assert!( + error.to_string().contains(if cancel_request { + "cancelled" + } else { + "HTTP 400" + }), + "{error:#}" + ); + } + } + + #[tokio::test] + async fn tool_round_trip_replays_reasoning_without_duplication() { + for (send_delta, send_done) in [(true, true), (true, false), (false, true)] { + let server = MockServer::start().await; + let mut events = vec![]; + if send_delta { + events.push(json!({"type":"response.reasoning_text.delta","delta":"Inspect the file first"})); + } + if send_done { + events.push(json!({"type":"response.output_item.done","item":{"type":"reasoning","id":"r1","content":[{"type":"reasoning_text","text":"Inspect the file first"}]}})); + } + events.extend([ + json!({"type":"response.output_item.done","item":{"type":"function_call","call_id":"call1","name":"read_file","arguments":"{}"}}), + json!({"type":"response.completed","response":{"id":"resp1"}}), + ]); + let body = events + .iter() + .map(|event| format!("data: {event}\n\n")) + .collect::(); + Mock::given(path("/v1/responses")) + .respond_with(ResponseTemplate::new(200).set_body_string(body)) + .mount(&server) + .await; + let client = MimoClient::new(MimoPlan::TokenPlan, server.uri(), "tp-test").unwrap(); + let mut req = request(CancellationToken::new()); + let thoughts = Arc::new(std::sync::Mutex::new(String::new())); + let captured = thoughts.clone(); + req.on_thought = Box::new(move |text| captured.lock().unwrap().push_str(text)); + let response = client.stream_chat(req).await.unwrap(); + let LlmResponse::ToolCalls { + text, + reasoning_content, + calls, + .. + } = response + else { + panic!("expected tool response"); + }; + assert_eq!(reasoning_content.as_deref(), Some("Inspect the file first")); + assert_eq!(*thoughts.lock().unwrap(), "Inspect the file first"); + let mut next = request(CancellationToken::new()); + next.messages.push( + ChatMessage::assistant_tool_calls_with_content_and_reasoning( + text, + calls, + reasoning_content, + ), + ); + next.messages.push(ChatMessage::tool_result( + "call1", + "read_file", + "file contents", + )); + client.stream_chat(next).await.unwrap(); + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = serde_json::from_slice(&requests[1].body).unwrap(); + let input = body["input"].as_array().unwrap(); + assert_eq!( + input + .iter() + .map(|item| item["type"].as_str().unwrap()) + .collect::>(), + vec![ + "message", + "reasoning", + "function_call", + "function_call_output" + ] + ); + assert!(input[1]["id"].as_str().is_some_and(|id| !id.is_empty())); + assert_eq!( + input[1]["content"], + json!([{"type":"reasoning_text","text":"Inspect the file first"}]) + ); + assert_eq!(input[2]["call_id"], input[3]["call_id"]); + assert_eq!(input[3]["output"], "file contents"); + } + } + + #[tokio::test] + async fn both_discovery_methods_exclude_audio_and_preserve_unknown_models() { + let server = MockServer::start().await; + Mock::given(path("/v1/models")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({"data":[ + {"id":"mimo-v2.5-asr"}, {"id":"mimo-v2.5-tts"}, + {"id":"mimo-v2.5-tts-voiceclone"}, {"id":"mimo-v2.5-tts-voicedesign"}, + {"id":"mimo-v2.6-pro"}, {"id":"future-mimo"} + ]}))) + .mount(&server) + .await; + for plan in [MimoPlan::TokenPlan, MimoPlan::PayAsYouGo] { + let client = MimoClient::new(plan, server.uri(), "test-key").unwrap(); + let metadata_ids: Vec<_> = client + .list_model_metadata() + .await + .unwrap() + .into_iter() + .map(|model| model.id) + .collect(); + assert_eq!(metadata_ids, vec!["mimo-v2.6-pro", "future-mimo"]); + assert_eq!(client.list_models().await.unwrap(), metadata_ids); + } + } + + #[tokio::test] + async fn responses_request_targets_token_plan_and_disables_parallel_tools() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/v1/responses")) + .and(header("Authorization", "Bearer tp-test-key")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("content-type", "text/event-stream") + .set_body_string(completed()), + ) + .mount(&server) + .await; + + let client = MimoClient::new(MimoPlan::TokenPlan, server.uri(), "tp-test-key").unwrap(); + client + .stream_chat(request(CancellationToken::new())) + .await + .unwrap(); + + let requests = server.received_requests().await.unwrap(); + let body: serde_json::Value = serde_json::from_slice(&requests[0].body).unwrap(); + assert_eq!(body["parallel_tool_calls"], false); + assert_eq!(body["reasoning"]["effort"], "high"); + assert_eq!(body["store"], false); + } + + #[tokio::test] + async fn model_discovery_is_enriched_without_pay_as_you_go_pricing() { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/v1/models")) + .and(header("Authorization", "Bearer tp-test-key")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "object": "list", + "data": [ + {"id": "mimo-v2.6-pro", "object": "model", "pricing": {"prompt": "0.0000036", "completion": "0.00000087"}}, + {"id": "future-mimo", "object": "model"} + ] + }))) + .mount(&server) + .await; + + let client = MimoClient::new(MimoPlan::TokenPlan, server.uri(), "tp-test-key").unwrap(); + let models = client.list_model_metadata().await.unwrap(); + assert_eq!(models.len(), 2); + + let known = &models[0]; + assert_eq!(known.default_reasoning_level.as_deref(), Some("low")); + assert_eq!(known.context_length, Some(1_048_576)); + assert_eq!(known.supports_images, Some(true)); + assert!(known.pricing.is_none()); + assert_eq!( + known + .supported_reasoning_levels + .iter() + .map(|preset| preset.effort.as_str()) + .collect::>(), + vec!["none", "low", "medium", "high"] + ); + + let unknown = &models[1]; + assert_eq!(unknown.id, "future-mimo"); + assert!(unknown.default_reasoning_level.is_none()); + } + + #[tokio::test] + async fn pay_as_you_go_discovery_preserves_provider_pricing() { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/v1/models")) + .and(header("Authorization", "Bearer sk-test-key")) + .respond_with(ResponseTemplate::new(200).set_body_json(json!({ + "object": "list", + "data": [ + {"id": "mimo-v2.6-pro", "object": "model", "pricing": {"prompt": "0.0000036", "completion": "0.00000087"}} + ] + }))) + .mount(&server) + .await; + + let client = MimoClient::new(MimoPlan::PayAsYouGo, server.uri(), "sk-test-key").unwrap(); + let models = client.list_model_metadata().await.unwrap(); + assert_eq!( + models[0].pricing.map(|pricing| ( + pricing.input_cost_per_token_usd, + pricing.output_cost_per_token_usd + )), + Some((0.0000036, 0.00000087)) + ); + } +} diff --git a/crates/anvil-client/src/responses_api.rs b/crates/anvil-client/src/responses_api.rs index c2405b0..7405e7c 100644 --- a/crates/anvil-client/src/responses_api.rs +++ b/crates/anvil-client/src/responses_api.rs @@ -52,6 +52,7 @@ pub(crate) struct ResponsesTextConfig { #[derive(Debug, Serialize)] #[serde(tag = "type", rename_all = "snake_case")] pub(crate) enum ResponsesTextFormat { + JsonObject, JsonSchema { name: String, schema: serde_json::Value, @@ -62,6 +63,10 @@ pub(crate) enum ResponsesTextFormat { #[derive(Debug, Serialize)] #[serde(tag = "type", rename_all = "snake_case")] pub(crate) enum ResponsesInputItem { + Reasoning { + id: String, + content: Vec, + }, Message { role: String, content: Vec, @@ -80,6 +85,7 @@ pub(crate) enum ResponsesInputItem { #[derive(Debug, Serialize)] #[serde(tag = "type", rename_all = "snake_case")] pub(crate) enum ResponsesContent { + ReasoningText { text: String }, InputText { text: String }, InputImage { image_url: String }, OutputText { text: String }, @@ -93,6 +99,15 @@ pub(crate) struct ResponsesToolDef { pub(crate) parameters: serde_json::Value, } +#[derive(Default)] +pub(crate) struct ResponsesRequestOptions<'a> { + pub(crate) store: bool, + pub(crate) previous_response_id: Option<&'a str>, + /// MiMo accepts plain reasoning text when replaying stateless history. + /// Other providers may require encrypted items or server-side state. + pub(crate) replay_reasoning: bool, +} + /// Builds a Responses API request body from `messages`. /// /// `messages` is either the *full* conversation (fresh turn, no prior @@ -110,13 +125,12 @@ pub(crate) fn build_responses_request( tools: Option<&[ToolDefinition]>, reasoning_effort: Option<&str>, structured_output: Option<&StructuredOutputRequest>, - store: bool, - previous_response_id: Option<&str>, + options: ResponsesRequestOptions<'_>, ) -> ResponsesRequest { let mut instructions_parts: Vec = Vec::new(); let mut input: Vec = Vec::new(); - for msg in messages { + for (index, msg) in messages.iter().enumerate() { match msg.role.as_str() { "system" => { let text = msg.content_text(); @@ -142,6 +156,16 @@ pub(crate) fn build_responses_request( }); } "assistant" => { + if options.replay_reasoning + && let Some(text) = msg.reasoning_content.as_ref().filter(|s| !s.is_empty()) + { + input.push(ResponsesInputItem::Reasoning { + // Stateless replay needs unique, stable item IDs, not + // a reference to a previous server-side response. + id: format!("rs_anvil_{index}"), + content: vec![ResponsesContent::ReasoningText { text: text.clone() }], + }); + } if let Some(calls) = &msg.tool_calls { for call in calls { input.push(ResponsesInputItem::FunctionCall { @@ -203,8 +227,8 @@ pub(crate) fn build_responses_request( tools, parallel_tool_calls: true, stream: true, - store, - previous_response_id: previous_response_id.map(str::to_string), + store: options.store, + previous_response_id: options.previous_response_id.map(str::to_string), reasoning: reasoning_effort.map(|effort| ReasoningConfig { effort: Some(effort.to_string()), summary: None, @@ -305,6 +329,10 @@ struct ResponseError { #[derive(Debug, Deserialize)] #[serde(tag = "type", rename_all = "snake_case")] enum OutputItem { + Reasoning { + #[serde(default)] + content: Vec, + }, Message { #[serde(default)] role: Option, @@ -326,6 +354,10 @@ enum OutputItem { #[derive(Debug, Deserialize)] #[serde(tag = "type", rename_all = "snake_case")] enum OutputItemContent { + ReasoningText { + #[serde(default)] + text: String, + }, OutputText { #[serde(default)] text: String, @@ -360,6 +392,8 @@ where S: Stream>> + Unpin, { let mut full_text = String::new(); + let mut full_reasoning = String::new(); + let mut pending_reasoning = String::new(); let mut tool_calls: Vec = Vec::new(); let mut raw_buf: Vec = Vec::new(); let mut deadline = tokio::time::Instant::now() + idle.first_progress; @@ -459,6 +493,24 @@ where && let Ok(item) = serde_json::from_value::(item_val) { match item { + OutputItem::Reasoning { content } => { + let text: String = content.into_iter().filter_map(|part| { + match part { + OutputItemContent::ReasoningText { text } => Some(text), + _ => None, + } + }).collect(); + if pending_reasoning.is_empty() && !text.is_empty() { + on_thought(&text); + } + full_reasoning.push_str(if text.is_empty() { + &pending_reasoning + } else { + &text + }); + pending_reasoning.clear(); + made_progress = true; + } OutputItem::Message { role, content } => { if role.as_deref() == Some("assistant") && !deltas_received { for c in content { @@ -556,8 +608,14 @@ where ); } } - "response.reasoning_text.delta" - | "response.reasoning_summary_text.delta" => { + "response.reasoning_text.delta" => { + if let Some(delta) = event.delta { + on_thought(&delta); + pending_reasoning.push_str(&delta); + made_progress = true; + } + } + "response.reasoning_summary_text.delta" => { if let Some(delta) = event.delta { on_thought(&delta); made_progress = true; @@ -586,11 +644,13 @@ where if let Some(err) = failure { return Err(err); } + full_reasoning.push_str(&pending_reasoning); + let reasoning_content = (!full_reasoning.is_empty()).then_some(full_reasoning); if cancel.is_cancelled() { return Ok(ResponsesStreamOutcome { response: LlmResponse::Text { text: full_text, - reasoning_content: None, + reasoning_content, usage, codex_reasoning: None, }, @@ -608,7 +668,7 @@ where Ok(ResponsesStreamOutcome { response: LlmResponse::Text { text: full_text, - reasoning_content: None, + reasoning_content, usage, codex_reasoning: None, }, @@ -619,7 +679,7 @@ where Ok(ResponsesStreamOutcome { response: LlmResponse::ToolCalls { text: full_text, - reasoning_content: None, + reasoning_content, calls: tool_calls, usage, codex_reasoning: None, @@ -651,6 +711,78 @@ mod tests { Box::new(|_| {}) } + #[test] + fn reasoning_replay_is_opt_in_and_uses_stable_unique_ids() { + let messages = [ + ChatMessage::user("question"), + ChatMessage::assistant_with_reasoning("first answer", Some("first thought".into())), + ChatMessage::user("follow-up"), + ChatMessage::assistant_with_reasoning("second answer", Some("second thought".into())), + ]; + let build = |replay_reasoning| { + serde_json::to_value(build_responses_request( + "model", + &messages, + None, + None, + None, + ResponsesRequestOptions { + replay_reasoning, + ..Default::default() + }, + )) + .unwrap() + }; + let default = build(false); + assert_eq!(default["input"].as_array().unwrap().len(), 4); + assert!(!default.to_string().contains("thought")); + let replay = build(true); + assert_eq!(replay, build(true)); + let input = replay["input"].as_array().unwrap(); + assert_eq!(input.len(), 6); + assert_eq!(input[1]["type"], "reasoning"); + assert_eq!(input[4]["type"], "reasoning"); + assert_ne!(input[1]["id"], input[4]["id"]); + assert_eq!(input[1]["content"][0]["text"], "first thought"); + assert_eq!(input[4]["content"][0]["text"], "second thought"); + } + + #[tokio::test] + async fn reasoning_items_are_accumulated_without_replaying_summaries() { + let events = [ + serde_json::json!({"type":"response.reasoning_summary_text.delta","delta":"summary"}), + serde_json::json!({"type":"response.reasoning_text.delta","delta":"first"}), + serde_json::json!({"type":"response.output_item.done","item":{"type":"reasoning","content":[{"type":"reasoning_text","text":"first"}]}}), + serde_json::json!({"type":"response.output_item.done","item":{"type":"reasoning","content":[{"type":"reasoning_text","text":"second"}]}}), + serde_json::json!({"type":"response.reasoning_text.delta","delta":"third"}), + serde_json::json!({"type":"response.output_text.delta","delta":"answer"}), + serde_json::json!({"type":"response.completed","response":{}}), + ]; + let stream = + stream::iter(events.map(|event| Ok(format!("data: {event}\n\n").into_bytes()))); + let (on_thought, thoughts) = collect_tokens(); + let outcome = drive_responses_sse_stream( + stream, + noop_sink(), + on_thought, + CancellationToken::new(), + IdleTimeouts::uniform(std::time::Duration::from_secs(1)), + ) + .await + .unwrap(); + let LlmResponse::Text { + text, + reasoning_content, + .. + } = outcome.response + else { + panic!("expected text response"); + }; + assert_eq!(text, "answer"); + assert_eq!(reasoning_content.as_deref(), Some("firstsecondthird")); + assert_eq!(*thoughts.lock().unwrap(), "summaryfirstsecondthird"); + } + fn delayed_chunks( chunks: Vec<(std::time::Duration, &'static str)>, ) -> BoxStream<'static, Result>> { diff --git a/crates/anvil-client/src/structured_output.rs b/crates/anvil-client/src/structured_output.rs index dbbdeca..73cb515 100644 --- a/crates/anvil-client/src/structured_output.rs +++ b/crates/anvil-client/src/structured_output.rs @@ -25,6 +25,14 @@ pub struct StructuredOutputRequest { pub prefer_json_object: bool, } +/// In-band schema instruction for providers that only support JSON-object mode. +pub(crate) fn json_schema_instruction(request: &StructuredOutputRequest) -> String { + format!( + "Return only JSON matching this JSON Schema: {}", + request.schema + ) +} + #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct StructuredOutputSchemaError { pub instance_location: String, diff --git a/docs/src/content/docs/providers.md b/docs/src/content/docs/providers.md index 95e7f55..77aa820 100644 --- a/docs/src/content/docs/providers.md +++ b/docs/src/content/docs/providers.md @@ -29,7 +29,7 @@ printf '%s' '{"messages":[{"role":"system","content":"Classify the item."},{"rol | anvil infer --model codex::gpt-5.5 --reasoning-effort medium ``` -This path accepts only system and user text, supplies no tools, and bypasses ACP sessions, project instructions, skills, hooks, history, and the agent loop. The required model prefix selects the backend so no provider fallback can pick a different model: `codex::`, `meta::`, `kimi::`, `grok::`, or `deepseek::`. The corresponding credentials are the Codex auth file, native Muse login, Kimi Code credentials, Grok Build OAuth credentials, and the DeepSeek API key. Omit `--service-tier` to use the provider default. Transport diagnostics go to stderr, while successful stdout contains the validated `output`, aggregate token `usage`, and effective request settings. `--validation-retries` controls additional attempts after local JSON Schema validation fails. +This path accepts only system and user text, supplies no tools, and bypasses ACP sessions, project instructions, skills, hooks, history, and the agent loop. The required model prefix selects the backend so no provider fallback can pick a different model: `codex::`, `meta::`, `kimi::`, `grok::`, `mimo::`, `mimo-payg::`, or `deepseek::`. The corresponding credentials are the Codex auth file, native Muse login, Kimi Code credentials, Grok Build OAuth credentials, the Xiaomi MiMo Token Plan and pay-as-you-go API keys, and the DeepSeek API key. Omit `--service-tier` to use the provider default. Transport diagnostics go to stderr, while successful stdout contains the validated `output`, aggregate token `usage`, and effective request settings. `--validation-retries` controls additional attempts after local JSON Schema validation fails. ## Meta / Muse @@ -54,6 +54,30 @@ Anvil probes Ollama at `http://localhost:11434/v1/models`. Start Ollama, then ru On macOS and Linux, a running `ds4-server` is discovered from its listening port. Set `DS4_BASE_URL` to point at a non-default or remote endpoint, then refresh local discovery. +## Xiaomi MiMo + +### Token Plan + +Set `MIMO_TOKEN_PLAN_API_KEY` to the `tp-...` individual key or `ttp-...` team key from Xiaomi's Token Plan console. Anvil routes `mimo::*` models through Xiaomi's dedicated subscription Responses endpoint (`https://token-plan-cn.xiaomimimo.com/v1`), not the pay-as-you-go endpoint: + +```bash +export MIMO_TOKEN_PLAN_API_KEY="tp-your-token-plan-key" +``` + +Xiaomi's broader `MIMO_API_KEY` convention is accepted as a fallback when its `tp-` or `ttp-` prefix identifies a Token Plan key. Set `MIMO_TOKEN_PLAN_BASE_URL` only if Xiaomi's console provides a different Token Plan base URL. Current models include `mimo::mimo-v2.6-pro`, `mimo::mimo-v2.6-flash`, and `mimo::mimo-v2.6-pro-ultraspeed`. + +### Pay-as-you-go + +Set `MIMO_PAY_AS_YOU_GO_API_KEY` to the `sk-...` key from Xiaomi's API Keys console. Anvil routes `mimo-payg::*` models through the billed Responses endpoint (`https://api.xiaomimimo.com/v1`): + +```bash +export MIMO_PAY_AS_YOU_GO_API_KEY="sk-your-pay-as-you-go-key" +``` + +Xiaomi's broader `MIMO_API_KEY` convention is accepted as a fallback only when its `sk-` prefix identifies a pay-as-you-go key. Set `MIMO_PAY_AS_YOU_GO_BASE_URL` only if you need to route through a compatible proxy. Both plans can coexist by setting their dedicated variables and explicitly selecting either `mimo::mimo-v2.6-pro` or `mimo-payg::mimo-v2.6-pro`. + +MiMo structured inference uses JSON-object mode with schema instructions in the prompt and local validation. Discovery excludes audio-only ASR and TTS models. + ## Hosted DeepSeek and Kimi Code DeepSeek uses `DEEPSEEK_API_KEY` or credentials saved through `/setup deepseek`. Models use `deepseek::*` wire IDs. @@ -122,4 +146,4 @@ These profiles use baseline Chat Completions with streaming, tools, usage, and s Clients that advertise ACP elicitation forms receive out-of-transcript credential fields for OpenRouter, Bedrock, and DeepSeek. In a text-only client, commands such as `/setup openrouter key ` remain available but the pasted secret becomes part of the session transcript. Prefer environment variables or elicitation forms for sensitive credentials. -Provider priority for automatic selection is Bedrock, Codex, Meta, local models (Ollama then ds4), DeepSeek, Kimi, Grok, generic OpenAI-compatible profiles, then OpenRouter. Override it for the current session with `/setup model `. +Provider priority for automatic selection is Bedrock, Codex, Meta, local models (Ollama then ds4), DeepSeek, Kimi, Grok, Xiaomi MiMo Token Plan then pay-as-you-go, generic OpenAI-compatible profiles, then OpenRouter. Override it for the current session with `/setup model `. diff --git a/src/main.rs b/src/main.rs index de324c3..fb6ff60 100644 --- a/src/main.rs +++ b/src/main.rs @@ -42,7 +42,8 @@ mod workspace_delta; // standalone `anvil_client` crate; these module imports keep the bare // `::` paths below working. use anvil_client::hosted::{ - build_deepseek_backend, build_grok_backend, build_kimi_backend, deepseek_backend_from_key, + build_deepseek_backend, build_grok_backend, build_kimi_backend, + build_mimo_pay_as_you_go_backend, build_mimo_token_plan_backend, deepseek_backend_from_key, }; use anvil_client::llm_client::LlmBackend; use anvil_client::multi_backend::{BackendRegistration, MultiBackend}; @@ -574,6 +575,8 @@ async fn build_multi_backend(transient_setup: bool) -> Result> let codex_backend = build_codex_backend().await; let deepseek_backend = build_deepseek_backend(); let kimi_backend = build_kimi_backend(); + let mimo_backend = build_mimo_token_plan_backend(); + let mimo_payg_backend = build_mimo_pay_as_you_go_backend(); let grok_backend = build_grok_backend(); let meta_backend = match anvil_client::meta_client::MetaClient::load() { Ok(backend) => backend, @@ -621,6 +624,19 @@ async fn build_multi_backend(transient_setup: bool) -> Result> "Grok backend not available; install Grok Build and run `grok login --oauth`." ); } + if mimo_backend.is_none() { + tracing::info!( + "Xiaomi MiMo Token Plan backend not available; set MIMO_TOKEN_PLAN_API_KEY (or \ + Xiaomi's MIMO_API_KEY) to a Token Plan key to enable it." + ); + } + if mimo_payg_backend.is_none() { + tracing::info!( + "Xiaomi MiMo pay-as-you-go backend not available; set \ + MIMO_PAY_AS_YOU_GO_API_KEY (or Xiaomi's MIMO_API_KEY) to a \ + pay-as-you-go key to enable it." + ); + } if openrouter_backend.is_none() { tracing::info!( "OpenRouter backend not available; set {} or run `/setup openrouter key ` \ @@ -645,6 +661,16 @@ async fn build_multi_backend(transient_setup: bool) -> Result> deepseek_backend, ), BackendRegistration::new(discovery::ModelSource::KIMI, "Kimi", kimi_backend), + BackendRegistration::new( + discovery::ModelSource::MIMO, + "Xiaomi MiMo Token Plan", + mimo_backend, + ), + BackendRegistration::new( + discovery::ModelSource::MIMO_PAYG, + "Xiaomi MiMo Pay-as-you-go", + mimo_payg_backend, + ), BackendRegistration::new(discovery::ModelSource::GROK, "Grok", grok_backend), BackendRegistration::new( discovery::ModelSource::OPENAI,