diff --git a/CHANGELOG.md b/CHANGELOG.md index e633652..bc54f44 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] ### Fixed +- **A permanent CLI error no longer retries because the echoed prompt contains a + three-digit number.** The retry policy matched any bare `5xx`, `overloaded`, + or `rate limit` anywhere in the error message, and `codex` echoes the prompt + into stderr ahead of its error line. A prompt mentioning "512 MiB" made + codex's `status: 400` "model is not supported" error look transient, so every + call burned about 137s of backoff before failing. The policy now reads only + the error message's last non-empty line, the line where `codex` prints + `ERROR: ...` and `claude` prints `API Error: ...`, and counts a 5xx only in + a status position, such as `"status":503`, `last status: 502`, `API Error: 500`, or + an HTTP backend's `exited 503:` header. A 4xx is never transient. - **A newline-free stderr blob past 64 KiB no longer crashes a CLI-backed run.** `_tee_stderr` drained stderr with `async for`, which uses `StreamReader.readline` and caps a line at 64 KiB — a longer newline-free diff --git a/conformance/vectors/resolve/codex-exit-400-echoed-prompt.json b/conformance/vectors/resolve/codex-exit-400-echoed-prompt.json new file mode 100644 index 0000000..e2d645c --- /dev/null +++ b/conformance/vectors/resolve/codex-exit-400-echoed-prompt.json @@ -0,0 +1,19 @@ +{ + "name": "codex-exit-400-echoed-prompt", + "op": "resolve", + "input": { + "provider": "codex", + "raw": "", + "returncode": 1, + "stderr": "user\nCap the sandbox at 512 MiB and retry on a 503.\nERROR: {\"type\":\"error\",\"status\":400,\"error\":{\"type\":\"invalid_request_error\",\"message\":\"The 'gpt-5.4-mini' model is not supported when using Codex with a ChatGPT account.\"}}", + "wants_value": false + }, + "expected": { + "status": "error", + "kind": "exit", + "msg": "codex exited 1: user\nCap the sandbox at 512 MiB and retry on a 503.\nERROR: {\"type\":\"error\",\"status\":400,\"error\":{\"type\":\"invalid_request_error\",\"message\":\"The 'gpt-5.4-mini' model is not supported when using Codex with a ChatGPT account.\"}}", + "transient": false, + "cost_usd": null, + "usage": null + } +} diff --git a/conformance/vectors/retry_decision/non-transient-codex-400-echoed-prompt.json b/conformance/vectors/retry_decision/non-transient-codex-400-echoed-prompt.json new file mode 100644 index 0000000..e5a8c91 --- /dev/null +++ b/conformance/vectors/retry_decision/non-transient-codex-400-echoed-prompt.json @@ -0,0 +1,13 @@ +{ + "name": "non-transient-codex-400-echoed-prompt", + "op": "retry_decision", + "input": { + "attempt": 0, + "max_attempts": 5, + "error_msg": "codex exited 1: user\nCap the sandbox at 512 MiB and retry on a 503.\nERROR: {\"type\":\"error\",\"status\":400,\"error\":{\"type\":\"invalid_request_error\",\"message\":\"The 'gpt-5.4-mini' model is not supported when using Codex with a ChatGPT account.\"}}" + }, + "expected": { + "retry": false, + "sleep_s": 0.0 + } +} diff --git a/conformance/vectors/retry_decision/non-transient-rate-limit-in-echoed-prompt.json b/conformance/vectors/retry_decision/non-transient-rate-limit-in-echoed-prompt.json new file mode 100644 index 0000000..48f44fa --- /dev/null +++ b/conformance/vectors/retry_decision/non-transient-rate-limit-in-echoed-prompt.json @@ -0,0 +1,13 @@ +{ + "name": "non-transient-rate-limit-in-echoed-prompt", + "op": "retry_decision", + "input": { + "attempt": 0, + "max_attempts": 5, + "error_msg": "codex exited 1: user\nBack off when you hit a rate limit.\nERROR: stream closed unexpectedly" + }, + "expected": { + "retry": false, + "sleep_s": 0.0 + } +} diff --git a/conformance/vectors/retry_decision/transient-500-attempt-3-caps-at-60.json b/conformance/vectors/retry_decision/transient-500-attempt-3-caps-at-60.json index c776f10..8593ed1 100644 --- a/conformance/vectors/retry_decision/transient-500-attempt-3-caps-at-60.json +++ b/conformance/vectors/retry_decision/transient-500-attempt-3-caps-at-60.json @@ -4,7 +4,7 @@ "input": { "attempt": 3, "max_attempts": 5, - "error_msg": "internal 500 error" + "error_msg": "ERROR: {\"type\":\"error\",\"status\":500,\"error\":{\"type\":\"server_error\",\"message\":\"internal error\"}}" }, "expected": { "retry": true, diff --git a/conformance/vectors/retry_decision/transient-503-attempt-2.json b/conformance/vectors/retry_decision/transient-503-attempt-2.json index 1d71e30..523bcbc 100644 --- a/conformance/vectors/retry_decision/transient-503-attempt-2.json +++ b/conformance/vectors/retry_decision/transient-503-attempt-2.json @@ -4,7 +4,7 @@ "input": { "attempt": 2, "max_attempts": 5, - "error_msg": "upstream returned 503" + "error_msg": "API Error: 503 {\"type\":\"error\",\"error\":{\"type\":\"api_error\",\"message\":\"Service Unavailable\"}}" }, "expected": { "retry": true, diff --git a/conformance/vectors/retry_decision/transient-codex-retry-limit-502.json b/conformance/vectors/retry_decision/transient-codex-retry-limit-502.json new file mode 100644 index 0000000..2768b04 --- /dev/null +++ b/conformance/vectors/retry_decision/transient-codex-retry-limit-502.json @@ -0,0 +1,13 @@ +{ + "name": "transient-codex-retry-limit-502", + "op": "retry_decision", + "input": { + "attempt": 0, + "max_attempts": 5, + "error_msg": "codex exited 1: user\nCap the sandbox at 512 MiB.\nERROR: exceeded retry limit, last status: 502 Bad Gateway" + }, + "expected": { + "retry": true, + "sleep_s": 5.0 + } +} diff --git a/rust/conformance-gen/src/cases.rs b/rust/conformance-gen/src/cases.rs index 871c33c..1cb1891 100644 --- a/rust/conformance-gen/src/cases.rs +++ b/rust/conformance-gen/src/cases.rs @@ -587,6 +587,10 @@ fn resolve_case( } } +const CODEX_400_ECHOED_PROMPT: &str = r#"codex exited 1: user +Cap the sandbox at 512 MiB and retry on a 503. +ERROR: {"type":"error","status":400,"error":{"type":"invalid_request_error","message":"The 'gpt-5.4-mini' model is not supported when using Codex with a ChatGPT account."}}"#; + fn resolve_cases() -> Vec { vec![ resolve_case( @@ -679,6 +683,14 @@ fn resolve_cases() -> Vec { "codex exec failed", false, ), + resolve_case( + "codex-exit-400-echoed-prompt", + "codex", + "", + 1, + CODEX_400_ECHOED_PROMPT.trim_start_matches("codex exited 1: "), + false, + ), resolve_case( "codex-exit-raw-tail", "codex", @@ -926,13 +938,17 @@ fn retry_cases() -> Vec { "transient-503-attempt-2", 2, 5, - Some("upstream returned 503"), + Some( + r#"API Error: 503 {"type":"error","error":{"type":"api_error","message":"Service Unavailable"}}"#, + ), ), retry_case( "transient-500-attempt-3-caps-at-60", 3, 5, - Some("internal 500 error"), + Some( + r#"ERROR: {"type":"error","status":500,"error":{"type":"server_error","message":"internal error"}}"#, + ), ), retry_case( "transient-overloaded-attempt-0", @@ -947,6 +963,28 @@ fn retry_cases() -> Vec { Some("529 overloaded"), ), retry_case("transient-max-attempts-one", 0, 1, Some("529 overloaded")), + retry_case( + "transient-codex-retry-limit-502", + 0, + 5, + Some( + "codex exited 1: user\nCap the sandbox at 512 MiB.\nERROR: exceeded retry limit, last status: 502 Bad Gateway", + ), + ), + retry_case( + "non-transient-codex-400-echoed-prompt", + 0, + 5, + Some(CODEX_400_ECHOED_PROMPT), + ), + retry_case( + "non-transient-rate-limit-in-echoed-prompt", + 0, + 5, + Some( + "codex exited 1: user\nBack off when you hit a rate limit.\nERROR: stream closed unexpectedly", + ), + ), retry_case( "non-transient-attempt-0", 0, diff --git a/rust/spawnllm-core/src/retry.rs b/rust/spawnllm-core/src/retry.rs index 6aa709f..4d591e5 100644 --- a/rust/spawnllm-core/src/retry.rs +++ b/rust/spawnllm-core/src/retry.rs @@ -6,8 +6,12 @@ use serde_json::Value; use crate::{OpError, OpResult, from_input}; -static TRANSIENT: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b529\b|overloaded|rate.?limit|\b5\d\d\b").unwrap()); +static TRANSIENT_ERROR_LINE: LazyLock = LazyLock::new(|| { + Regex::new(r#"(?i)overloaded|rate.?limit|(?:\bstatus|\bAPI Error)[\s":=]*5\d\d\b"#).unwrap() +}); + +static TRANSIENT_HTTP_EXIT: LazyLock = + LazyLock::new(|| Regex::new(r"^\w+ exited 5\d\d:").unwrap()); #[derive(Debug, Clone, Deserialize)] pub struct RetryInput { @@ -23,7 +27,12 @@ pub struct RetryDecision { } pub(crate) fn is_transient(msg: &str) -> bool { - TRANSIENT.is_match(msg) + TRANSIENT_HTTP_EXIT.is_match(msg) + || msg + .lines() + .rev() + .find(|line| !line.trim().is_empty()) + .is_some_and(|line| TRANSIENT_ERROR_LINE.is_match(line)) } pub fn backoff(attempt: u32) -> f64 {