From d2e77e283ae31d949eedb64c3e9822c6a2501429 Mon Sep 17 00:00:00 2001 From: drewOrc <36374426+drewOrc@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:34:16 +0800 Subject: [PATCH] Send temperature through extra_body; check kwargs against SDK signatures The first make llm-smoke failed on every call with "Messages.create() got an unexpected keyword argument 'temperature'": anthropic 1.x removed the sampling parameters from messages.create, though the API and Haiku 4.5 still accept them. No request reached the API; nothing was spent. temperature 0 is part of the comparison with the old project, so it now goes in extra_body, which the SDK merges into the request JSON as is. The identity still records temperature 0 and its hash is unchanged. The fake client accepts any keyword, which is why no test caught this. A new test binds the kwargs we send to the installed SDK's Messages.create and Messages.count_tokens signatures (no network, no key), and another checks the fake client receives exactly those kwargs. --- DEVLOG.md | 24 ++++++++++++++++++++++++ src/tinyrouter/llm.py | 28 ++++++++++++++++++++-------- tests/llm_fakes.py | 2 ++ tests/test_llm.py | 30 ++++++++++++++++++++++++++++-- 4 files changed, 74 insertions(+), 10 deletions(-) diff --git a/DEVLOG.md b/DEVLOG.md index 1789a61..be5cdc6 100644 --- a/DEVLOG.md +++ b/DEVLOG.md @@ -4,6 +4,30 @@ --- +## 2026-09-29(夜):llm-smoke 第一次實跑失敗,temperature 改走 extra_body + +### 本次工作 / 執行摘要 +- 事故:PR #14 合併後第一次 `make llm-smoke`,每筆呼叫都在送出前就失敗,錯誤是 `TypeError: Messages.create() got an unexpected keyword argument 'temperature'`。沒有請求到達 API,花費 US$0,輸出裡沒有 key。 +- 原因:anthropic 1.x 把 `temperature`、`top_p`、`top_k` 從 `messages.create()` 的簽名拿掉了,但 API 本身沒有移除,Haiku 4.5 仍然接受。測試用的假 client 什麼參數都收,所以測不出和真 SDK 簽名不符。 +- 修法:`request_params` 改成 `extra_body={"temperature": 0.0}`,SDK 會把它原樣併進 request JSON(claude-api skill,`python/claude-api/sdk-upgrade.md` Step 6 的建議:模型仍接受、程式又依賴這個設定時,移到 extra_body,不要刪)。identity 仍記 temperature 0,語意沒變,所以身分雜湊不變,smoke journal 裡的 failed 紀錄下次會照常重試。 +- `count_tokens` 只送 model、system、messages,沒有同樣的問題;它的參數也抽成 `count_tokens_params()`,跟 `request_params()` 一樣是唯一產生參數的地方。 +- 疤痕變腳本:新增 `test_the_arguments_we_send_fit_the_installed_sdk_signatures`,把實際送出的 kwargs `bind` 到真 SDK 的 `Messages.create` 與 `Messages.count_tokens` 簽名上,不需要網路或 key;另一條測試確認假 client 收到的參數正是這兩個函式產生的,簽名測試因此涵蓋實際送出的內容。 + +### 核心發現 / 數據 +- (無實跑數據)。定價仍是 Haiku 4.5 輸入 US$1、輸出 US$5 per MTok。 + +### Blockers / 遇到的問題 +- (無) + +### Next +- [ ] 合併後重跑 `make llm-smoke` + +### Files / Budget +- `src/tinyrouter/llm.py`、`tests/test_llm.py`、`tests/llm_fakes.py` +- API 花費:US$0 + +--- + ## 2026-09-29(晚):PR #14 審查修正(4 medium、6 low) ### 本次工作 / 執行摘要 diff --git a/src/tinyrouter/llm.py b/src/tinyrouter/llm.py index c437e63..c5916a8 100644 --- a/src/tinyrouter/llm.py +++ b/src/tinyrouter/llm.py @@ -136,13 +136,29 @@ def cost_usd( def request_params(query: str) -> dict[str, Any]: - """The exact Messages API arguments for one query; the single place they are built.""" + """The exact ``messages.create`` arguments for one query; the single place they are built. + + anthropic 1.x removed ``temperature`` from ``messages.create`` (passing + it is a TypeError), but the API did not: Haiku 4.5 still accepts it. The + comparison with the old project depends on temperature 0, so it goes in + ``extra_body``, which the SDK merges into the request JSON unchanged + (claude-api skill, python/claude-api/sdk-upgrade.md, Step 6). + """ return { "model": MODEL, "max_tokens": MAX_TOKENS, - "temperature": TEMPERATURE, "system": SYSTEM_PROMPT, "messages": [{"role": "user", "content": query}], + "extra_body": {"temperature": float(TEMPERATURE)}, + } + + +def count_tokens_params() -> dict[str, Any]: + """The ``messages.count_tokens`` arguments: the prompt with a one-character query.""" + return { + "model": MODEL, + "system": SYSTEM_PROMPT, + "messages": [{"role": "user", "content": "x"}], } @@ -300,12 +316,8 @@ def prompt_base_tokens(client: Any, sleep: Callable[[float], None] = time.sleep) Retried and redacted like ``classify``; LLMCallError when it gives up. """ - counted, _, _ = with_retries( - lambda: client.messages.count_tokens( - model=MODEL, system=SYSTEM_PROMPT, messages=[{"role": "user", "content": "x"}] - ), - sleep, - ) + params = count_tokens_params() + counted, _, _ = with_retries(lambda: client.messages.count_tokens(**params), sleep) return int(counted.input_tokens) diff --git a/tests/llm_fakes.py b/tests/llm_fakes.py index 1bfbdb8..34ec4c2 100644 --- a/tests/llm_fakes.py +++ b/tests/llm_fakes.py @@ -56,6 +56,7 @@ def __init__( self.base_tokens = base_tokens self.calls: list[dict] = [] self.count_calls = 0 + self.count_kwargs: list[dict] = [] self.count_failures: list[BaseException] = [] self.response_override = None self._lock = threading.Lock() @@ -87,6 +88,7 @@ def create(self, **kwargs): def count_tokens(self, **kwargs): self.count_calls += 1 + self.count_kwargs.append(kwargs) if self.count_failures: raise self.count_failures.pop(0) return SimpleNamespace(input_tokens=self.base_tokens) diff --git a/tests/test_llm.py b/tests/test_llm.py index 9ae34de..a20b9bf 100644 --- a/tests/test_llm.py +++ b/tests/test_llm.py @@ -1,4 +1,5 @@ import hashlib +import inspect import pytest @@ -58,10 +59,35 @@ def test_request_matches_the_old_project_settings(): assert params == { "model": "claude-haiku-4-5-20251001", "max_tokens": 20, - "temperature": 0, "system": SYSTEM_PROMPT, "messages": [{"role": "user", "content": "book a flight"}], + "extra_body": {"temperature": 0.0}, } + assert "temperature" not in params + assert llm.identity()["temperature"] == 0 + + +def test_the_arguments_we_send_fit_the_installed_sdk_signatures(): + """2026-09-29 smoke incident: every call raised TypeError before reaching the API. + + anthropic 1.8.0 dropped ``temperature`` from ``Messages.create``; the + fake client accepts any keyword, so no test noticed. Binding the real + kwargs to the real SDK signatures catches that without a network or key. + """ + from anthropic.resources.messages import Messages + + create = inspect.signature(Messages.create) + create.bind(None, **llm.request_params("book a flight")) + count = inspect.signature(Messages.count_tokens) + count.bind(None, **llm.count_tokens_params()) + + +def test_the_fake_client_receives_exactly_the_built_arguments(): + client = fake_client() + classify("book a flight", client, sleep=lambda _: None) + llm.prompt_base_tokens(client, sleep=lambda _: None) + assert client.messages.calls == [llm.request_params("book a flight")] + assert client.messages.count_kwargs == [llm.count_tokens_params()] @pytest.mark.parametrize( @@ -92,7 +118,7 @@ def test_classify_returns_prediction_with_usage_and_request_id(): assert (result.input_tokens, result.output_tokens) == (250, 4) assert result.request_id == "req_fake_1" assert result.attempts == 1 - assert client.messages.calls[-1]["temperature"] == 0 + assert client.messages.calls[-1]["extra_body"] == {"temperature": 0.0} def test_classify_retries_a_429_then_succeeds_honouring_retry_after():