From f1e5c14fcf3d66d3dfa930c389950f62f8b12f66 Mon Sep 17 00:00:00 2001 From: Yasyf Mohamedali Date: Thu, 24 Sep 2026 19:47:17 -0700 Subject: [PATCH] =?UTF-8?q?packs:=20=E2=9A=A1=EF=B8=8F=20turn=20off=20Cere?= =?UTF-8?q?bras=20reasoning=20in=20the=20plain-English=20rewrite?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Context: Follows #165. The plain-English rewrite calls Cerebras `qwen-3.8-27b`, which reasons by default. On a 2,300-character reply it spent a median 6,300 reasoning tokens for about 520 output tokens, so even with #165's 6-second cap many rewrites timed out and showed the original. Summary: `plain_english` passes `reasoning_effort="none"` to `OpenAiEndpointBackend`, and the spawnllm pin moves to `>=0.14.0,<0.15`, the first release with that field (yasyf/spawnllm#10). Motivation: The owner chose to make `reasoning_effort` a typed field in spawnllm rather than override the backend's request planning here. Details: 20 calls through the real `plain_english.plain_english()` path, same prompt and harness, a fresh nonce per call: | | p50 | p95 | max | fell back to original | |---|---|---|---|---| | #165 (spawnllm 0.13.4, reasoning on) | 4.42s | 6.02s | 6.47s | 3 of 20 (6s cap) | | this PR (spawnllm with reasoning_effort, `"none"`) | 0.52s | 0.98s | 1.13s | 0 of 20 | The "after" row ran against yasyf/spawnllm#10 installed from its branch. After v0.14.0 was published, 12 more calls against the locked release gave 0.52s p50 and 0.74s p95, with no fallbacks. `uv lock --upgrade-package spawnllm` moves the lock from 0.13.4 to 0.14.0. `test_streamed_message_is_rewritten_once` asserts that the backend carries `reasoning_effort == "none"`. Both changelog entries move to Unreleased: #165's landed under 12.56.0, which was tagged before #165 merged. `pytest tests/test_plain_english.py tests/test_plugin_hooks.py` passes. --- CHANGELOG.md | 24 +++++++++++++------ .../general/hooks/plain_english.py | 4 +++- pyproject.toml | 2 +- tests/test_plain_english.py | 3 ++- uv.lock | 10 ++++---- 5 files changed, 28 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index dd825eb2..f20fa478 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,23 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Fixed + +- **A slow plain-English rewrite no longer hides the reply.** Claude Code + cancels a `MessageDisplay` hook after 10 seconds whatever `hooks.json` asks + for, and then shows only the last chunk's original text. The rewrite waited + up to 20 seconds for Cerebras after up to 2 seconds of chunk assembly, so a + long reply lost everything but its last chunk. The rewrite now gives up + after 6 seconds and shows the whole original reply, and `hooks.json` + registers `MessageDisplay` with the 10-second timeout Claude Code enforces. +- **The plain-English rewrite answers in about half a second.** Cerebras' + `qwen-3.8-27b` reasons by default, and on a 2,300-character reply it spent + a median 6,300 reasoning tokens for about 520 output tokens. That made a + rewrite take 4.4s at p50, and 3 of 20 hit the 6-second cap and showed the + original. The rewrite now sends `reasoning_effort: "none"` through spawnllm + 0.14.0's `OpenAiEndpointBackend`, which brings the same 20 rewrites to + 0.52s at p50 and 0.98s at p95 with none falling back. + ## [12.56.0] - 2026-09-24 ### Added @@ -63,13 +80,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `additionalContext` envelope a warn rendered failed schema validation. A warn, context, or allow message now prints as plain text, and a block renders `{"decision": "block"}`, which cancels the compaction. -- **A slow plain-English rewrite no longer hides the reply.** Claude Code - cancels a `MessageDisplay` hook after 10 seconds whatever `hooks.json` asks - for, and then shows only the last chunk's original text. The rewrite waited - up to 20 seconds for Cerebras after up to 2 seconds of chunk assembly, so a - long reply lost everything but its last chunk. The rewrite now gives up - after 6 seconds and shows the whole original reply, and `hooks.json` - registers `MessageDisplay` with the 10-second timeout Claude Code enforces. ### Removed diff --git a/captain_hook/builtin_packs/general/hooks/plain_english.py b/captain_hook/builtin_packs/general/hooks/plain_english.py index ef246598..1b2da458 100644 --- a/captain_hook/builtin_packs/general/hooks/plain_english.py +++ b/captain_hook/builtin_packs/general/hooks/plain_english.py @@ -73,7 +73,9 @@ def plain_english(evt: MessageDisplayEvent, text: str, api_key: str) -> str: contextvars.copy_context().run, evt.ctx.call_llm, rewrite_prompt(evt, text), - backend=OpenAiEndpointBackend("https://api.cerebras.ai/v1", "qwen-3.8-27b", api_key=api_key), + backend=OpenAiEndpointBackend( + "https://api.cerebras.ai/v1", "qwen-3.8-27b", api_key=api_key, reasoning_effort="none" + ), timeout=max(1, int(budget)), ) try: diff --git a/pyproject.toml b/pyproject.toml index 7fe6b34f..106903ff 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -43,7 +43,7 @@ dependencies = [ "filelock>=3", "pathspec>=0.12", "loguru>=0.7.3", - "spawnllm[sdk]>=0.13.4,<0.14", + "spawnllm[sdk]>=0.14.0,<0.15", "mcp>=2.0,<3", ] diff --git a/tests/test_plain_english.py b/tests/test_plain_english.py index 59b3ef7f..065259f1 100644 --- a/tests/test_plain_english.py +++ b/tests/test_plain_english.py @@ -123,10 +123,11 @@ def test_streamed_message_is_rewritten_once(ctx: CerebrasStub) -> None: ) backend = kwargs["backend"] assert isinstance(backend, OpenAiEndpointBackend) - assert (backend.base_url, backend.model, backend.api_key) == ( + assert (backend.base_url, backend.model, backend.api_key, backend.reasoning_effort) == ( "https://api.cerebras.ai/v1", "qwen-3.8-27b", "test-key", + "none", ) assert kwargs["timeout"] == 6 assert ctx.session.load(plain_english.PlainEnglishBuffer).messages == {} diff --git a/uv.lock b/uv.lock index 75ff8440..49bc89ac 100644 --- a/uv.lock +++ b/uv.lock @@ -294,7 +294,7 @@ requires-dist = [ { name = "rich", specifier = ">=13" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.8" }, { name = "spacy", specifier = ">=3.7" }, - { name = "spawnllm", extras = ["sdk"], specifier = ">=0.13.4,<0.14" }, + { name = "spawnllm", extras = ["sdk"], specifier = ">=0.14.0,<0.15" }, { name = "wn", specifier = ">=1.1.0" }, ] provides-extras = ["dev"] @@ -2494,7 +2494,7 @@ wheels = [ [[package]] name = "spawnllm" -version = "0.13.4" +version = "0.14.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "click" }, @@ -2503,10 +2503,10 @@ dependencies = [ { name = "pydantic" }, { name = "wasmtime" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/e6/19/f9e69af9b0775a2a95a2b3da81eb62097c4f2a772b1b5d16102dc9e55171/spawnllm-0.13.4.tar.gz", hash = "sha256:bc9d7d62cfb2a3599888c50c530f4d548c4f8612f7a1fb5d29dd4053c340d8dc", size = 279205, upload-time = "2026-09-17T20:21:26.48Z" } +sdist = { url = "https://files.pythonhosted.org/packages/51/66/7395d6df9f72d6338ce08fd8353991eed2d457d13e2f04157e9c5bff9595/spawnllm-0.14.0.tar.gz", hash = "sha256:e0e026c41d37ac56295dd3e4cf5642cb13fa2f36cdb80d6e8bb982c955baa3d6", size = 279854, upload-time = "2026-09-25T03:04:50.588Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/24/94/da1411382cd7b114527b6d539519b979543d8aa868d3e79e2123cf2837c9/spawnllm-0.13.4-py3-none-any.whl", hash = "sha256:e6390e943193720d22e0aff10d454ba6590988dfe9a165f7cf8ceffad4c9830e", size = 290125, upload-time = "2026-09-17T20:21:23.553Z" }, - { url = "https://files.pythonhosted.org/packages/ce/57/c43fc9bdd0517855ebd01bb504fbdc95e702ff8bca0131765a7960c7ca40/spawnllm-0.13.4-py3-none-macosx_26_0_arm64.whl", hash = "sha256:18275a8bdc2cced7bcca1013ec1dfb46d2335f16a440501da2c20a6558b5bfc3", size = 376439, upload-time = "2026-09-17T20:21:25.191Z" }, + { url = "https://files.pythonhosted.org/packages/77/20/b061d62c2f9a064a865155e752a1ad9255b8f2c09699d5744e2b29b12373/spawnllm-0.14.0-py3-none-any.whl", hash = "sha256:a60d6f243a8ac75b5ea7e92551b5607346d8d73376a5d10934a90b03c1cfbfa9", size = 290809, upload-time = "2026-09-25T03:04:47.351Z" }, + { url = "https://files.pythonhosted.org/packages/a7/f9/44c108ee6cd3f07b4379962105c76afc7fd2fc01ecad4cf524da1ed00d6f/spawnllm-0.14.0-py3-none-macosx_26_0_arm64.whl", hash = "sha256:c480df4b7da100afb48dac2c8d877d2aa6a0e68edbce823675a2d845258a61c4", size = 377105, upload-time = "2026-09-25T03:04:48.981Z" }, ] [package.optional-dependencies]