From 8ea6443998f8675d74fb8646aec7295f3738d1b5 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 24 Sep 2026 08:40:27 +0000 Subject: [PATCH] =?UTF-8?q?=E5=9B=9B=E5=A4=84=E5=A4=8D=E7=8E=B0=E8=BF=87?= =?UTF-8?q?=E7=9A=84=E6=AF=9B=E7=97=85:=E4=B8=8A=E4=B8=8B=E6=96=87?= =?UTF-8?q?=E5=B0=BA=E5=AD=90=E6=BC=8F=E4=BA=86=20tool=5Fcalls=20=E5=8F=82?= =?UTF-8?q?=E6=95=B0=E3=80=81500=20=E4=B8=8D=E9=87=8D=E8=AF=95=E3=80=81/re?= =?UTF-8?q?sume=20=E4=B8=8D=E8=AE=A4=E5=8E=9F=E8=AF=9D=E3=80=81=E6=A3=80?= =?UTF-8?q?=E7=B4=A2=E5=BB=BA=E5=9B=BE=E4=B8=A4=E4=B8=A4=E6=AF=94=E8=BE=83?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 一次全仓复核里挑出来的四条,每条都先写判据、看它红,再改。 一、上下文预算只量了一半 _ctx_chars 只数 content,而 write_file 写的整份文件躺在 tool_calls[].arguments 里。 实测 6 次 write_file、每次 1.8 万字符:真实历史 123,922 字符,它报 73。 压缩永远不触发,_prune_old_tool_results 也只剪 role=tool —— 写得越多的任务 越是每一步把所有写过的文件原样重发,直到 400 超长。 · _msg_chars:正文 + 参数,_ctx_chars 和 _tail_start 都走它 · 窗口外的大块参数跟旧工具输出同一条规则剪掉;剪完仍是合法 JSON、path 留着, spawn_subagent 不剪(理由同它的返回不剪) · 连写 12 个 6000 字符文件的一轮:最大一次请求 84,602 → 压在 COMPACT_AT 以下 · SECURITY.md 那条「会话文件不保证是原始内容」补上参数这一半 二、500 一次就放弃 重试判据是在报错文本里找关键词,表里有 502/503 没有 500。用真 openai 异常对象核对: 500 试 1 次;504 的 "Gateway Timeout" 被记成读超时、只多给一次; 429「余额不足」反而重试满三趟。FINDINGS 七十八节那个任务三趟死法正是 500、500、429。 · 按 status_code 认 408/429/5xx,文本表留着给没有状态码的异常 · 有状态码就不算读超时 · 余额/额度耗尽直接报(只认明确写法 —— Gemini 每分钟限流的文本也是 "exceeded your current quota",那是真限流,照样重试) · 认 Retry-After / retry-after-ms,封顶 60s;没有就 2/4/8s(原来 1/2s) · 四条走真退避的旧判据补上 sleep 桩,全套从 12s 回到 9s 三、REPL 里 /resume 不认人的原话 启动时 --resume 会 _human_asks 重建 state["asks"],REPL 里 /resume 和 /delete 掉当前会话这两条路没有:切过去之后 asks 还是上一个会话的对象, 续上的会话一压缩人的原话就被转述掉,「点名要过」也拿上一个会话的话去比。 抽成 _adopt_history,三条路都走它。旧判据只查「repl 里调过 _human_asks」, 启动那一处就满足了;新判据真把 REPL 跑一遍。 四、检索建图两两比较 _edges 对全部节点两两求交,每轮顶层、复盘、子 agent 各建一遍,节点数随会话数涨。 实测每次 recall():500 个会话 0.23s,1500 个 1.3s,3000 个 4.6s。 改成关键词倒排,只数真共享关键词的那几对;每行邻居按下标升序,跟原来逐项相同 (_activate 按这个顺序累加浮点,顺序一变同分的可能换位)。 FINDINGS 句子做语料:3000 个会话 recall() 1540ms → 319ms,输出逐字相同。 336 判据:334 过、5 skip、1 xfail(原有);py3.10 / 3.11 / 3.13 都跑过。 三个 selfcheck 绿。 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014gdkHx6rLSMVsUmQiErVSn --- DEVELOPMENT.md | 4 +- README.md | 10 +- SECURITY.md | 2 +- agent.py | 125 ++++++++++++++++--- recall.py | 27 ++++- tests/test_goal.py | 2 + tests/test_loop.py | 280 ++++++++++++++++++++++++++++++++++++++++++- tests/test_memory.py | 62 ++++++++++ 8 files changed, 482 insertions(+), 30 deletions(-) diff --git a/DEVELOPMENT.md b/DEVELOPMENT.md index 91717b6..0838a40 100644 --- a/DEVELOPMENT.md +++ b/DEVELOPMENT.md @@ -150,7 +150,7 @@ def check_permission(state, cls, name, args): # 决策 + ```bash python agent.py --selfcheck # 零依赖、零网络、零 key。改完先跑这个 -python -m pytest tests/ -q # 326 条。CI 跑 ubuntu/windows × 3.10/3.13 +python -m pytest tests/ -q # 336 条。CI 跑 ubuntu/windows × 3.10/3.13 ``` `--selfcheck` 覆盖 4 个工具的读/写/改、`edit_file` 的"找不到 / 不唯一"两种报错、 @@ -190,7 +190,7 @@ set TALOS_MAX_STEPS=12 | **行为类** | **模型读到守卫那条消息之后干什么** | ❌ 必须用你真在用的那个 | 「拒绝之后会不会改用 `python -c` 绕路」是那个具体模型的性格,换了模型不算数。 -而**行为类正是 live 测试唯一还值钱的部分** —— 管道通不通,326 条判据已经免费覆盖了。 +而**行为类正是 live 测试唯一还值钱的部分** —— 管道通不通,336 条判据已经免费覆盖了。 按这条界,**目标闸的 live 判据是行为类的**:离线判据已经证明判断器拿得到 `read_file`、 越界会被驳回、判不出来时不假装成功;它们证明不了的是**这个具体模型拿到只读工具之后 diff --git a/README.md b/README.md index 49e7ad5..377c16b 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ [![tests](https://github.com/Jerry-TZ/Talos/actions/workflows/test.yml/badge.svg)](https://github.com/Jerry-TZ/Talos/actions/workflows/test.yml) -**一个你能完整读完的编程 agent。** 5197 行 Python,326 条离线判据,一份不糊弄人的安全说明。 +**一个你能完整读完的编程 agent。** 5309 行 Python,336 条离线判据,一份不糊弄人的安全说明。 Talos 终端界面:动盘之前先弹确认框 @@ -19,9 +19,9 @@ | 文件 | 行数 | 职责 | |---|---|---| -| `agent.py` | 4112 | 循环 + 工具 + 权限门 + 自学习 | +| `agent.py` | 4207 | 循环 + 工具 + 权限门 + 自学习 | | `console_ui.py` | 214 | 终端界面(可整体替换) | -| `recall.py` | 565 | 联想记忆:扩散激活检索 | +| `recall.py` | 582 | 联想记忆:扩散激活检索 | | `session.py` | 306 | 会话持久化(想换 SQLite 只改这个) | 行数自己数,别信我:`wc -l agent.py console_ui.py recall.py session.py` @@ -33,7 +33,7 @@ 极简 agent 赛道很挤,有人用 Zig 做到 678KB 二进制。**Talos 不比谁小,它比谁都好读。** - **能读完** — 四个文件,注释解释的是*为什么*,不是*是什么*。几乎每条防御旁边都写着它挡的那次真实翻车。 -- **能验证** — 326 个测试,**离线、免 API key、几秒跑完**,CI 在 Linux/Windows × Python 3.10/3.13 上都跑。clone 下来立刻知道它没坏。 +- **能验证** — 336 个测试,**离线、免 API key、几秒跑完**,CI 在 Linux/Windows × Python 3.10/3.13 上都跑。clone 下来立刻知道它没坏。 - **不吹牛** — [`SECURITY.md`](SECURITY.md) 明写 `create_tool` 就是进程内 RCE、正则黑名单只是减速带。**没有沙箱就是没有沙箱** —— 真要隔离,[三条现成方案](SECURITY.md#真要隔离怎么办)按代价从低到高列在那儿。 - **有考卷** — [`EXAM.md`](EXAM.md) / [`EXAM2.md`](EXAM2.md) 是两份可复现的能力测试,带标准答案和作弊检测(比如逐个核验 arXiv ID 真伪,防止编造引用)。记录的是"我怎么验证它真的有用",不是功能列表。 - **有实测** — [`FINDINGS.md`](FINDINGS.md) 记了二十二个真实任务量出来的东西:哪条提示词生效、哪条从头到尾没生效、六次翻车、检索改动的前后数字,以及**两个被数据否掉的自己的方案**。样本小,局限写在最前面。 @@ -200,7 +200,7 @@ Ctrl+C 停下当前这轮 —— 做过的都留着,可以直接说 ```bash .venv\Scripts\python.exe -m pip install -r requirements-dev.txt -.venv\Scripts\python.exe -m pytest tests/ -q # 326 条,约 8 秒,不联网、不需要 key +.venv\Scripts\python.exe -m pytest tests/ -q # 336 条,约 8 秒,不联网、不需要 key python agent.py --selfcheck # 免依赖的冒烟检查 ``` diff --git a/SECURITY.md b/SECURITY.md index 9e6f46f..325c0f6 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -147,7 +147,7 @@ run_bash 能跑任何 shell 命令;一个恶意工具里的 `while True` 会挂 **它也不是完整的审计日志。** 两个缺口: -1. 为了省 token,较旧的大工具输出会被替换成 `[已省略工具输出,N 字符]` 后覆盖保存 —— 会话文件里的历史结果**不保证是原始内容**,`/view` 看到的是打过桩的版本。 +1. 为了省 token,较旧的大工具输出会被替换成 `[已省略工具输出,N 字符]` 后覆盖保存;较旧调用里的大块参数(比如 `write_file` 写进去的整份文件内容)同样被换成 `[已省略 N 字符 …]` —— 会话文件里的历史**不保证是原始内容**,`/view` 看到的是打过桩的版本。文件本身的旧版本在回收站 `.talos/trash/` 里找(有大小和数量上限,见上),不在会话文件里。 2. **复盘那一整段不进会话文件。** `reflect()` 在 `messages` 的**副本**上跑(为了不让复盘提示词污染后续对话),而 `sess.save()` 存的是原列表。于是复盘读了哪条技能、改了哪个文件、写了什么,`/view` 一行都看不到 —— 尽管那一段是会**写文件**的。想看只能看终端。 想留完整记录,自己另存。 diff --git a/agent.py b/agent.py index 745abac..375b69f 100644 --- a/agent.py +++ b/agent.py @@ -2410,11 +2410,35 @@ def _usage(resp): cached = (getattr(d, "cached_tokens", 0) or 0) if d is not None else 0 return (getattr(u, "prompt_tokens", 0) or 0, getattr(u, "completion_tokens", 0) or 0, cached) +CHAT_TRIES = 4 # 一次调用最多试几趟(含第一趟) +RETRY_AFTER_MAX = 60 # 服务器让等的秒数封顶 —— 别让一个离谱的头把人晾一小时 +# 服务器**答了话**、而且话是「等一下再来」的那几个状态码。按状态码认,不按文本: +# 原来的关键词表里有 502、503 却没有 500 —— 500 试一次就放弃(FINDINGS 七十八节那个 +# 任务三趟死法是 500、500、429)。 +_RETRY_STATUS = {408, 429, 500, 502, 503, 504} +# **钱没了不是对面忙。** 这几种 429 等多久都不会好,重试只是把报错往后推, +# 还一路打「模型繁忙」让人以为在等。只认明确的写法,**不认 "quota" 这个词**: +# Gemini 免费档的每分钟限流文本也是 "You exceeded your current quota",那是真限流。 +_NO_BALANCE = ("insufficient_quota", "insufficient balance", "insufficient_balance", + "余额不足", "exceeded_current_quota_error", "欠费") + +def _retry_after(e): + """服务器在响应头里说的等待秒数;没说、或者说的是 HTTP 日期,返回 None。""" + headers = getattr(getattr(e, "response", None), "headers", None) or {} + try: + ms = headers.get("retry-after-ms") + if ms is not None: + return float(ms) / 1000 + s = headers.get("retry-after") + return float(s) if s is not None else None + except (TypeError, ValueError): + return None + def _chat(client, **kwargs): """Call the model, retrying briefly on rate-limit / 'busy' / transient errors. Free tiers (esp. glm-4.7-flash) get congested — '当前模型用户多' is just a busy signal.""" timeouts = 0 - for attempt in range(3): + for attempt in range(CHAT_TRIES): t0 = time.time() try: resp = client.chat.completions.create(**kwargs) @@ -2432,14 +2456,21 @@ def _chat(client, **kwargs): # 掉了一次网,`APIConnectionError("Connection error.")` —— 这串里一个既有关键词 # 都不含,于是**不重试、直接抛**,整轮的复盘就没了(工作本身已经存盘,丢的是学习)。 # 断网比 429 更该重试:429 是对面忙,掉线常常下一秒就好。 - transient = (any(k in s for k in ("429", "rate limit", "ratelimit", "timeout", - "timed out", "overload", "too many", "busy", "503", "502", - "connection", "并发", "繁忙")) - or "用户多" in str(e)) + # 文本那张表留着:没有状态码的异常(掉线、SDK 自己的超时)只能靠它认, + # 而「当前模型用户多」这类说法也不保证配着 429 来。状态码是加上去的,不是换掉。 + status = getattr(e, "status_code", None) + transient = ((status in _RETRY_STATUS + or any(k in s for k in ("429", "rate limit", "ratelimit", "timeout", + "timed out", "overload", "too many", "busy", "503", "502", + "connection", "并发", "繁忙")) + or "用户多" in str(e)) + and not any(k in s for k in _NO_BALANCE)) # 但超时跟"忙"不是一回事。忙是等一下就好;超时说明这次调用本来就要跑过 # CHAT_TIMEOUT,重试三遍就是三遍注定失败 —— 默认 300s 下,一次失败要 15 分钟 # 才告诉你。推理模型上这是常态,不是意外。只给它一次机会。 - timeouts += ("timed out" in s or "timeout" in s) + # **有状态码就不是这一种**:504 的文本是 "Gateway Timeout",原来被记成读超时、 + # 只多给一次机会 —— 可它是网关替上游答的话,这边并没有等满 CHAT_TIMEOUT。 + timeouts += status is None and ("timed out" in s or "timeout" in s) # **连不上 ≠ 对面忙,不重试。** 上面那条「掉线常常下一秒就通」说的是 # `APIConnectionError`(连接被拒/被断,几毫秒就回来);连接**超时**是另一回事: # 每次要烧满 `CONNECT_TIMEOUT × 主机的地址条数`,重一次就是再等一遍。 @@ -2462,10 +2493,16 @@ def _chat(client, **kwargs): f" 注意 httpx 的连接超时是**按地址**算的:多 A 记录的主机会乘上去," f"所以真实等待≈{CONNECT_TIMEOUT:g}s × 地址条数。") raise - if attempt < 2 and transient and timeouts < 2: + if attempt < CHAT_TRIES - 1 and transient and timeouts < 2: + # 服务器说了等多久就等多久;没说就 2、4、8 秒。原来是 1、2 秒, + # 免费档的 429 三趟在三秒里就烧完了。 + wait = _retry_after(e) + wait = min(wait, RETRY_AFTER_MAX) if wait is not None else 2 ** (attempt + 1) if ui is not None: - ui.note(f"模型繁忙,{2 ** attempt}s 后重试 ({attempt + 1}/2)…") - time.sleep(2 ** attempt) + # 带上状态码:429(限流,等等就好)和 500(对面出错)人该分得清 + ui.note(f"模型繁忙{f'(HTTP {status})' if status else ''}," + f"{wait:g}s 后重试 ({attempt + 1}/{CHAT_TRIES - 1})…") + time.sleep(wait) continue raise @@ -3342,8 +3379,21 @@ def consolidate(client, model: str, state: dict) -> str: # ── context compression (仿 Claude Code 的 auto-compact) ─────────────────────── +def _msg_chars(m: dict) -> int: + """一条消息**真正发出去**的量:正文 + 它发起的工具调用参数。 + + 原来只数 `content`。而 `write_file` 写的整份文件躺在 `tool_calls[].arguments` 里 —— + 实测 6 次 write_file、真实 12 万字符的历史,这里数出 73。压缩闸看不见它们, + 写得越多的任务越是每步整份重发,直到 `400 Prompt exceeds max length`。""" + n = len(str(m.get("content") or "")) + for c in m.get("tool_calls") or (): + fn = c.get("function") if isinstance(c, dict) else getattr(c, "function", None) + args = fn.get("arguments") if isinstance(fn, dict) else getattr(fn, "arguments", "") + n += len(str(args or "")) + return n + def _ctx_chars(messages: list) -> int: - return sum(len(str(m.get("content") or "")) for m in messages) + return sum(_msg_chars(m) for m in messages) COMPACT_KEEP = 8 # 压缩之后原样留在末尾的消息数,见 _tail_start COMPACT_TAIL_CHARS = COMPACT_AT // 3 # 尾部最多占预算的三分之一 —— 光按条数留会压不下去 @@ -3362,9 +3412,11 @@ def _tail_start(messages: list, keep: int) -> int: # 长回答就能让「压缩完还是超阈值」,于是下一步立刻再压一次:实测连着两次 # 「33 条 → 10 条」「11 条 → 7 条」,每一次都是一次全量缓存作废加一次额外的模型调用。 # 这是我加尾部保留时没算到的代价 —— 保留是按条数写的,而预算一直是按字符算的。 + # 按 `_msg_chars` 数,不按 `content`:尾部躺着几份 write_file 参数时,只数正文的话 + # 8 条全留,压完照样超阈值 —— 跟上面那句「两个单位」是同一个根因的另一半。 spent, j = 0, len(messages) while j > i: - spent += len(str(messages[j - 1].get("content") or "")) + spent += _msg_chars(messages[j - 1]) if spent > COMPACT_TAIL_CHARS: break j -= 1 @@ -3408,6 +3460,16 @@ def _human_asks(messages: list) -> list: if m.get("role") == "user" and isinstance(m.get("content"), str) and not m["content"].startswith(_HARNESS_SAYS)] +def _adopt_history(state: dict, messages: list) -> None: + """换了一段历史,就重认一遍哪些是人说的。 + + **三条路换历史**:启动时 `--resume`/`--continue`、REPL 里 `/resume`、`/delete` 掉 + 当前会话。上一版只有启动那条认了,另外两条切过去之后 `asks` 还是上一个会话的对象 —— + 续上的会话一压缩,人的原话就被转述掉;`asked` 也还是旧的,「点名要过」拿上一个 + 会话的话去比这个会话的删除命令。""" + state["asks"] = _human_asks(messages) + state["asked"] = "\n".join(m["content"] for m in state["asks"]) + def maybe_compact(client, model: str, messages: list, force: bool = False, asks=()) -> list: """History got long? Summarize the head, keep the tail verbatim. Returns the new list. @@ -3490,8 +3552,9 @@ def maybe_compact(client, model: str, messages: list, force: bool = False, asks= return kept + [{"role": "user", "content": "【早前对话的压缩摘要】\n" + summary}] + tail def _prune_old_tool_results(messages: list, keep: int = 8) -> None: - """Stub OLD, bulky tool outputs in place so they stop getting resent every step (token saver). - Keeps the last `keep` messages untouched. Note: this rewrites saved history (/view shows stubs).""" + """Stub OLD, bulky tool outputs — and the bulky arguments of old tool calls — in place so they + stop getting resent every step (token saver). Keeps the last `keep` messages untouched. + Note: this rewrites saved history (/view shows stubs).""" # 「需要就重新读」得说清读的是什么。只报字符数的话,模型此时已经不知道当初读的是哪个 # 文件了 —— 那是一句无法执行的建议。出处不用另存:发起这次调用的 tool_call 里就写着。 by_id = {} @@ -3526,6 +3589,37 @@ def _prune_old_tool_results(messages: list, keep: int = 8) -> None: m["content"] = (f"[已省略 {what} 的输出,{len(m['content'])} 字符 — 需要就重新读]" if what else f"[已省略工具输出,{len(m['content'])} 字符 — 需要就重新读]") + # **调用那一侧也有大块:模型自己写的参数。** write_file 的整份文件躺在 + # `tool_calls[].arguments` 里,上面那段只剪 `role == "tool"`,于是写过的每个文件 + # 永远留在历史里、每步重发。同一条规则(窗口外、超过门槛),剪法有两处不同: + # 剪完还得是**合法 JSON**(有的 provider 回放历史时会解析它),`path` 这类短字段留着。 + # spawn_subagent 不剪,理由同上:那是派活的原话,重来就是重新派一次。 + for m in old: + for c in (m.get("tool_calls") or []) if m.get("role") == "assistant" else (): + fn = c.get("function") if isinstance(c, dict) else None + raw = fn.get("arguments") if isinstance(fn, dict) else None + if (isinstance(raw, str) and len(raw) > 600 + and fn.get("name") != "spawn_subagent"): + fn["arguments"] = _stub_args(raw) + +def _stub_args(raw: str) -> str: + """把一次旧调用的参数里的大块字段换成一句话,其余原样,仍是合法 JSON。 + + **只说确定的事**:这是旧调用的参数,结果在紧跟着的那条工具消息里。 + 不说「已经写进去了」—— 那次调用可能被拒了、可能报错了。""" + try: + a = json.loads(raw) + except (json.JSONDecodeError, TypeError): + a = None + if not isinstance(a, dict): # 截断的半截 JSON(输出撞上 max_tokens 就是这样) + return json.dumps({"已省略": f"{len(raw)} 字符的参数(旧调用,结果见紧跟着的工具消息)"}, + ensure_ascii=False) + path = a.get("path") if isinstance(a.get("path"), str) else "" + for k, v in a.items(): + if isinstance(v, str) and len(v) > 600: + a[k] = (f"[已省略 {len(v)} 字符 — 旧调用的参数,结果见紧跟着的工具消息" + + (f";文件现状用 read_file {path} 看" if path else "") + "]") + return json.dumps(a, ensure_ascii=False) _CORRECTION_MARKERS = ("不对", "错了", "搞错", "不是这样", "不应该", "不该", "重来", "别这样", "不是这个", "写错", "wrong", "incorrect", "not what", "should have", "instead") @@ -3773,8 +3867,7 @@ def repl(resume=None) -> None: messages = [] state = {"mode": "default", "allow": set(), "view": "normal"} # ← permission + display state if messages: # 续上来的历史:身份过不了磁盘,只能按内容认回来 - state["asks"] = _human_asks(messages) - state["asked"] = "\n".join(m["content"] for m in state["asks"]) + _adopt_history(state, messages) ui.banner(state["mode"], PROVIDER, model) ui.note(f"会话 {sess.sid}" + (f" · 续上 {len(messages)} 条消息" if messages else " · 存于 .talos/sessions/")) @@ -3920,6 +4013,7 @@ def repl(resume=None) -> None: if sid: sess = S.open_session(sid) messages[:] = sess.load() + _adopt_history(state, messages) # 跟启动时 --resume 同一件事 ui.note(f"已切到会话 {sess.sid} · {len(messages)} 条消息,可继续编辑") else: ui.note("没找到该会话 — 用 /history 看编号") @@ -3937,6 +4031,7 @@ def repl(resume=None) -> None: if sid == sess.sid: # 删的是当前会话 → 开一个新空会话 sess = S.Session.new() messages[:] = [] + _adopt_history(state, messages) ui.note(f"已删除 {sid}") else: ui.note(f"⚠️ 没删掉 {sid} —— 文件还在 .talos/sessions/,会话没动") diff --git a/recall.py b/recall.py index dfb1d47..1f4810a 100644 --- a/recall.py +++ b/recall.py @@ -265,12 +265,29 @@ def _drop_dont_use(text: str) -> str: return _DONT_USE.sub("", text) def _edges(nodes: list) -> dict: + """边权 `shared / max(|A|,|B|)`,至少共享 `EDGE_MIN` 个关键词才连。 + + **只数真共享关键词的那几对。** 原来两两比较,每轮顶层请求、复盘、子 agent 各建一遍, + 而节点数跟着会话数涨(每个会话的第一句是一个「往事」)。实测每次 `recall()`: + 500 个会话 0.23s,1500 个 1.3s,3000 个 4.6s。倒排之后在真实中文语料 + (FINDINGS 的句子)上快 5~8 倍,图一模一样。 + + **每行的邻居按下标升序排**,跟两两比较那一版逐项相同:`_activate` 按这个顺序 + 累加浮点数,顺序变了末位就可能变,同分的两条在排名里就可能换位。""" + idx = {} + for i, n in enumerate(nodes): + for k in n["kw"]: + idx.setdefault(k, []).append(i) E = {} - for i in range(len(nodes)): - for j in range(i + 1, len(nodes)): - shared = len(nodes[i]["kw"] & nodes[j]["kw"]) - if shared >= EDGE_MIN: - w = shared / max(len(nodes[i]["kw"]), len(nodes[j]["kw"])) + for i, n in enumerate(nodes): + shared = {} + for k in n["kw"]: + for j in idx[k]: + if j > i: + shared[j] = shared.get(j, 0) + 1 + for j in sorted(shared): + if shared[j] >= EDGE_MIN: + w = shared[j] / max(len(n["kw"]), len(nodes[j]["kw"])) E.setdefault(i, {})[j] = w E.setdefault(j, {})[i] = w return E diff --git a/tests/test_goal.py b/tests/test_goal.py index 7f2ad3e..d7adbbf 100644 --- a/tests/test_goal.py +++ b/tests/test_goal.py @@ -28,6 +28,7 @@ def _judge_reads(path, cid="j1"): def _run(monkeypatch, script, goal="产物正确", view="quiet", **extra): import agent as A monkeypatch.setattr(A, "ui", _ui()) + monkeypatch.setattr(A.time, "sleep", lambda _s: None) # 判断器崩的那几条会走重试退避 state = {"mode": "bypass", "allow": set(), "view": view, "goal": goal, **extra} messages = [{"role": "user", "content": "干活"}] out = A.agent_turn(_Client(script), "m", messages, state, top=True) @@ -516,6 +517,7 @@ def test_a_crash_after_a_block_says_the_deliverable_is_known_wrong(ws, monkeypat notes = [] monkeypatch.setattr(console_ui, "note", lambda s, *a, **k: notes.append(s)) monkeypatch.setattr(console_ui, "error", lambda *a, **k: None) + monkeypatch.setattr(A.time, "sleep", lambda _s: None) # 那个 429 会先走一次重试退避 monkeypatch.chdir(ws) monkeypatch.setenv("TALOS_GOAL", "report.md 里 8 个数都对") client = _Client([ diff --git a/tests/test_loop.py b/tests/test_loop.py index ea64443..753b3f2 100644 --- a/tests/test_loop.py +++ b/tests/test_loop.py @@ -1024,10 +1024,11 @@ def test_a_subagent_cannot_reset_the_parents_repeat_counter(ws, monkeypatch): if m.get("role") == "tool" and "第 3 次" in str(m.get("content"))] assert hits, "子agent 插了一脚,父轮的打转计数就被清零了" -def test_a_timeout_is_retried_like_any_other_busy_signal(): +def test_a_timeout_is_retried_like_any_other_busy_signal(monkeypatch): """SDK 抛的是 "Request timed out.",里面没有 "timeout" 这个词 —— 刚给客户端设完超时 才发现,超时本身正好落在重试判据之外,等于设了个「一次就放弃」的开关。""" import agent as A + monkeypatch.setattr(A.time, "sleep", lambda _s: None) assert A._chat(_Client([Exception("Request timed out."), "OK"])) == "OK" def test_only_a_slow_call_reports_how_long_it_took(monkeypatch): @@ -2309,13 +2310,18 @@ def test_resuming_a_session_does_not_lose_track_of_what_the_human_asked(ws): assert got == [human, "再补一句:顺便看看 signals.py"], f"认错了:{got}" # 认出来了还得**接上** —— 上一版就是「函数有了、续会话那条路没调」。 + # 走 `_adopt_history`:repl 里换历史的路不止启动这一条,每条路是否都接上了, + # 由 `test_switching_sessions_inside_the_repl_re_reads_who_said_what` 真跑一遍来钉。 src = _io.open(_os.path.join(_os.path.dirname(_os.path.dirname( _os.path.abspath(__file__))), "agent.py"), encoding="utf-8").read() fn = next(n for n in ast.walk(ast.parse(src)) if isinstance(n, ast.FunctionDef) and n.name == "repl") called = {n.func.id for n in ast.walk(fn) if isinstance(n, ast.Call) and isinstance(n.func, ast.Name)} - assert "_human_asks" in called, "repl 续会话那条路没把 asks 重建回来" + assert "_adopt_history" in called, "repl 续会话那条路没把 asks 重建回来" + state = {} + A._adopt_history(state, loaded) + assert [m["content"] for m in state["asks"]] == got and human in state["asked"] def test_piping_a_script_run_into_findstr_is_not_reading_that_script(ws): @@ -2411,3 +2417,273 @@ def test_no_session_save_can_take_the_repl_down_with_it(): and c.func.attr == "save" and c.lineno not in guarded] assert not naked, (f"agent.py 第 {naked} 行的落盘没人接着 —— 一次 PermissionError " "就把整个 REPL 带走了。包进 try,或者走 `_save_quietly`。") + + +# ── 上下文预算里漏掉的那一半:模型写进 tool_calls 的参数 ───────────────────────── +def _sent_chars(messages): + """一段历史**真正发出去**的量:正文 + 每条 assistant 的 tool_calls 参数。 + + **故意不调 `_ctx_chars`** —— 被量的就是它,拿它当尺子等于让它给自己打分。""" + n = 0 + for m in messages: + n += len(str(m.get("content") or "")) + for c in m.get("tool_calls") or (): + fn = c["function"] if isinstance(c, dict) else c.function + n += len(str((fn.get("arguments") if isinstance(fn, dict) else fn.arguments) or "")) + return n + + +def _writes(n, size, start=0): + """n 次 write_file,每次写 size 字符 —— 模型一个接一个写大文件的那种历史。""" + import json + out = [] + for i in range(start, start + n): + body = f"# m{i}\n" + "x = 1\n" * (size // 6) + out.append({"role": "assistant", "content": "", "tool_calls": [ + {"id": f"w{i}", "type": "function", + "function": {"name": "write_file", + "arguments": json.dumps({"path": f"m{i}.py", "content": body})}}]}) + out.append({"role": "tool", "tool_call_id": f"w{i}", "content": f"wrote m{i}.py"}) + return out + + +def test_what_the_model_wrote_counts_toward_the_context_budget(): + """`write_file` 的内容躺在 assistant 消息的 `tool_calls[].arguments` 里,不在 `content` 里。 + + `_ctx_chars` 原来只数 `content`。实测:6 次 write_file、每次 1.8 万字符, + 真实历史 123,922 字符,它报 **73**。于是压缩永远不触发,而写得越多的任务 + —— 恰恰是最贵的那种 —— 每一步都把所有写过的文件原样重发一遍,直到 + `400 Prompt exceeds max length`。尺子只量了一半,闸就只守了一半。""" + import agent as A + msgs = [{"role": "user", "content": "写 6 个模块"}] + _writes(6, 18000) + real = _sent_chars(msgs) + assert real > A.COMPACT_AT, "样本本身没超预算,这条判据测不出东西" + assert A._ctx_chars(msgs) >= real, ( + f"历史真实有 {real} 字符,`_ctx_chars` 只数出 {A._ctx_chars(msgs)} —— " + "tool_calls 里的参数没算进去,压缩闸看不见它们") + + +def test_old_tool_call_arguments_are_stubbed_like_old_tool_output(): + """旧的大块**参数**跟旧的大块**工具输出**是同一种东西:早就用过了,还在每步重发。 + + `_prune_old_tool_results` 只剪 `role == "tool"`,于是 write_file 写进去的整份文件 + 永远留在历史里。剪法跟工具输出同一条规则(窗口外、超过门槛),而且: + - 剪完的参数还得是**合法 JSON** —— 有的 provider 回放历史时会解析它; + - `path` 留着 —— 省略提示要能照着走(「要看现状就 read_file 它」); + - `spawn_subagent` 不剪 —— 跟它的返回不剪是同一个理由,那是派活的原话。""" + import json + import agent as A + task = "读一下季报,把要点列出来" + "要点" * 400 + msgs = [{"role": "user", "content": "写模块"}, + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "s", "type": "function", + "function": {"name": "spawn_subagent", "arguments": json.dumps({"task": task})}}]}, + {"role": "tool", "tool_call_id": "s", "content": "子agent 的结论"}] + msgs += _writes(10, 5000) + A._prune_old_tool_results(msgs) + + assert json.loads(msgs[1]["tool_calls"][0]["function"]["arguments"])["task"] == task, \ + "派子 agent 的原话被剪了 —— 那一份重来就是重新派一次活" + old = [m for m in msgs[3:-8] if m["role"] == "assistant"] + recent = [m for m in msgs[-8:] if m["role"] == "assistant"] + assert old and recent, "样本切分不对" + for m in old: + a = json.loads(m["tool_calls"][0]["function"]["arguments"]) # 剪完还得能解析 + assert a["path"].startswith("m"), f"path 被一起剪没了:{a}" + assert a["content"].startswith("[已省略"), f"窗口外的大块参数没剪:{str(a)[:80]}" + assert a["path"] in a["content"], f"省略提示没说去哪儿看:{a['content']}" + for m in recent: + a = json.loads(m["tool_calls"][0]["function"]["arguments"]) + assert len(a["content"]) >= 5000, "窗口里的也剪了 —— 正在用的那几条不能动" + assert _sent_chars(msgs) < A.COMPACT_AT, ( + f"剪完还有 {_sent_chars(msgs)} 字符 —— 这一级便宜手段没起作用") + + +def test_a_turn_that_writes_big_files_does_not_resend_all_of_them(ws, monkeypatch): + """上面两条是函数级的;这条钉**循环真的在用它们** —— 每一次请求都在预算里。 + + 脚本化模型连写 12 个 6000 字符的文件。修之前最后一次请求带着全部 12 份 + (约 7.2 万字符);修之后窗口外的参数每步被剪掉,每次请求都压在 `COMPACT_AT` 以下, + 而且不用花钱叫模型压缩(这份脚本里没有给压缩准备回答,叫了就会错位)。""" + import agent as A + monkeypatch.setattr(A, "ui", _ui()) + monkeypatch.setattr(A, "run_tool", lambda name, args: (f"wrote {args.get('path')}", False)) + sizes = [] + + class _Rec(_Client): + def _c(self, **k): + sizes.append(_sent_chars(k["messages"][1:])) # 在发出那一刻量 —— 之后还会被剪 + return super()._c(**k) + + script = [] + for m in _writes(12, 6000)[::2]: + c = m["tool_calls"][0] + script.append(_msg(tool_calls=[_tc("write_file", c["function"]["arguments"], cid=c["id"])])) + script.append(_msg(content="done")) + msgs = [{"role": "user", "content": "写 12 个模块"}] + A.agent_turn(_Rec(script), "m", msgs, {"mode": "bypass", "allow": set()}) + assert len(sizes) == 13, f"脚本没按预期走完:{len(sizes)} 次请求" + assert max(sizes) < A.COMPACT_AT, ( + f"有一次请求带了 {max(sizes)} 字符(预算 {A.COMPACT_AT})—— " + "写过的文件在每一步被原样重发") + + +def test_compaction_budgets_the_tail_by_what_is_actually_sent(ws, monkeypatch): + """尾部的字符预算也得按**真正发出去的量**算。 + + `_tail_start` 原来只数 `content`:尾部 8 条里躺着 4 份 9000 字符的 write_file 参数, + 它数出来是几十个字符,于是 8 条全留 —— 压完还有 3.6 万字符,超过阈值, + 下一步立刻再压一次(`test_compaction_must_actually_get_under_the_threshold` + 记过这个形状,当时的根因是「单位不一致」,这是同一个根因的另一半)。""" + import agent as A + monkeypatch.setattr(A, "ui", _ui()) + monkeypatch.setattr(A, "_chat", lambda c, **kw: _msg(content="简报")) + ask = {"role": "user", "content": "写 9 个模块"} + out = A.maybe_compact(None, "m", [ask] + _writes(9, 9000), force=True, asks=[ask]) + assert _sent_chars(out) < A.COMPACT_AT, ( + f"压完还有 {_sent_chars(out)} 字符(阈值 {A.COMPACT_AT})—— 下一步会再压一次") + seen = set() + for m in out: # 缩尾巴不许缩出落单的工具结果 + for c in m.get("tool_calls") or (): + seen.add(c["id"]) + if m.get("role") == "tool": + assert m["tool_call_id"] in seen, f"落单的工具结果 {m['tool_call_id']}" + assert any(m.get("role") == "tool" for m in out), "尾巴全丢了" + + +# ── 重试按状态码认,不按报错里恰好有哪几个字 ─────────────────────────────────── +class _StatusError(Exception): + """openai `APIStatusError` 的形状:`.status_code`、`.response.headers`, + 文本是 `Error code: 500 - {...}`。离线判据不 import openai,照着它的形状造一个 + (真对象的这三处在本机用 openai 2.x 核对过)。""" + def __init__(self, code, body, headers=None): + super().__init__(f"Error code: {code} - {body}") + self.status_code = code + self.body = body + self.response = types.SimpleNamespace(status_code=code, headers=headers or {}) + + +def test_a_server_error_is_retried_and_a_gateway_timeout_is_not_a_read_timeout(monkeypatch): + """重试判据原来是在报错文本里找关键词,表里有 502、503,**没有 500**。 + + 实测三处:500 试 1 次就放弃;504 的文本是 `Gateway Timeout`,撞上 "timeout" 被记成 + **读超时**,只多给一次机会;真正的客户端错(400)照旧不该重试。FINDINGS 七十八节 + 那个任务连排三趟,死法是 500、500、429 —— 前两趟一次重试都没有。 + + 状态码是服务器**答了话**的证据:504 是网关替上游说「没等到」,不是这边等满了 + `CHAT_TIMEOUT`,所以不该吃「超时只给一次机会」那条规矩。""" + import agent as A + monkeypatch.setattr(A.time, "sleep", lambda _s: None) + monkeypatch.setattr(A, "ui", _ui()) + e500 = lambda: _StatusError(500, {"error": {"code": "500", "message": "Internal Error"}}) + assert A._chat(_Client([e500(), "OK"])) == "OK", "500 没重试" + e504 = lambda: _StatusError(504, {"error": {"message": "Gateway Timeout"}}) + assert A._chat(_Client([e504(), e504(), "OK"])) == "OK", \ + "504 被当成读超时了 —— 第二次就放弃" + bad = _Client([_StatusError(400, {"error": {"message": "invalid request"}}), "OK"]) + with pytest.raises(Exception, match="400"): + A._chat(bad) + assert bad.i == 1, "400 是请求本身错了,重试只是把同一个错再拿一遍" + + +def test_an_exhausted_balance_is_not_retried(monkeypatch): + """**钱没了不是对面忙。** 429 有两种:限流(等一下就好)和额度耗尽(等多久都不会好)。 + 原来两种一样重试,还打「模型繁忙,Ns 后重试」—— 人照着那句话等,等来的还是同一个错。 + 退避拉长之后这一档更贵,所以要认出来直接报。 + + **反例必须一起钉**:Gemini 免费档的**每分钟限流**文本也是 "You exceeded your current + quota",那是真的限流,要重试。所以判据只认明确是「余额/额度用完」的那几种写法, + 不认 "quota" 这个词。""" + import agent as A + sleeps = [] + monkeypatch.setattr(A.time, "sleep", sleeps.append) + monkeypatch.setattr(A, "ui", _ui()) + for body in ({"error": {"code": "1113", "message": "余额不足或无可用资源包,请充值。"}}, # glm + {"error": {"code": "insufficient_quota", # openai + "message": "You exceeded your current quota, please check your plan"}}): + c = _Client([_StatusError(429, body), "OK"]) + with pytest.raises(Exception, match="429"): + A._chat(c) + assert c.i == 1 and sleeps == [], f"额度耗尽还在重试:{body}" + gemini = _StatusError(429, [{"error": {"code": 429, "status": "RESOURCE_EXHAUSTED", + "message": "You exceeded your current quota, please " + "check your plan and billing details."}}]) + assert A._chat(_Client([gemini, "OK"])) == "OK", \ + "Gemini 的每分钟限流被当成额度耗尽了 —— 它等一下就好" + + +def test_retry_after_is_honoured_and_capped(monkeypatch): + """服务器说了等多久,就等多久 —— 原来一律 1s、2s,对免费档的 429 太短, + 三次在三秒里烧完。上限是为了别让一个离谱的头把人晾一小时:超过就按上限等, + 等完再试一次,还不行照旧报错。""" + import agent as A + sleeps = [] + monkeypatch.setattr(A.time, "sleep", sleeps.append) + monkeypatch.setattr(A, "ui", _ui()) + busy = {"error": {"message": "rate limit reached"}} + for headers, want in (({"retry-after": "7"}, 7), ({"retry-after-ms": "1500"}, 1.5), + ({"retry-after": "86400"}, A.RETRY_AFTER_MAX)): + sleeps.clear() + assert A._chat(_Client([_StatusError(429, busy, headers), "OK"])) == "OK" + assert sleeps == [want], f"{headers} → 实际等了 {sleeps}" + # 头是 HTTP 日期(或者根本没有)就退回指数退避,不许因为解析不了而崩 + sleeps.clear() + date = {"retry-after": "Wed, 21 Oct 2026 07:28:00 GMT"} + assert A._chat(_Client([_StatusError(503, busy, date), "OK"])) == "OK" + assert len(sleeps) == 1 and sleeps[0] > 0 + + +def test_switching_sessions_inside_the_repl_re_reads_who_said_what(ws, monkeypatch): + """`--resume` 启动那条路会按内容把人的原话认回来(`_human_asks`); + REPL 里敲 `/resume` 那条路**没有**,`/delete` 掉当前会话那条也没有。 + + 于是切过去之后 `state["asks"]` 还是**上一个会话**的对象:续上的会话一压缩, + 人的原话被转述掉 —— 正是 `_human_asks` 当初要修的那件事。`state["asked"]` 也是旧的, + 「你在请求里点名要过它」那句提示拿上一个会话的话去比这个会话的删除命令。 + + 上一版的判据只查「repl 里调用过 `_human_asks`」—— 启动那一处就满足了, + 另外两条路一直漏着。这里改成真的把 REPL 跑一遍。""" + import json + import os + import sys + import agent as A + import session as S + b_human = "把 report.md 的标题改成季度总结" + os.makedirs(S.SESS_DIR, exist_ok=True) + with open(os.path.join(S.SESS_DIR, "20260101-000000__b.jsonl"), "w", encoding="utf-8") as f: + for m in ({"role": "user", "content": b_human}, {"role": "assistant", "content": "好"}): + f.write(json.dumps(m, ensure_ascii=False) + "\n") + + tasks = iter(["先删掉 alpha.md", "/resume 20260101-000000", "继续", + "/delete 20260101-000000", "新的活"]) + + def read_task(_mode): + try: + return next(tasks) + except StopIteration: + raise EOFError + n = lambda *a, **k: None + monkeypatch.setitem(sys.modules, "console_ui", types.SimpleNamespace( + read_task=read_task, banner=n, note=n, answer=n, error=n, + ask_yes=lambda *a: True, thinking=lambda: contextlib.nullcontext())) + monkeypatch.setattr(A, "ui", None) # repl 会 `global ui` 换掉它,这行负责还原 + monkeypatch.setattr(A, "make_client", lambda: (None, "m")) + seen = [] + + def turn(client, model, messages, state, top=False): + seen.append((list(messages), list(state.get("asks", ())), state.get("asked", ""))) + return "ok" + monkeypatch.setattr(A, "agent_turn", turn) + A.repl() + assert len(seen) == 3, f"脚本没按预期走完:{len(seen)} 轮" + + msgs, asks, asked = seen[1] # /resume 之后的第一轮 + resumed = next(m for m in msgs if m.get("content") == b_human) + assert any(a is resumed for a in asks), \ + "续上的会话里人的原话没认回来 —— 一压缩就会被转述掉" + assert "alpha.md" not in asked, "「点名要过」还在拿上一个会话的话来比" + + msgs, asks, asked = seen[2] # /delete 当前会话、开新会话之后 + assert [a["content"] for a in asks] == ["新的活"], \ + f"删掉的会话里的原话还挂在 asks 上:{[a['content'] for a in asks]}" + assert b_human not in asked, "已删掉的会话的原话还在参与「点名」判断" diff --git a/tests/test_memory.py b/tests/test_memory.py index 850fc33..356b9b9 100644 --- a/tests/test_memory.py +++ b/tests/test_memory.py @@ -2035,3 +2035,65 @@ def test_a_command_line_switch_is_not_the_rm_command(ws): for cmd in ("rm -rf build", "sudo rm x", "del a.txt", "Remove-Item x", "python -c 'x' && rm y"): assert A._DESTRUCTIVE.search(cmd), f"这条真的在删东西:{cmd}" + + +# ── 建图:只比共享关键词的那几对 ────────────────────────────────────────────── +def _pairwise_edges(nodes): + """两两比较的参照实现 —— 就是 `recall._edges` 原来的样子,留在这儿当尺子。""" + import recall as R + E = {} + for i in range(len(nodes)): + for j in range(i + 1, len(nodes)): + shared = len(nodes[i]["kw"] & nodes[j]["kw"]) + if shared >= R.EDGE_MIN: + w = shared / max(len(nodes[i]["kw"]), len(nodes[j]["kw"])) + E.setdefault(i, {})[j] = w + E.setdefault(j, {})[i] = w + return E + + +def test_the_recall_graph_is_exactly_the_pairwise_one(): + """换建图算法不许换图。**连每一行里邻居的顺序都要一样**:`_activate` 按这个顺序 + 累加浮点数,顺序一变末位就可能变,同分的两条在排名里就可能换位 —— + 那就不是「只是变快了」,是悄悄改了检索结果。 + + 语料故意挑小词表:大量节点对恰好共享 1 个(不连边)、2 个(`EDGE_MIN`,刚好连边) + 或更多关键词,还有完全相同的集合 —— 边界都踩到。""" + import random + import recall as R + rnd = random.Random(0) + vocab = [f"k{i}" for i in range(40)] + nodes = [{"kw": set(rnd.sample(vocab, rnd.randint(1, 8)))} for _ in range(300)] + nodes += [{"kw": set(nodes[i]["kw"])} for i in range(0, 300, 37)] # 完全相同的集合 + want, got = _pairwise_edges(nodes), R._edges(nodes) + assert sum(len(r) for r in want.values()) > 1000, "样本太稀,边界没踩到" + assert set(got) == set(want), "连了边的节点集合变了" + diff = [i for i in want if list(got[i].items()) != list(want[i].items())] + assert not diff, f"{len(diff)} 行的邻居或顺序变了,例如第 {diff[0]} 行" + + +def test_building_the_recall_graph_does_not_compare_every_pair(): + """原来的 `_edges` 两两比较,每一轮顶层请求、复盘、子 agent 都各建一遍图。 + 节点数随会话数涨(每个会话的第一句是一个「往事」节点,`-p` 和 benchmark 每跑一次 + 就多一个)。实测每次 `recall()`:500 个会话 0.23s,1500 个 1.3s,3000 个 4.6s。 + + 这条拿**同一台机器上的两两比较**当基线,不写死秒数:稀疏语料(每个节点只跟 + 同组两三个节点共享关键词)上,只比共享关键词的那几对应该快出一个量级。 + 门槛放在 5 倍,实测是几十倍,留足 CI 机器的抖动。""" + import time + import recall as R + nodes = [] + for g in range(300): # 300 组 × 3 个节点,组内共享 3 个词 + common = {f"g{g}a", f"g{g}b", f"g{g}c"} + nodes += [{"kw": common | {f"n{g}_{k}_{j}" for j in range(12)}} for k in range(3)] + + def timed(fn): + t = time.perf_counter() + out = fn(nodes) + return time.perf_counter() - t, out + slow, want = timed(_pairwise_edges) # 抖动只会让基线更慢,测一次就够 + fast, got = min((timed(R._edges) for _ in range(3)), key=lambda r: r[0]) + assert got == want, "先得是同一张图,快才有意义" + assert fast * 5 < slow, ( + f"建图 {fast * 1000:.0f} ms,两两比较 {slow * 1000:.0f} ms —— " + "它还在比较每一对节点,会话一多每轮检索就要几秒")