-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmemory_report.py
More file actions
184 lines (162 loc) · 9.81 KB
/
Copy pathmemory_report.py
File metadata and controls
184 lines (162 loc) · 9.81 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
"""把记忆网络画成一张单文件 HTML —— 只读,不联网,不起服务器。
python memory_report.py → .talos/memory_report.html
**不画节点关系图。** 78 个节点 810 条边,画出来是一团毛线:好看,但看完还是不知道
该动哪里。真正能指导行动的是三件事,所以只画这三件:
1. 哪几条技能老被捞到,哪几条从没被想起 —— FINDINGS 第四节那条「泛用技能是负资产」
当初是手工对照发现的,这张表就是把那次手工活变成每次都能看一眼。
2. 分数的分布 —— 那几个阈值(THRESH / BODY_LEAD / BODY_FLOOR)全是手调的,
看见真实分数落在哪儿,比看常数有用。
3. 正文注入了几次 —— 每次至多 1200 字,是这套机制唯一真花钱的地方。
数据只来自本地 .talos/,里面有 memory.md 的片段(前 80 字)。**这份 HTML 别往外发。**
"""
import collections
import html
import json
import os
import statistics
import sys
HOME = os.path.dirname(os.path.abspath(__file__))
D = os.path.join(HOME, ".talos")
OUT = os.path.join(D, "memory_report.html")
def _rows(name):
p = os.path.join(D, name)
if not os.path.exists(p):
return []
out = []
for ln in open(p, encoding="utf-8"):
try:
out.append(json.loads(ln))
except ValueError:
continue # 坏行跳过 —— 报告不该被一行脏数据弄崩
return out
def _bar(frac, w=180):
n = max(1, round(frac * w))
return f'<div class=bar><i style="width:{n}px"></i></div>'
def build() -> str:
# 一次检索一行。老文件里还留着一批带 `out` 的结果行(那阵子两种行混在一个文件里,
# 数「几轮检索」直接翻倍)—— 结果行已经搬去 cache_trace,这里只是跳过历史遗留。
trace = [r for r in _rows("recall_trace.jsonl") if "out" not in r]
# 一次顶层请求一行:缓存命中 + 这一轮花了多少 + 注入了几条技能正文,全在同一行,
# 所以任何两列都能交叉 —— 上一版分在两个文件,交叉不了。
cache = _rows("cache_trace.jsonl")
outs = cache
hits = {}
hp = os.path.join(D, "recall_hits.json")
if os.path.exists(hp):
try:
hits = json.load(open(hp, encoding="utf-8"))
except ValueError:
pass
picked = collections.Counter() # 被捞到几次
bodied = collections.Counter() # 其中给了正文几次
scores = []
for r in trace:
for p in r.get("picked", []):
k = p.get("key", "")[:70]
picked[k] += 1
if p.get("body"):
bodied[k] += 1
if isinstance(p.get("score"), (int, float)):
scores.append(p["score"])
parts = ["<h1>记忆网络</h1>",
f"<p class=dim>{len(trace)} 轮检索 · {len(hits)} 条记忆有命中记录 · "
f"{sum(bodied.values())} 次注入过正文</p>"]
# ① 谁老被捞到,谁从没被想起 —— 前者可能是噪声,后者是死重
parts.append("<h2>① 被捞到的次数(和其中给了正文的次数)</h2>"
"<p class=dim>经常进前五、却从没当过第一名的,大概率是噪声技能 —— "
"它靠“什么都沾边”挤掉真正对题的那条。</p><table>")
for k, n in picked.most_common(18):
b = bodied[k]
parts.append(f"<tr><td class=k>{html.escape(k)}</td><td>{_bar(n / max(picked.values()))}</td>"
f"<td class=n>{n}</td><td class=n>{'正文 ' + str(b) if b else ''}</td></tr>")
parts.append("</table>")
dead = [k for k, v in hits.items()
if isinstance(v, list) and len(v) >= 2 and v[0] >= 8 and v[1] == 0]
if dead:
parts.append("<h2>② 见过很多次、一次都没被想起</h2>"
"<p class=dim>存了个寂寞。/forget 会提议删这些(只提议 Talos 自己写的)。</p><ul>")
for k in dead[:12]:
parts.append(f"<li>{html.escape(k[:90])} <span class=dim>(出现 {hits[k][0]} 次)</span></li>")
parts.append("</ul>")
# ③ 分数落在哪儿 —— 阈值是手调的,看真实分布比看常数有用
if scores:
buckets = collections.Counter(min(9, int(s * 10)) for s in scores)
parts.append("<h2>③ 激活分数的分布</h2>"
f"<p class=dim>中位数 {statistics.median(scores):.2f} · "
f"范围 {min(scores):.2f}~{max(scores):.2f} · 共 {len(scores)} 次命中。"
"阈值 THRESH/BODY_FLOOR 都是手调的,这张图是它们该不该动的唯一依据。</p><table>")
top = max(buckets.values())
for b in range(10):
parts.append(f"<tr><td class=k>{b/10:.1f} – {(b+1)/10:.1f}</td>"
f"<td>{_bar(buckets.get(b, 0) / top)}</td>"
f"<td class=n>{buckets.get(b, 0)}</td></tr>")
parts.append("</table>")
# ④ 注入正文到底有没有让这一轮更省 —— 复盘写完技能就结束,从来不知道哪条真管用。
# **按结果行自己带的 bodies 分组,不按 q 去跟检索行对。** 复盘用同一个 query 再检索
# 一遍、同一个问题问两次,按 q 分组就把没拿到正文的那些轮也算进「注入过」那一组。
groups = {True: [], False: []}
for out in outs:
if isinstance(out.get("calls"), int) and not out.get("capped"):
groups[bool(out.get("bodies"))].append(out["calls"])
parts.append("<h2>④ 注入了技能正文的轮,是不是更省</h2>")
if not any(groups.values()):
parts.append("<p class=dim>还没有数据 —— 结果是这一轮跑完才回填的,正常用几轮再来看。</p>")
else:
for label, want in (("注入过技能正文", True), ("只给了描述", False)):
g = groups[want]
if g:
parts.append(f"<p><b>{label}</b> n={len(g)} · 工具调用中位数 "
f"{statistics.median(g):.1f} · 范围 {min(g)}~{max(g)}</p>")
capped_n = sum(1 for o in outs if o.get("capped"))
parts.append(f"<p class=dim>差值小于 2 次、或任一组 n<8 时,别下结论 —— "
f"任务难度本身的方差比这大。撞了步数上限的 {capped_n} 轮不计入"
f"(那是「卡住了」,不是「花得多」)。这是燃尽表,每跑一个真任务加一条。</p>")
# ⑤ 缓存 —— 拆成跨轮和轮内两个数,一个混起来的命中率答不了「谁在漏」
parts.append("<h2>⑤ KV 缓存:跨轮 vs 轮内</h2>")
firsts = [r["hit_first"] for r in cache if isinstance(r.get("hit_first"), (int, float))]
rests = [r["hit_rest"] for r in cache if isinstance(r.get("hit_rest"), (int, float))]
if firsts or rests:
for label, g, why in (
("第 1 次调用(跨轮前缀)", firsts,
"低 = 前缀在轮之间断了:system 变了、压缩重排了、历史被改写了"),
("第 2..N 次(轮内前缀)", rests,
"理论上该接近 100%(循环只往后追加)。明显低于 1 = 有人在轮内改写历史,"
"头号嫌疑是 <code>_prune_old_tool_results</code> 每步把旧工具输出换成「已省略」")):
if g:
parts.append(f"<p><b>{label}</b> n={len(g)} · 中位数 "
f"{statistics.median(g):.0%} · 范围 {min(g):.0%}~{max(g):.0%}<br>"
f"<span class=dim>{why}</span></p>")
parts.append("<p class=dim>任一组 n<8 时别下结论。这两个数是**日常使用自动攒的**,"
"不用专门跑对照实验 —— 上一版只有一个混起来的命中率,而一轮里这两种"
"缓存的性质完全不同。</p>")
else:
parts.append("<p class=dim>还没有数据 —— 正常用几轮再来看。</p>")
parts.append("<h2>⑥ system 变没变 × 命中多少</h2>")
if not cache:
parts.append("<p class=dim>还没有数据。正常用几轮,复盘写过技能之后再来看 —— "
"要比的是「system 块变了的那些轮」和「没变的那些轮」。</p>")
else:
for label, want in (("system 没变", False), ("system 变了(复盘写过技能)", True)):
g = [r["hit"] for r in cache if r.get("sys_changed") is want and r.get("hit") is not None]
if g:
parts.append(f"<p><b>{label}</b> n={len(g)} · 中位数 "
f"{statistics.median(g):.0%} · 范围 {min(g):.0%}~{max(g):.0%}</p>")
parts.append("<p class=dim>差值小于 10 个百分点、或任一组 n<8 时,别下结论。</p>")
return ("<meta charset=utf-8><title>Talos 记忆网络</title><style>"
"body{font:15px/1.7 system-ui,sans-serif;max-width:900px;margin:40px auto;padding:0 20px}"
"h1{font-size:22px}h2{font-size:17px;margin-top:32px}"
".dim{color:#777;font-size:13px}table{width:100%;border-collapse:collapse}"
"td{padding:3px 6px;vertical-align:middle}.k{font-size:12px;color:#333}"
".n{text-align:right;font-variant-numeric:tabular-nums;color:#666;width:70px}"
".bar{background:#eee;height:9px;border-radius:5px;width:180px}"
".bar i{display:block;height:9px;background:#7a5af8;border-radius:5px}"
"@media(prefers-color-scheme:dark){body{background:#111;color:#ddd}"
".k{color:#bbb}.bar{background:#333}}</style>" + "\n".join(parts))
if __name__ == "__main__":
if not os.path.isdir(D):
sys.exit("没有 .talos/ —— 先正常用几轮。")
os.makedirs(D, exist_ok=True)
with open(OUT, "w", encoding="utf-8") as f:
f.write(build())
print("写好了:", OUT)
print("(里面有 memory.md 的片段,别往外发)")