From eef76288f19914ea1715ea2d8b8a5b653517af69 Mon Sep 17 00:00:00 2001 From: vivekchand Date: Wed, 9 Sep 2026 11:48:59 +0200 Subject: [PATCH] fix(audit): the harness auditor judged absence from a file it had only partly read Closes #5750. #5750 was filed automatically at severity **high**: "No code path in the OpenClaw/NemoClaw adapter reads NEMOCLAW_TRACE_FILE / the .e2e/traces directory or parses this trace artifact." That code has shipped. `clawmetry/adapters/openclaw.py:1705` reads `NEMOCLAW_TRACE_FILE`, falls back to `NEMOCLAW_TRACE_DIR` and then the harness default, and puts `nemoclawOnboardTraceStatus`, `nemoclawOnboardTraceSpanCount`, `nemoclawOnboardTraceErrors` and `nemoclawOnboardSlowSpans` on the detection record. `clawmetry/adapters/nemo.py` reads it too. REQ-OBS-RSO-034 specifies the whole capability, delivered by PR #5198. The auditor never saw any of it. `_adapter_source` read the first 60,000 characters; `openclaw.py` is 193,340. The reader sits at line 1690, roughly 18k characters past the cut, so **69% of the adapter was invisible** -- and both runtimes this OSS audit covers map to that same file, so every run judged a two-thirds-clipped adapter. Then the prompt told the model: "verify against the FULL adapter above (it is provided in full)". For a task that is entirely about reporting ABSENCE, that sentence converts "I did not see it" into "it is not there". The model did exactly what it was told the evidence supported. This is the second time. The docstring records the first: aider's conditional COST at line ~527 was cut, the audit flagged "no COST", and the cap was raised to 60k in response. Raising a number is not a fix for a file that grows. * the budget is now far above any adapter here, and `_adapter_source` returns whether it trimmed rather than trimming silently; * when it DOES trim, the prompt says so instead of claiming completeness; * `_adapter_index` always carries every `def` and every UPPER_CASE string from the WHOLE file, so an absence claim stays checkable even under trimming. Small, complete, and built from the full source regardless of the body. tests/test_harness_audit_reads_whole_adapter.py auto-discovers from the manifest, so an adapter that outgrows the budget tomorrow fails here instead of quietly filing fiction. Guard proven: restoring the 60k cap and the old completeness claim reds 5 of 6, including both audited runtimes and the `NEMOCLAW_TRACE_FILE`-past-60k witness. The four sibling issues #5746 to #5749 were filed by the same truncated runs and each needs the same check against the full adapter before anyone works it. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01Xb6A5G74JiMe3zHFs1JZEP --- .github/workflows/ci.yml | 1 + scripts/harness/audit.py | 86 ++++++++++++-- .../test_harness_audit_reads_whole_adapter.py | 108 ++++++++++++++++++ 3 files changed, 183 insertions(+), 12 deletions(-) create mode 100644 tests/test_harness_audit_reads_whole_adapter.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 97e5344a29..27b1ff62d3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -569,6 +569,7 @@ jobs: tests/test_lovable_runtime_wiring.py \ tests/test_delegated_usage.py \ tests/test_local_query_api.py \ + tests/test_harness_audit_reads_whole_adapter.py \ tests/test_otlp_sessionless_spans.py \ tests/test_local_query_dispatch_edge_cases.py \ tests/test_store_unreachable_is_not_empty.py \ diff --git a/scripts/harness/audit.py b/scripts/harness/audit.py index 23d7f562d5..1b49f4e757 100644 --- a/scripts/harness/audit.py +++ b/scripts/harness/audit.py @@ -26,6 +26,7 @@ import hashlib import json import os +import re import subprocess import sys @@ -100,17 +101,57 @@ def _harness_surface(clone: str) -> str: return "\n".join(out)[:32000] -def _adapter_source(h: dict) -> str: - """Read the FULL ClawMetry adapter for this runtime. Read the whole file (up - to 60k) — truncating drops the tail, and capabilities()/cost-derivation often - live at the BOTTOM of the adapter, which caused false-positive gaps (e.g. - aider's conditional COST at line ~527 was cut, so the audit wrongly flagged - 'no COST'). +# Big enough for every adapter this repo audits today, with headroom. The cap +# is not the safety mechanism -- ``_adapter_source`` refuses to lie about +# trimming, and ``tests/test_harness_audit_reads_whole_adapter.py`` fails if an +# adapter outgrows it -- but a runaway file must not blow the model's context. +_ADAPTER_BUDGET = 400000 + + +def _adapter_index(src: str) -> str: + """Every definition and env var in the WHOLE file, one per line. + + The auditor's job is to decide what the adapter does NOT do. That makes it + uniquely vulnerable to truncation: absence of evidence in a clipped file + reads exactly like evidence of absence, and the model has no way to tell. + This index is small, complete, and always built from the full source, so a + claim like "no code path reads NEMOCLAW_TRACE_FILE" can be checked against + it even in the (now guarded) case where the body had to be trimmed. + """ + names = sorted(set(re.findall(r"^\s*(?:async\s+)?def\s+(\w+)", src, re.M))) + envs = sorted(set(re.findall(r"[\"\']([A-Z][A-Z0-9_]{4,})[\"\']", src))) + return ("DEFINITIONS (complete, from the whole file):\n " + + ", ".join(names) + + "\n\nUPPER_CASE STRINGS / ENV VARS (complete, from the whole file):\n " + + ", ".join(envs)) + + +def _adapter_source(h: dict) -> tuple: + """Read the ClawMetry adapter for this runtime. Returns ``(source, trimmed)``. + + Truncating drops the tail, and capabilities()/cost-derivation often live at + the BOTTOM of the adapter, which caused false-positive gaps (e.g. aider's + conditional COST at line ~527 was cut, so the audit wrongly flagged 'no + COST'). The cap was raised to 60k for that, and then ``openclaw.py`` grew to + 193k -- so 69% of the adapter went unread again, silently, for BOTH runtimes + this audit covers, while the prompt kept asserting the file was "provided in + full". That is how #5750 was filed at severity high against a + ``NEMOCLAW_TRACE_FILE`` reader sitting at line 1690, roughly 18k characters + past the cut. + + Two changes, because raising a number is not a fix for a file that grows: + the budget is now far above any adapter here, and when it IS exceeded the + caller is told, so the prompt can say the source was trimmed instead of + claiming it was complete. OSS audits only the FREE runtimes (openclaw + nemoclaw), whose adapters are in this repo. The 12 closed pro adapters are audited by clawmetry-pro's own private copy of this script, so no closed adapter path is referenced here.""" - return _read(os.path.join(REPO_ROOT, h["adapter"]), 60000) + path = os.path.join(REPO_ROOT, h["adapter"]) + src = _read(path, _ADAPTER_BUDGET + 1) + if len(src) > _ADAPTER_BUDGET: + return src[:_ADAPTER_BUDGET], True + return src, False def _capabilities_enum() -> str: @@ -144,11 +185,22 @@ def _channel_coverage_context() -> str: return "\n".join(parts) -def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: str = "") -> str: +def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: str = "", + trimmed: bool = False) -> str: _channel_block = ( f"\nClawMetry channel-ingest coverage (sync daemon _CHANNEL_DIRS + HTTP routes — " f"a channel present here is already handled; do NOT report it as a gap):\n```\n{channel_ctx}\n```" ) if channel_ctx else "" + _index = _adapter_index(adapter) + # Never assert completeness we cannot guarantee. The model is being asked to + # report ABSENCE, so an unqualified "provided in full" over a clipped file + # is the one sentence most likely to manufacture a false positive. + _completeness = ( + "NOTE: the adapter body above was TRIMMED to fit. Absence from it is NOT " + "evidence the adapter lacks something. Use the complete index below." + if trimmed else + "The adapter above is the COMPLETE file, start to end." + ) return f"""You audit observability coverage for ClawMetry, which monitors AI agent runtimes. RUNTIME: {h['runtime']} ({h.get('display', h['runtime'])}) @@ -163,6 +215,9 @@ def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: s ``` {adapter} ``` +{_completeness} + +{_index} {_channel_block} The upstream harness — its observable surface (recent commits + data/telemetry files): ``` @@ -176,8 +231,11 @@ def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: s path you can point to in the harness; if you can't point to where the harness exposes it, DO NOT include it (no speculation). -CRITICAL — verify against the FULL adapter above before reporting (it is provided -in full): do NOT flag something the adapter already handles. In particular check +CRITICAL — verify against the adapter above before reporting: do NOT flag +something the adapter already handles. Check the DEFINITIONS and ENV VARS index, +which is COMPLETE for the whole file even when the body above was trimmed: if a +symbol or env var you are about to call missing appears there, it is NOT a gap. +In particular check ``capabilities()`` (capabilities are often added CONDITIONALLY at the bottom of the file), any ``derive_cost_usd`` / cost-derivation, and the field mapping. If the adapter already captures or derives it, it is NOT a gap. @@ -313,8 +371,12 @@ def main() -> int: if not surface: print(f" [skip] no clone at {clone} — run scripts/harness/sync.sh first") continue - adapter = _adapter_source(h) - raw = _run_claude(_build_prompt(h, surface, adapter, caps, channel_ctx)) + adapter, trimmed = _adapter_source(h) + if trimmed: + print(f" [warn] {h['adapter']} exceeded the source budget and was " + f"trimmed; the prompt says so and carries a complete index") + raw = _run_claude(_build_prompt(h, surface, adapter, caps, channel_ctx, + trimmed=trimmed)) gaps = _extract_json_array(raw) print(f" {len(gaps)} gap(s) reported") # Ground (anti-hallucination), severity-sort, drop low, cap per runtime so a diff --git a/tests/test_harness_audit_reads_whole_adapter.py b/tests/test_harness_audit_reads_whole_adapter.py new file mode 100644 index 0000000000..41ab3aecf3 --- /dev/null +++ b/tests/test_harness_audit_reads_whole_adapter.py @@ -0,0 +1,108 @@ +"""The harness audit must not report absence from a file it only partly read. + +`scripts/harness/audit.py` asks a model to list signals a runtime exposes that +ClawMetry's adapter does NOT capture. That task is uniquely vulnerable to +truncation: the model is reasoning about ABSENCE, so a clipped file reads +exactly like a missing feature, and it has no way to tell the difference. + +It happened twice. The cap was raised to 60k after aider's conditional COST at +line ~527 was cut and the audit wrongly flagged "no COST". Then +`clawmetry/adapters/openclaw.py` grew to 193k, so 69% of it went unread again, +silently, for BOTH runtimes this audit covers -- while the prompt kept telling +the model the adapter was "provided in full". That is how #5750 was filed at +severity **high** against a `NEMOCLAW_TRACE_FILE` reader sitting at line 1690, +roughly 18k characters past the cut, with `nemoclawOnboardTraceStatus` on the +detection record and REQ-OBS-RSO-034 specifying the whole capability. + +The cost is not a wasted CI minute. It is a high-severity issue in the tracker +that a human or an agent has to read, reproduce and disprove, against code that +was already shipped. + +These guards auto-discover from the manifest, so an adapter that grows past the +budget tomorrow fails here instead of quietly filing fiction. +""" +from __future__ import annotations + +import json +import os +import sys + +import pytest + +REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, os.path.join(REPO, "scripts", "harness")) + +audit = pytest.importorskip("audit") + + +def _harnesses(): + with open(os.path.join(REPO, "scripts", "harness", "manifest.json")) as f: + m = json.load(f) + rows = m if isinstance(m, list) else m.get("harnesses", m) + out = [] + for h in rows: + if not isinstance(h, dict) or not h.get("adapter"): + continue + if os.path.exists(os.path.join(REPO, h["adapter"])): + out.append(h) + return out + + +@pytest.mark.parametrize("h", _harnesses(), ids=lambda h: h.get("runtime", "?")) +def test_the_audited_adapter_is_read_whole(h): + """The budget must exceed every adapter this repo actually audits.""" + src, trimmed = audit._adapter_source(h) + on_disk = os.path.getsize(os.path.join(REPO, h["adapter"])) + assert not trimmed, ( + f"{h['adapter']} is {on_disk} bytes and exceeds the audit's source " + f"budget, so the model would judge absence from a partial file" + ) + # The tail is where capabilities() and cost-derivation live. + with open(os.path.join(REPO, h["adapter"]), encoding="utf-8", + errors="replace") as f: + whole = f.read() + assert src.endswith(whole[-200:]), "the adapter's tail did not survive" + + +@pytest.mark.parametrize("h", _harnesses(), ids=lambda h: h.get("runtime", "?")) +def test_the_prompt_never_claims_completeness_it_does_not_have(h): + """The regression that made #5750 confident. + + Telling a model the file is complete when it is not is worse than saying + nothing: it converts "I did not see it" into "it is not there". + """ + src, trimmed = audit._adapter_source(h) + whole = audit._build_prompt(h, "surface", src, "caps", "", trimmed=False) + assert "COMPLETE file" in whole + + clipped = audit._build_prompt(h, "surface", src[:1000], "caps", "", trimmed=True) + assert "TRIMMED" in clipped + assert "COMPLETE file, start to end" not in clipped + assert "provided in\nfull" not in clipped and "provided in full" not in clipped + + +def test_the_index_is_complete_even_when_the_body_is_not(): + """The index is what keeps an absence claim checkable under trimming.""" + src = ( + 'def alpha():\n pass\n' + + "# filler\n" * 500 + + 'def omega():\n x = os.environ.get("NEMOCLAW_TRACE_FILE", "")\n' + ) + idx = audit._adapter_index(src) + assert "alpha" in idx and "omega" in idx + assert "NEMOCLAW_TRACE_FILE" in idx + + +def test_the_index_reaches_symbols_past_the_old_60k_cap(): + """Concretely: the thing #5750 said did not exist.""" + h = next((x for x in _harnesses() + if x.get("adapter", "").endswith("openclaw.py")), None) + if h is None: + pytest.skip("openclaw adapter not in the manifest") + src, _ = audit._adapter_source(h) + assert "NEMOCLAW_TRACE_FILE" in src, "the reader must be in the body now" + assert src.index("NEMOCLAW_TRACE_FILE") > 60000, ( + "this symbol is the regression witness: it must sit past the old cap, " + "or this test stops proving anything" + ) + assert "NEMOCLAW_TRACE_FILE" in audit._adapter_index(src)