Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -569,6 +569,7 @@ jobs:
tests/test_lovable_runtime_wiring.py \
tests/test_delegated_usage.py \
tests/test_local_query_api.py \
tests/test_harness_audit_reads_whole_adapter.py \
tests/test_otlp_sessionless_spans.py \
tests/test_local_query_dispatch_edge_cases.py \
tests/test_store_unreachable_is_not_empty.py \
Expand Down
86 changes: 74 additions & 12 deletions scripts/harness/audit.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
import hashlib
import json
import os
import re
import subprocess
import sys

Expand Down Expand Up @@ -100,17 +101,57 @@ def _harness_surface(clone: str) -> str:
return "\n".join(out)[:32000]


def _adapter_source(h: dict) -> str:
"""Read the FULL ClawMetry adapter for this runtime. Read the whole file (up
to 60k) — truncating drops the tail, and capabilities()/cost-derivation often
live at the BOTTOM of the adapter, which caused false-positive gaps (e.g.
aider's conditional COST at line ~527 was cut, so the audit wrongly flagged
'no COST').
# Big enough for every adapter this repo audits today, with headroom. The cap
# is not the safety mechanism -- ``_adapter_source`` refuses to lie about
# trimming, and ``tests/test_harness_audit_reads_whole_adapter.py`` fails if an
# adapter outgrows it -- but a runaway file must not blow the model's context.
_ADAPTER_BUDGET = 400000


def _adapter_index(src: str) -> str:
"""Every definition and env var in the WHOLE file, one per line.

The auditor's job is to decide what the adapter does NOT do. That makes it
uniquely vulnerable to truncation: absence of evidence in a clipped file
reads exactly like evidence of absence, and the model has no way to tell.
This index is small, complete, and always built from the full source, so a
claim like "no code path reads NEMOCLAW_TRACE_FILE" can be checked against
it even in the (now guarded) case where the body had to be trimmed.
"""
names = sorted(set(re.findall(r"^\s*(?:async\s+)?def\s+(\w+)", src, re.M)))
envs = sorted(set(re.findall(r"[\"\']([A-Z][A-Z0-9_]{4,})[\"\']", src)))
return ("DEFINITIONS (complete, from the whole file):\n "
+ ", ".join(names)
+ "\n\nUPPER_CASE STRINGS / ENV VARS (complete, from the whole file):\n "
+ ", ".join(envs))


def _adapter_source(h: dict) -> tuple:
"""Read the ClawMetry adapter for this runtime. Returns ``(source, trimmed)``.

Truncating drops the tail, and capabilities()/cost-derivation often live at
the BOTTOM of the adapter, which caused false-positive gaps (e.g. aider's
conditional COST at line ~527 was cut, so the audit wrongly flagged 'no
COST'). The cap was raised to 60k for that, and then ``openclaw.py`` grew to
193k -- so 69% of the adapter went unread again, silently, for BOTH runtimes
this audit covers, while the prompt kept asserting the file was "provided in
full". That is how #5750 was filed at severity high against a
``NEMOCLAW_TRACE_FILE`` reader sitting at line 1690, roughly 18k characters
past the cut.

Two changes, because raising a number is not a fix for a file that grows:
the budget is now far above any adapter here, and when it IS exceeded the
caller is told, so the prompt can say the source was trimmed instead of
claiming it was complete.

OSS audits only the FREE runtimes (openclaw + nemoclaw), whose adapters are in
this repo. The 12 closed pro adapters are audited by clawmetry-pro's own
private copy of this script, so no closed adapter path is referenced here."""
return _read(os.path.join(REPO_ROOT, h["adapter"]), 60000)
path = os.path.join(REPO_ROOT, h["adapter"])
src = _read(path, _ADAPTER_BUDGET + 1)
if len(src) > _ADAPTER_BUDGET:
return src[:_ADAPTER_BUDGET], True
return src, False


def _capabilities_enum() -> str:
Expand Down Expand Up @@ -144,11 +185,22 @@ def _channel_coverage_context() -> str:
return "\n".join(parts)


def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: str = "") -> str:
def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: str = "",
trimmed: bool = False) -> str:
_channel_block = (
f"\nClawMetry channel-ingest coverage (sync daemon _CHANNEL_DIRS + HTTP routes — "
f"a channel present here is already handled; do NOT report it as a gap):\n```\n{channel_ctx}\n```"
) if channel_ctx else ""
_index = _adapter_index(adapter)
# Never assert completeness we cannot guarantee. The model is being asked to
# report ABSENCE, so an unqualified "provided in full" over a clipped file
# is the one sentence most likely to manufacture a false positive.
_completeness = (
"NOTE: the adapter body above was TRIMMED to fit. Absence from it is NOT "
"evidence the adapter lacks something. Use the complete index below."
if trimmed else
"The adapter above is the COMPLETE file, start to end."
)
return f"""You audit observability coverage for ClawMetry, which monitors AI agent runtimes.

RUNTIME: {h['runtime']} ({h.get('display', h['runtime'])})
Expand All @@ -163,6 +215,9 @@ def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: s
```
{adapter}
```
{_completeness}

{_index}
{_channel_block}
The upstream harness — its observable surface (recent commits + data/telemetry files):
```
Expand All @@ -176,8 +231,11 @@ def _build_prompt(h: dict, surface: str, adapter: str, caps: str, channel_ctx: s
path you can point to in the harness; if you can't point to where the harness
exposes it, DO NOT include it (no speculation).

CRITICAL — verify against the FULL adapter above before reporting (it is provided
in full): do NOT flag something the adapter already handles. In particular check
CRITICAL — verify against the adapter above before reporting: do NOT flag
something the adapter already handles. Check the DEFINITIONS and ENV VARS index,
which is COMPLETE for the whole file even when the body above was trimmed: if a
symbol or env var you are about to call missing appears there, it is NOT a gap.
In particular check
``capabilities()`` (capabilities are often added CONDITIONALLY at the bottom of
the file), any ``derive_cost_usd`` / cost-derivation, and the field mapping. If
the adapter already captures or derives it, it is NOT a gap.
Expand Down Expand Up @@ -313,8 +371,12 @@ def main() -> int:
if not surface:
print(f" [skip] no clone at {clone} — run scripts/harness/sync.sh first")
continue
adapter = _adapter_source(h)
raw = _run_claude(_build_prompt(h, surface, adapter, caps, channel_ctx))
adapter, trimmed = _adapter_source(h)
if trimmed:
print(f" [warn] {h['adapter']} exceeded the source budget and was "
f"trimmed; the prompt says so and carries a complete index")
raw = _run_claude(_build_prompt(h, surface, adapter, caps, channel_ctx,
trimmed=trimmed))
gaps = _extract_json_array(raw)
print(f" {len(gaps)} gap(s) reported")
# Ground (anti-hallucination), severity-sort, drop low, cap per runtime so a
Expand Down
108 changes: 108 additions & 0 deletions tests/test_harness_audit_reads_whole_adapter.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,108 @@
"""The harness audit must not report absence from a file it only partly read.

`scripts/harness/audit.py` asks a model to list signals a runtime exposes that
ClawMetry's adapter does NOT capture. That task is uniquely vulnerable to
truncation: the model is reasoning about ABSENCE, so a clipped file reads
exactly like a missing feature, and it has no way to tell the difference.

It happened twice. The cap was raised to 60k after aider's conditional COST at
line ~527 was cut and the audit wrongly flagged "no COST". Then
`clawmetry/adapters/openclaw.py` grew to 193k, so 69% of it went unread again,
silently, for BOTH runtimes this audit covers -- while the prompt kept telling
the model the adapter was "provided in full". That is how #5750 was filed at
severity **high** against a `NEMOCLAW_TRACE_FILE` reader sitting at line 1690,
roughly 18k characters past the cut, with `nemoclawOnboardTraceStatus` on the
detection record and REQ-OBS-RSO-034 specifying the whole capability.

The cost is not a wasted CI minute. It is a high-severity issue in the tracker
that a human or an agent has to read, reproduce and disprove, against code that
was already shipped.

These guards auto-discover from the manifest, so an adapter that grows past the
budget tomorrow fails here instead of quietly filing fiction.
"""
from __future__ import annotations

import json
import os
import sys

import pytest

REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, os.path.join(REPO, "scripts", "harness"))

audit = pytest.importorskip("audit")


def _harnesses():
with open(os.path.join(REPO, "scripts", "harness", "manifest.json")) as f:
m = json.load(f)
rows = m if isinstance(m, list) else m.get("harnesses", m)
out = []
for h in rows:
if not isinstance(h, dict) or not h.get("adapter"):
continue
if os.path.exists(os.path.join(REPO, h["adapter"])):
out.append(h)
return out


@pytest.mark.parametrize("h", _harnesses(), ids=lambda h: h.get("runtime", "?"))
def test_the_audited_adapter_is_read_whole(h):
"""The budget must exceed every adapter this repo actually audits."""
src, trimmed = audit._adapter_source(h)
on_disk = os.path.getsize(os.path.join(REPO, h["adapter"]))
assert not trimmed, (
f"{h['adapter']} is {on_disk} bytes and exceeds the audit's source "
f"budget, so the model would judge absence from a partial file"
)
# The tail is where capabilities() and cost-derivation live.
with open(os.path.join(REPO, h["adapter"]), encoding="utf-8",
errors="replace") as f:
whole = f.read()
assert src.endswith(whole[-200:]), "the adapter's tail did not survive"


@pytest.mark.parametrize("h", _harnesses(), ids=lambda h: h.get("runtime", "?"))
def test_the_prompt_never_claims_completeness_it_does_not_have(h):
"""The regression that made #5750 confident.

Telling a model the file is complete when it is not is worse than saying
nothing: it converts "I did not see it" into "it is not there".
"""
src, trimmed = audit._adapter_source(h)
whole = audit._build_prompt(h, "surface", src, "caps", "", trimmed=False)
assert "COMPLETE file" in whole

clipped = audit._build_prompt(h, "surface", src[:1000], "caps", "", trimmed=True)
assert "TRIMMED" in clipped
assert "COMPLETE file, start to end" not in clipped
assert "provided in\nfull" not in clipped and "provided in full" not in clipped


def test_the_index_is_complete_even_when_the_body_is_not():
"""The index is what keeps an absence claim checkable under trimming."""
src = (
'def alpha():\n pass\n'
+ "# filler\n" * 500
+ 'def omega():\n x = os.environ.get("NEMOCLAW_TRACE_FILE", "")\n'
)
idx = audit._adapter_index(src)
assert "alpha" in idx and "omega" in idx
assert "NEMOCLAW_TRACE_FILE" in idx


def test_the_index_reaches_symbols_past_the_old_60k_cap():
"""Concretely: the thing #5750 said did not exist."""
h = next((x for x in _harnesses()
if x.get("adapter", "").endswith("openclaw.py")), None)
if h is None:
pytest.skip("openclaw adapter not in the manifest")
src, _ = audit._adapter_source(h)
assert "NEMOCLAW_TRACE_FILE" in src, "the reader must be in the body now"
assert src.index("NEMOCLAW_TRACE_FILE") > 60000, (
"this symbol is the regression witness: it must sit past the old cap, "
"or this test stops proving anything"
)
assert "NEMOCLAW_TRACE_FILE" in audit._adapter_index(src)
Loading