From c48c675b3c8d689a6582e6695e6013cc9427b71e Mon Sep 17 00:00:00 2001 From: Christopher Sarkissian Date: Wed, 2 Sep 2026 21:49:37 -0700 Subject: [PATCH] test: migrate agent suites to direct execution --- .env.example | 11 ++++-- .gitignore | 2 +- .skeleton/review-lock.json | 10 ++--- AGENTS.md | 2 +- README.md | 2 +- agent-suites/README.md | 34 ++++++++--------- agent-suites/evidence/README.md | 4 +- agent-suites/evidence/SUMMARY.md | 2 +- agent-suites/evidence/charts/pass-rates.svg | 2 +- agent-suites/fixtures/README.md | 2 +- .../clean/docs/fixture/billing-api-legacy.md | 4 +- .../fixtures/seeds/grounding.patch | 14 +------ agent-suites/skeleton-clean/scenarios.json | 4 +- .../fixtures/seeds/customize-demo.patch | 38 ------------------- .../fixtures/seeds/grounding.patch | 38 ------------------- .../fixtures/seeds/routing-docs.patch | 38 ------------------- .../fixtures/seeds/routing-skill-staged.patch | 38 ------------------- agent-suites/skeleton-messy/scenarios.json | 4 +- package.json | 6 +-- refs/llm-harness.md | 4 +- scripts/agent-evidence/aggregate-compares.ts | 4 +- scripts/agent-evidence/render-charts.ts | 2 +- 22 files changed, 54 insertions(+), 211 deletions(-) diff --git a/.env.example b/.env.example index d39abef..c999b03 100644 --- a/.env.example +++ b/.env.example @@ -1,11 +1,16 @@ # This package requires no runtime environment variables for the CLI itself. # Copy to `.env` only if you add private overrides locally (do not commit `.env`). -# Agent-test live dogfood (`bun run agent:test:live` / `agent:test:live:compare`). -# Export the key in your shell — the agent-test CLI does not auto-load `.env`. +# Direct Cursor agent tests (`bun run agent:test:direct` / `agent:test:direct:compare`). +# Export the key in your shell — agent-test does not auto-load `.env`. +# Every test execution launches a real provider agent and can incur usage. # CURSOR_API_KEY= -# Optional model overrides for live agent / judge +# Required instead of CURSOR_API_KEY when running with `--host claude`. +# The default Cursor judge still requires CURSOR_API_KEY unless `--no-judge` is set. +# ANTHROPIC_API_KEY= + +# Optional model overrides for the agent and judge # CURSOR_AGENT_MODEL= # CURSOR_JUDGE_MODEL= diff --git a/.gitignore b/.gitignore index ea0bf95..a628de3 100644 --- a/.gitignore +++ b/.gitignore @@ -15,5 +15,5 @@ src/audit/__tests__/fixtures/plugins/**/*.mjs src/audit/__tests__/fixtures/plugins/**/*.mjs.stamp .skeleton/catalog.md -# Raw live compare dumps (aggregate into SUMMARY.*) +# Raw direct-agent compare dumps (aggregate into SUMMARY.*) agent-suites/evidence/runs/ diff --git a/.skeleton/review-lock.json b/.skeleton/review-lock.json index 8b0858a..bbaf4f8 100644 --- a/.skeleton/review-lock.json +++ b/.skeleton/review-lock.json @@ -5,24 +5,24 @@ "reviewedAt": "2026-09-02", "documentHash": "sha256:f9df3e875c53499eda112a7c70d2fb76c0cbbb0dfec250dc55e6d1201b04c33f", "reviewDependencies": { - "AGENTS.md": "sha256:f3059ccf7ba993212a97be4e628e3e4c7688739036e7b0151b2005f8dcb4393b", + "AGENTS.md": "sha256:cb166286f3b8d469149488c9bf20941cffe3edb8891b1998467d600e192d7903", "docs/developer/validation.md": "sha256:6e843854039b7a62700ccccabea9284cc00f19ce70deb3115bf77f3acc42f94a", "src/validate/changed.ts": "sha256:c2210201db3962ca99e603a5ece323d5ffbd8a8145ee577bfa191831f977be5a" } }, "AGENTS.md": { "reviewedAt": "2026-09-02", - "documentHash": "sha256:f3059ccf7ba993212a97be4e628e3e4c7688739036e7b0151b2005f8dcb4393b", + "documentHash": "sha256:cb166286f3b8d469149488c9bf20941cffe3edb8891b1998467d600e192d7903", "reviewDependencies": { - "package.json": "sha256:33c803d5fc5280994f12cf4b3e69c7706378d111dbd3a026b092f742b6d99c04", + "package.json": "sha256:3411d4e9efaac85b784328fef349e51fe5dc47751611c9d86f7517c56da2be7a", "src/cli.ts": "sha256:1fc7a773f0295fd209a4d3baed536c24dc7bdf3d3c079118dad6f94a2de6026a" } }, "README.md": { "reviewedAt": "2026-09-02", - "documentHash": "sha256:265b48daf1265ce64be229be53c5545c43dc882585cb2e96f6858fc0410d182d", + "documentHash": "sha256:2cddd454afcfdf9acda2ff2402b0a338ea2ad10a0bc551fa27c51e24a0cb04b6", "reviewDependencies": { - "package.json": "sha256:33c803d5fc5280994f12cf4b3e69c7706378d111dbd3a026b092f742b6d99c04", + "package.json": "sha256:3411d4e9efaac85b784328fef349e51fe5dc47751611c9d86f7517c56da2be7a", "src/cli.ts": "sha256:1fc7a773f0295fd209a4d3baed536c24dc7bdf3d3c079118dad6f94a2de6026a" } }, diff --git a/AGENTS.md b/AGENTS.md index 9f974da..f810480 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -59,7 +59,7 @@ bun src/cli.ts audit docs --paths=docs/a.md --fix=doc-meta --confirm-reviewed Optional local hooks: install [pre-commit](https://pre-commit.com/) (`brew install pre-commit` or `pipx install pre-commit`), then `pre-commit install`. Customize IDE hooks from `skeleton init` are optional — not required for audit. -Behavioral A/B dogfood (live Cursor, not part of `bun run check`): [agent-suites/README.md](agent-suites/README.md) · [refs/llm-harness.md](refs/llm-harness.md). +Behavioral A/B dogfood (direct Cursor, not part of `bun run check`): [agent-suites/README.md](agent-suites/README.md) · [refs/llm-harness.md](refs/llm-harness.md). Consumer-facing decision table and routing: [docs/developer/validation.md](docs/developer/validation.md). Common failures: [docs/developer/troubleshooting.md](docs/developer/troubleshooting.md). Day-one setup: [docs/developer/getting-started.md](docs/developer/getting-started.md). diff --git a/README.md b/README.md index e3cc9be..2912064 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ Does an intact Skeleton contract change agent behavior — grounding on the righ ### What we did -We ran a paired live A/B self-benchmark with [`@post-print/agent-test`](https://www.npmjs.com/package/@post-print/agent-test): `skeleton-clean` vs `skeleton-messy`. The same authored prompts compare an intact fixture with a conflicting fixture. +We ran a paired direct-agent A/B self-benchmark with [`@post-print/agent-test`](https://www.npmjs.com/package/@post-print/agent-test): `skeleton-clean` vs `skeleton-messy`. The same authored prompts compare an intact fixture with a conflicting fixture. Scenarios covered contested grounding (conflicting docs), docs-only validation routing, canonical grounding, owned-skill routing, and customize ownership. Protocol: **N=10** sequential paired compares on 2026-07-17; McNemar on paired pass/fail; median token deltas with a bootstrap CI on the mean. diff --git a/agent-suites/README.md b/agent-suites/README.md index 889879d..ead2d6e 100644 --- a/agent-suites/README.md +++ b/agent-suites/README.md @@ -1,10 +1,10 @@ # Agent suites (skeleton behavioral benchmark) -**Source of truth for** live A/B dogfood of the Skeleton SSOT contract via `@post-print/agent-test`. +**Source of truth for** direct-agent A/B dogfood of the Skeleton SSOT contract via `@post-print/agent-test`. -These suites measure whether a clean Skeleton structure (registry, validation lanes, customize) improves **grounding**, **validation routing**, and **token efficiency** versus a messy control tree — not portable skill conformance (that lives in [toolbox](https://github.com/csark0812/toolbox) `agent-suites/`). +These suites measure whether a clean Skeleton structure (catalog, validation lanes, customize) improves **grounding**, **validation routing**, and **token efficiency** versus a messy control tree — not portable skill conformance (that lives in [toolbox](https://github.com/csark0812/toolbox) `agent-suites/`). Committed stats and transcript excerpts: [`evidence/`](evidence/). Protocol SSOT: [`refs/llm-harness.md`](../refs/llm-harness.md). @@ -12,8 +12,8 @@ Committed stats and transcript excerpts: [`evidence/`](evidence/). Protocol SSOT ``` agent-suites/ - skeleton-clean/ # profile: skeleton — registry in preamble + seeded fixture docs - skeleton-messy/ # profile: shared — emptied worktree registry + conflicting docs + skeleton-clean/ # profile: skeleton — catalog routing in preamble + canonical fixture doc + skeleton-messy/ # profile: shared — no catalog routing + conflicting fixture docs fixtures/ # source trees for seed patches evidence/ # SUMMARY + curated transcripts (runs/ gitignored) ``` @@ -22,36 +22,36 @@ Paired scenario **names** match across clean/messy for `--compare-pairs skeleton | Scenario | Theme | | ----------------------------- | --------------------------------------------------- | -| `grounding: canonical topic` | Registry-first path + webhook citation | -| `grounding: conflicting docs` | SoT winner via registry | +| `grounding: canonical topic` | Catalog-first path + webhook citation | +| `grounding: conflicting docs` | SoT winner via the generated catalog | | `routing: docs-only change` | `validate:changed` lane | | `routing: owned skill body` | `audit:skills` lane | | `customize: project binding` | `.skeleton/customize/` vs editing synced `SKILL.md` | ## Commands -Requires **Node ≥ 22**, exported `CURSOR_API_KEY` (see [`.env.example`](../.env.example)), and `@post-print/agent-test` ≥ 0.2.7. +Requires **Node ≥ 22**, a direct-only `@post-print/agent-test` release, and an exported provider credential (see [`.env.example`](../.env.example)). Cursor is the suite default and requires `CURSOR_API_KEY`. Claude runs use `--host claude` and require `ANTHROPIC_API_KEY`; the default judge still requires `CURSOR_API_KEY` unless `--no-judge` is set. Every execution launches a real provider agent and can incur usage. ```bash bun run agent:test:doctor bun run agent:test:validate -bun run agent:test:live:compare +bun run agent:test:direct:compare ``` Optional debug (staging under `$TMPDIR` by default): ```bash -bun run agent:test:live:debug -- --suite skeleton-clean -bun run agent:test:live:compare -- --debug --out-dir "$TMPDIR/skeleton-compare" +bun run agent:test:direct:debug -- --suite skeleton-clean +bun run agent:test:direct:compare -- --debug --out-dir "$TMPDIR/skeleton-compare" ``` -Live is the primary signal. Suites default to `host: "replay"` so accidental non-live runs are not CI gates — always pass `--live` (the npm scripts do). Golden replay traces are deferred until rubrics stabilize. +Direct execution is the primary signal. Both suites default to `host: "cursor"`; use `--host claude` to run the same scenarios with Claude. `agent:test:validate` is the offline configuration and path check. Do not add direct execution to deterministic CI without explicit credentials, provider-usage approval, and a budget. -**Note:** Live worktrees load **preamble context from the caller checkout** (`AGENTS.md`, profile sources). Seed patches change the **worktree disk** the agent tools see. Clean vs messy therefore differs by `profile` (`skeleton` vs `shared`) plus seeded fixture/registry state. Skill seeds use worktree path `fixture-skills/` (not `.claude/skills/`) because root `.claude/` is gitignored and harness seeding must `git add` the files. +**Note:** Direct runs load **preamble context from the caller checkout** (`AGENTS.md`, profile sources). Seed patches change the **worktree disk** the agent tools see. Clean vs messy therefore differs by `profile` (`skeleton` vs `shared`) plus the seeded fixture's SSOT shape. Skill seeds use worktree path `fixture-skills/` (not `.claude/skills/`) because root `.claude/` is gitignored and harness seeding must `git add` the files. ## Success criteria (KPIs) -Score from `compare-report.json` after paired live runs. **Protocol target: N=10** independent compares before final README claims (see gates in [`evidence/SUMMARY.md`](evidence/SUMMARY.md)). +Score from `compare-report.json` after paired direct runs. **Protocol target: N=10** independent compares before final README claims (see gates in [`evidence/SUMMARY.md`](evidence/SUMMARY.md)). | KPI | Definition | Success | | ------------------------- | --------------------------------------------- | --------------------------------------------------------- | @@ -64,13 +64,13 @@ Score from `compare-report.json` after paired live runs. **Protocol target: N=10 ## Dogfood SOP (deposit + aggregate) -1. Export `CURSOR_API_KEY` (CLI does not load `.env`). +1. Export `CURSOR_API_KEY` for the default Cursor host, or export `ANTHROPIC_API_KEY` and pass `--host claude`. The CLI does not load `.env`; the default judge still needs `CURSOR_API_KEY` unless disabled. 2. `bun run agent:test:doctor` 3. Run a compare into a unique out dir: ```bash OUT="$TMPDIR/skeleton-compare-run-002" -bunx agent-test --suites-dir agent-suites --live \ +bunx agent-test --suites-dir agent-suites \ --compare-pairs skeleton-clean:skeleton-messy \ --fail-on=behavior --out-dir "$OUT" mkdir -p agent-suites/evidence/runs/$(date +%Y-%m-%d)-run-002 @@ -87,8 +87,8 @@ bun run agent:evidence:excerpt -- --run-dir agent-suites/evidence/runs/ ``` 5. Update the metric log row in [refs/llm-harness.md](../refs/llm-harness.md). -6. If seeds drift after registry edits, re-check with `git apply --check` on `**/fixtures/seeds/*.patch`. +6. If seeds drift after catalog or fixture changes, rerun `bunx agent-test --validate-seeds --suites-dir agent-suites`. `evidence/runs/` is gitignored. Commit `SUMMARY.*`, `transcripts/`, and `samples/` after meaningful batches. -Do **not** fold live agent-test into `bun run check`. +Do **not** fold direct agent execution into `bun run check`. diff --git a/agent-suites/evidence/README.md b/agent-suites/evidence/README.md index 3f3b522..021e59a 100644 --- a/agent-suites/evidence/README.md +++ b/agent-suites/evidence/README.md @@ -1,6 +1,6 @@ # Behavioral evidence -**Source of truth for** committed artifacts from the Skeleton live A/B benchmark. +**Source of truth for** committed artifacts from the Skeleton direct-agent A/B benchmark. @@ -18,7 +18,7 @@ `runs/` — raw per-run `compare-report.json` (+ optional suite reports). Deposit locally, then aggregate: ```bash -# after each live compare: +# after each direct compare: mkdir -p agent-suites/evidence/runs/$(date +%Y-%m-%d)-run-NNN cp "$TMPDIR/skeleton-compare-run-NNN/compare-report.json" \ agent-suites/evidence/runs/$(date +%Y-%m-%d)-run-NNN/ diff --git a/agent-suites/evidence/SUMMARY.md b/agent-suites/evidence/SUMMARY.md index 8a51b11..42dd593 100644 --- a/agent-suites/evidence/SUMMARY.md +++ b/agent-suites/evidence/SUMMARY.md @@ -1,6 +1,6 @@ # Behavioral evidence summary -**Source of truth for** aggregated Skeleton A/B live compares (`skeleton-clean` vs `skeleton-messy`). +**Source of truth for** aggregated Skeleton A/B direct-agent compares (`skeleton-clean` vs `skeleton-messy`). diff --git a/agent-suites/evidence/charts/pass-rates.svg b/agent-suites/evidence/charts/pass-rates.svg index 9d0db8d..299334d 100644 --- a/agent-suites/evidence/charts/pass-rates.svg +++ b/agent-suites/evidence/charts/pass-rates.svg @@ -2,7 +2,7 @@ Pass rate by scenario - N=10 paired live compares · higher is better + N=10 paired direct compares · higher is better skeleton-clean diff --git a/agent-suites/fixtures/README.md b/agent-suites/fixtures/README.md index 8d9f6f0..b56da62 100644 --- a/agent-suites/fixtures/README.md +++ b/agent-suites/fixtures/README.md @@ -4,7 +4,7 @@ Readable trees used to build `skeleton-clean` / `skeleton-messy` seed patches un | Tree | Role | | ---- | ---- | -| `clean/docs/fixture/` | Registered Billing API SoT + legacy decoy | +| `clean/docs/fixture/` | Canonical Billing API SSOT + legacy decoy | | `clean/skill-trees/` | Bodies seeded into worktree `fixture-skills/` | | `messy/docs/fixture/` | Conflicting SoT claims (no correct webhook) | | `messy/skill-trees/` | Same skill bodies for messy arm | diff --git a/agent-suites/fixtures/clean/docs/fixture/billing-api-legacy.md b/agent-suites/fixtures/clean/docs/fixture/billing-api-legacy.md index 73c131b..3af41e4 100644 --- a/agent-suites/fixtures/clean/docs/fixture/billing-api-legacy.md +++ b/agent-suites/fixtures/clean/docs/fixture/billing-api-legacy.md @@ -1,5 +1,5 @@ # Billing API (legacy notes) -This draft is **not** registered. Wrong webhook: `https://legacy.example.com/hooks/bill` +This draft is **not** canonical. Wrong webhook: `https://legacy.example.com/hooks/bill` -Agents must not cite this URL when the registry points at billing-api.md. +Agents must not cite this URL when the catalog points at billing-api.md. diff --git a/agent-suites/skeleton-clean/fixtures/seeds/grounding.patch b/agent-suites/skeleton-clean/fixtures/seeds/grounding.patch index 8c8c04c..9a17b93 100644 --- a/agent-suites/skeleton-clean/fixtures/seeds/grounding.patch +++ b/agent-suites/skeleton-clean/fixtures/seeds/grounding.patch @@ -20,17 +20,7 @@ new file mode 100644 @@ -0,0 +1,5 @@ +# Billing API (legacy notes) + -+This draft is **not** registered. Wrong webhook: `https://legacy.example.com/hooks/bill` ++This draft is **not** canonical. Wrong webhook: `https://legacy.example.com/hooks/bill` + -+Agents must not cite this URL when the registry points at billing-api.md. - ---- a/.skeleton/registry.md -+++ b/.skeleton/registry.md -@@ -22,6 +22,7 @@ - | Plugins | [plugins.md](../docs/developer/plugins.md) | - | Customize | [customize.md](../docs/developer/customize.md) | - | Skeleton skill | [SKILL.md](../skeleton/SKILL.md) | -+| Billing API | [billing-api.md](../docs/fixture/billing-api.md) | - - ## Customizations ++Agents must not cite this URL when the catalog points at billing-api.md. diff --git a/agent-suites/skeleton-clean/scenarios.json b/agent-suites/skeleton-clean/scenarios.json index c5a7e91..12d80a3 100644 --- a/agent-suites/skeleton-clean/scenarios.json +++ b/agent-suites/skeleton-clean/scenarios.json @@ -1,8 +1,8 @@ { "name": "skeleton-clean", - "description": "Live A/B clean arm: profile=skeleton + catalog/SSOT fixture docs. Pair with skeleton-messy via --compare-pairs.", + "description": "Direct-agent A/B clean arm: profile=skeleton + catalog/SSOT fixture docs. Pair with skeleton-messy via --compare-pairs.", "defaults": { - "host": "replay", + "host": "cursor", "profile": "skeleton", "skills": "none" }, diff --git a/agent-suites/skeleton-messy/fixtures/seeds/customize-demo.patch b/agent-suites/skeleton-messy/fixtures/seeds/customize-demo.patch index c5a778d..e1036e8 100644 --- a/agent-suites/skeleton-messy/fixtures/seeds/customize-demo.patch +++ b/agent-suites/skeleton-messy/fixtures/seeds/customize-demo.patch @@ -10,41 +10,3 @@ new file mode 100644 + + +Protocol: answer with DEMO_SKILL_BASELINE when invoked with no customize overlay. - ---- a/.skeleton/registry.md -+++ b/.skeleton/registry.md -@@ -1,30 +1,10 @@ - # Registry - -- -+ - --**Source of truth for** topic routing in this repo. Edit rows here; edit content in canonical files only. -+**Source of truth for** topic routing in this repo. - - ## Documentation - --| Topic | Canonical file | --| --------------------- | ---------------------------------------------------------- | --| Package overview | [README.md](../README.md) | --| Agent cold-start | [AGENTS.md](../AGENTS.md) | --| Authoring conventions | [authoring.md](../docs/authoring.md) | --| Ecosystem tiers | [tiers.md](../docs/tiers.md) | --| Getting started | [getting-started.md](../docs/developer/getting-started.md) | --| Install | [install.md](../docs/developer/install.md) | --| Config | [config.md](../docs/developer/config.md) | --| Validation | [validation.md](../docs/developer/validation.md) | --| Troubleshooting | [troubleshooting.md](../docs/developer/troubleshooting.md) | --| Doc system | [doc-system.md](../docs/developer/doc-system.md) | --| Audit | [audit.md](../docs/developer/audit.md) | --| Plugins | [plugins.md](../docs/developer/plugins.md) | --| Customize | [customize.md](../docs/developer/customize.md) | --| Skeleton skill | [SKILL.md](../skeleton/SKILL.md) | -- --## Customizations -- --| Topic | Canonical file | --| ------------------------------------------------------------------------------------------------------ | ------------------------------------------ | --| Customize: skeleton-specific code-review overlays (validation ladder, invariant matrices, Action bar). | [code-review.md](customize/code-review.md) | -+| Topic | Canonical file | -+| ----- | -------------- | diff --git a/agent-suites/skeleton-messy/fixtures/seeds/grounding.patch b/agent-suites/skeleton-messy/fixtures/seeds/grounding.patch index 83947ad..448f31b 100644 --- a/agent-suites/skeleton-messy/fixtures/seeds/grounding.patch +++ b/agent-suites/skeleton-messy/fixtures/seeds/grounding.patch @@ -32,41 +32,3 @@ new file mode 100644 +Webhook: `https://messy-c.example.com/pay` + +Also try `bun run audit all` before merge (invented gate). - ---- a/.skeleton/registry.md -+++ b/.skeleton/registry.md -@@ -1,30 +1,10 @@ - # Registry - -- -+ - --**Source of truth for** topic routing in this repo. Edit rows here; edit content in canonical files only. -+**Source of truth for** topic routing in this repo. - - ## Documentation - --| Topic | Canonical file | --| --------------------- | ---------------------------------------------------------- | --| Package overview | [README.md](../README.md) | --| Agent cold-start | [AGENTS.md](../AGENTS.md) | --| Authoring conventions | [authoring.md](../docs/authoring.md) | --| Ecosystem tiers | [tiers.md](../docs/tiers.md) | --| Getting started | [getting-started.md](../docs/developer/getting-started.md) | --| Install | [install.md](../docs/developer/install.md) | --| Config | [config.md](../docs/developer/config.md) | --| Validation | [validation.md](../docs/developer/validation.md) | --| Troubleshooting | [troubleshooting.md](../docs/developer/troubleshooting.md) | --| Doc system | [doc-system.md](../docs/developer/doc-system.md) | --| Audit | [audit.md](../docs/developer/audit.md) | --| Plugins | [plugins.md](../docs/developer/plugins.md) | --| Customize | [customize.md](../docs/developer/customize.md) | --| Skeleton skill | [SKILL.md](../skeleton/SKILL.md) | -- --## Customizations -- --| Topic | Canonical file | --| ------------------------------------------------------------------------------------------------------ | ------------------------------------------ | --| Customize: skeleton-specific code-review overlays (validation ladder, invariant matrices, Action bar). | [code-review.md](customize/code-review.md) | -+| Topic | Canonical file | -+| ----- | -------------- | diff --git a/agent-suites/skeleton-messy/fixtures/seeds/routing-docs.patch b/agent-suites/skeleton-messy/fixtures/seeds/routing-docs.patch index 2005524..1940abd 100644 --- a/agent-suites/skeleton-messy/fixtures/seeds/routing-docs.patch +++ b/agent-suites/skeleton-messy/fixtures/seeds/routing-docs.patch @@ -34,41 +34,3 @@ new file mode 100644 +**Source of truth for** Billing API. + +Webhook: `https://messy-b.example.com/pay` - ---- a/.skeleton/registry.md -+++ b/.skeleton/registry.md -@@ -1,30 +1,10 @@ - # Registry - -- -+ - --**Source of truth for** topic routing in this repo. Edit rows here; edit content in canonical files only. -+**Source of truth for** topic routing in this repo. - - ## Documentation - --| Topic | Canonical file | --| --------------------- | ---------------------------------------------------------- | --| Package overview | [README.md](../README.md) | --| Agent cold-start | [AGENTS.md](../AGENTS.md) | --| Authoring conventions | [authoring.md](../docs/authoring.md) | --| Ecosystem tiers | [tiers.md](../docs/tiers.md) | --| Getting started | [getting-started.md](../docs/developer/getting-started.md) | --| Install | [install.md](../docs/developer/install.md) | --| Config | [config.md](../docs/developer/config.md) | --| Validation | [validation.md](../docs/developer/validation.md) | --| Troubleshooting | [troubleshooting.md](../docs/developer/troubleshooting.md) | --| Doc system | [doc-system.md](../docs/developer/doc-system.md) | --| Audit | [audit.md](../docs/developer/audit.md) | --| Plugins | [plugins.md](../docs/developer/plugins.md) | --| Customize | [customize.md](../docs/developer/customize.md) | --| Skeleton skill | [SKILL.md](../skeleton/SKILL.md) | -- --## Customizations -- --| Topic | Canonical file | --| ------------------------------------------------------------------------------------------------------ | ------------------------------------------ | --| Customize: skeleton-specific code-review overlays (validation ladder, invariant matrices, Action bar). | [code-review.md](customize/code-review.md) | -+| Topic | Canonical file | -+| ----- | -------------- | diff --git a/agent-suites/skeleton-messy/fixtures/seeds/routing-skill-staged.patch b/agent-suites/skeleton-messy/fixtures/seeds/routing-skill-staged.patch index 1bbf76b..8912ea4 100644 --- a/agent-suites/skeleton-messy/fixtures/seeds/routing-skill-staged.patch +++ b/agent-suites/skeleton-messy/fixtures/seeds/routing-skill-staged.patch @@ -10,41 +10,3 @@ new file mode 100644 + + +Baseline line: OWNED_DEMO_V2_STAGED - ---- a/.skeleton/registry.md -+++ b/.skeleton/registry.md -@@ -1,30 +1,10 @@ - # Registry - -- -+ - --**Source of truth for** topic routing in this repo. Edit rows here; edit content in canonical files only. -+**Source of truth for** topic routing in this repo. - - ## Documentation - --| Topic | Canonical file | --| --------------------- | ---------------------------------------------------------- | --| Package overview | [README.md](../README.md) | --| Agent cold-start | [AGENTS.md](../AGENTS.md) | --| Authoring conventions | [authoring.md](../docs/authoring.md) | --| Ecosystem tiers | [tiers.md](../docs/tiers.md) | --| Getting started | [getting-started.md](../docs/developer/getting-started.md) | --| Install | [install.md](../docs/developer/install.md) | --| Config | [config.md](../docs/developer/config.md) | --| Validation | [validation.md](../docs/developer/validation.md) | --| Troubleshooting | [troubleshooting.md](../docs/developer/troubleshooting.md) | --| Doc system | [doc-system.md](../docs/developer/doc-system.md) | --| Audit | [audit.md](../docs/developer/audit.md) | --| Plugins | [plugins.md](../docs/developer/plugins.md) | --| Customize | [customize.md](../docs/developer/customize.md) | --| Skeleton skill | [SKILL.md](../skeleton/SKILL.md) | -- --## Customizations -- --| Topic | Canonical file | --| ------------------------------------------------------------------------------------------------------ | ------------------------------------------ | --| Customize: skeleton-specific code-review overlays (validation ladder, invariant matrices, Action bar). | [code-review.md](customize/code-review.md) | -+| Topic | Canonical file | -+| ----- | -------------- | diff --git a/agent-suites/skeleton-messy/scenarios.json b/agent-suites/skeleton-messy/scenarios.json index a5a8156..95c7591 100644 --- a/agent-suites/skeleton-messy/scenarios.json +++ b/agent-suites/skeleton-messy/scenarios.json @@ -1,8 +1,8 @@ { "name": "skeleton-messy", - "description": "Live A/B messy arm: profile=shared, no useful catalog, conflicting fixture docs. Pair with skeleton-clean via --compare-pairs.", + "description": "Direct-agent A/B messy arm: profile=shared, no useful catalog, conflicting fixture docs. Pair with skeleton-clean via --compare-pairs.", "defaults": { - "host": "replay", + "host": "cursor", "profile": "shared", "skills": "none" }, diff --git a/package.json b/package.json index 5abf982..b06ae1d 100644 --- a/package.json +++ b/package.json @@ -53,9 +53,9 @@ "verify:package": "bun run build && bun scripts/verify-package.ts", "start": "bun src/cli.ts --help", "dev": "bun src/cli.ts --help", - "agent:test:live": "agent-test --suites-dir agent-suites --live --fail-on=behavior", - "agent:test:live:compare": "agent-test --suites-dir agent-suites --live --compare-pairs skeleton-clean:skeleton-messy --fail-on=behavior --out-dir \"${TMPDIR:-/tmp}/skeleton-compare\"", - "agent:test:live:debug": "agent-test --suites-dir agent-suites --live --fail-on=behavior --debug", + "agent:test:direct": "agent-test --suites-dir agent-suites --fail-on=behavior", + "agent:test:direct:compare": "agent-test --suites-dir agent-suites --compare-pairs skeleton-clean:skeleton-messy --fail-on=behavior --out-dir \"${TMPDIR:-/tmp}/skeleton-compare\"", + "agent:test:direct:debug": "agent-test --suites-dir agent-suites --fail-on=behavior --debug", "agent:test:doctor": "agent-test --doctor", "agent:test:validate": "agent-test --validate-only --validate-paths --suites-dir agent-suites", "agent:evidence:aggregate": "bun scripts/agent-evidence/aggregate-compares.ts", diff --git a/refs/llm-harness.md b/refs/llm-harness.md index 247881a..15a4812 100644 --- a/refs/llm-harness.md +++ b/refs/llm-harness.md @@ -11,7 +11,7 @@ Implementation: [`agent-suites/`](../agent-suites/) via [`@post-print/agent-test ## Design - **Paired A/B:** `skeleton-clean` vs `skeleton-messy` (identical scenario names) -- **Primary signal:** live Cursor (`bun run agent:test:live:compare`) +- **Primary signal:** direct Cursor (`bun run agent:test:direct:compare`) - **Not in `bun run check`:** keeps deterministic CI fast ## Protocol (N=10) @@ -22,7 +22,7 @@ Implementation: [`agent-suites/`](../agent-suites/) via [`@post-print/agent-test ```bash set -a && source .env && set +a OUT="$TMPDIR/skeleton-compare-run-$(printf '%03d' "$i")" -bunx agent-test --suites-dir agent-suites --live \ +bunx agent-test --suites-dir agent-suites \ --compare-pairs skeleton-clean:skeleton-messy \ --fail-on=behavior --out-dir "$OUT" mkdir -p "agent-suites/evidence/runs/$(date +%Y-%m-%d)-run-$(printf '%03d' "$i")" diff --git a/scripts/agent-evidence/aggregate-compares.ts b/scripts/agent-evidence/aggregate-compares.ts index 7d7d1ca..3d18d6d 100644 --- a/scripts/agent-evidence/aggregate-compares.ts +++ b/scripts/agent-evidence/aggregate-compares.ts @@ -380,7 +380,7 @@ function buildSummaryMarkdown(input: BuildMarkdownInput): string { const lines: string[] = [ "# Behavioral evidence summary", "", - `**Source of truth for** aggregated Skeleton A/B live compares (\`skeleton-clean\` vs \`skeleton-messy\`).`, + `**Source of truth for** aggregated Skeleton A/B direct-agent compares (\`skeleton-clean\` vs \`skeleton-messy\`).`, "", ``, "", @@ -471,7 +471,7 @@ async function main(): Promise { const runs = await collectRuns(runsDir); if (runs.length === 0) { console.error(`No compare-report.json files under ${runsDir}`); - console.error("Deposit runs after: bun run agent:test:live:compare"); + console.error("Deposit runs after: bun run agent:test:direct:compare"); process.exit(1); } await writeSummaryOutputs(outDir, buildAggregateContext(runs), runs); diff --git a/scripts/agent-evidence/render-charts.ts b/scripts/agent-evidence/render-charts.ts index 5cb85a0..53e34b4 100644 --- a/scripts/agent-evidence/render-charts.ts +++ b/scripts/agent-evidence/render-charts.ts @@ -112,7 +112,7 @@ function passRateChart(summary: Summary): string { Pass rate by scenario - N=${summary.nRuns} paired live compares · higher is better + N=${summary.nRuns} paired direct compares · higher is better skeleton-clean