diff --git a/CHANGELOG.md b/CHANGELOG.md index c144e26..b3c5d67 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,34 @@ All notable changes to MARGINAL are documented here. The project follows Semantic Versioning. +## [Unreleased] + +### Added + +- opt-in, provider-neutral `DiminishingReturnDetector` with same-state/evidence-aware gain decay; +- `GovernanceTracker` for MARGINAL decision latency and external governance tokens, USD and latency; +- explicit counterfactual stop review through `Treasury.record_stop_review(...)` without inferring action causality from task outcomes; +- public-evaluation fields for repeated calls, governance overhead, reviewed stops and false stops; +- gross-versus-net savings and intervention status including Graceful Irrelevance through `pass_through`; +- governance evidence standard, Codex benchmark-readiness guide and Community Feedback Log; +- structured documentation information architecture by user intent. + +### Changed + +- public benchmark efficiency counts governance overhead in effective tokens/USD while retaining agent-only metrics and backward-compatible rows; +- `MarginalPolicy` can optionally discount or reject repeated semantic same-state work; +- `Treasury` records policy-decision latency and exposes governance evidence in summaries/traces; +- website and README now lead with a concrete illustrative trace and proof standard before architecture theory; +- roadmap now treats governance tax, false-stop rate, matched OFF/ON evaluation and pass-through as first-class success criteria; +- the 10-task Codex canary is explicitly classified as integration validation rather than public performance evidence. + +### Scientific limitations + +- diminishing-return thresholds are transparent heuristics until calibrated on representative engine telemetry; +- false stops require external review/counterfactual labels and are not automatically causal estimates; +- Graceful Irrelevance classifies the measured configuration, not the universal usefulness of MARGINAL; +- vendor-specific Codex integration and measured public savings remain future v0.3 evidence. + ## [0.2.0] - 2026-08-06 ### Added @@ -34,32 +62,27 @@ All notable changes to MARGINAL are documented here. The project follows Semanti ### Changed -- redesigned the repository README as a concise, SEO-oriented technical landing page; +- redesigned the repository README as a concise technical landing page; - added a dependency-free, responsive, accessible GitHub Pages product website; -- consolidated the product website, hero asset, and Killer Demo into one Pages deployment; -- documented website ownership, deployment, accessibility, privacy, and evidence guardrails. - -- `Decision` is backward compatible but now carries recommendation, mode, reason-code, uncertainty, confidence, and estimator metadata; -- `MarginalPolicy` now has a stable identity and supports both versioned and legacy custom estimators; -- `Treasury.summary()` includes mode, policy, estimator, observed overruns, failed settlements, outcomes, and estimator observations; -- child treasuries inherit the parent execution mode; -- trace events include execution mode plus policy and estimator identity; -- Decision Ledger records now declare their privacy profile; `local_full` remains backward compatible while strict profiles are opt-in; -- actual failed-call spend can be accounted without replacing the original execution exception; extraction failures conservatively settle the reserved estimate and remain chained as secondary errors; -- strict public-benchmark parsing rejects string-like booleans instead of silently coercing them; -- project description and documentation now consistently describe MARGINAL as a learning-loop foundation rather than only a static wrapper. +- consolidated product website, hero asset and Killer Demo into one Pages deployment; +- `Decision` carries recommendation, mode, reason-code, uncertainty, confidence and estimator metadata; +- `MarginalPolicy` has stable identity and supports versioned/legacy custom estimators; +- `Treasury.summary()` includes mode, policy, estimator, observed overruns, failed settlements, outcomes and estimator observations; +- child treasuries inherit parent execution mode; +- traces include execution mode plus policy and estimator identity; +- strict public-benchmark parsing rejects string-like booleans. ### Compatibility -- existing v0.1 constructors, enforced execution, `JsonlTraceSink`, synchronous and asynchronous wrappers, demos, and CLI commands remain supported; +- existing v0.1 constructors, enforced execution, `JsonlTraceSink`, synchronous/asynchronous wrappers, demos and CLI commands remain supported; - new dataclass fields have backward-compatible defaults; -- the runtime core continues to have zero mandatory dependencies. +- runtime core continues to have zero mandatory dependencies. ### Scientific limitations - historical estimates are observational and do not establish causal action value; - policy replay does not simulate unobserved trajectories or prove quality preservation; -- vendor-specific Codex, Claude Code, GitHub Copilot, and OpenCode adapters remain future milestones; +- vendor-specific Codex, Claude Code, GitHub Copilot and OpenCode adapters remain future milestones; - reference profiles are transparent defaults, not universal calibrations. ## [0.1.0] - 2026-08-04 @@ -68,12 +91,12 @@ All notable changes to MARGINAL are documented here. The project follows Semanti - provider-neutral `Action`, `Cost`, `Decision`, and `Allocation` value objects; - deterministic candidate ranking, authorization, reservation, settlement, and abort; -- hard budgets for tokens, direct USD, latency, and risk; -- pending reservations and hierarchical parent/child accounting; +- hard token, USD, latency and risk budgets; +- pending reservations and hierarchical accounting; - protected verification reserves; -- marginal-value policy with token, latency, and risk shadow prices; -- exact action and callable-input fingerprinting; -- synchronous and asynchronous guarded-call adapters; +- marginal-value policy with token, latency and risk shadow prices; +- exact action/callable-input fingerprinting; +- synchronous/asynchronous guarded-call adapters; - append-only JSONL traces and CLI reporting; -- synthetic benchmark, public comparison utility, and Killer Demo; -- Python 3.10–3.13 CI, CodeQL, packaging, and community documentation. +- synthetic benchmark, public comparison utility and Killer Demo; +- Python 3.10–3.13 CI, CodeQL, packaging and community documentation. diff --git a/MIGRATION_MANIFEST.json b/MIGRATION_MANIFEST.json new file mode 100644 index 0000000..e41ce12 --- /dev/null +++ b/MIGRATION_MANIFEST.json @@ -0,0 +1,48 @@ +{ + "package": "MARGINAL community evidence hardening overlay", + "target_repository": "SignalLayerLabs/Marginal", + "target_branch": "main", + "prepared_against_commit": "d6ab5c745f1a2ec19b2cb2395a1e6bfae66f2de5", + "created": "2026-08-07", + "package_type": "repository-overlay", + "rules": [ + "Apply to a clean checkout of the target repository.", + "Do not add this ZIP or the external Visual Studio prompt to Git.", + "Run scripts/reorganize_docs.py after overlay extraction.", + "Do not invent benchmark results or mark the Codex adapter as implemented.", + "Run all validation commands before committing." + ], + "moves": { + "docs/quickstart.md": "docs/getting-started/quickstart.md", + "docs/concepts.md": "docs/product/concepts.md", + "docs/architecture.md": "docs/product/architecture.md", + "docs/faq.md": "docs/product/faq.md", + "docs/api.md": "docs/reference/api.md", + "docs/integrations.md": "docs/integrations/overview.md", + "docs/benchmarking.md": "docs/evaluation/benchmarking.md", + "docs/public-benchmarks.md": "docs/evaluation/public-benchmarks.md", + "docs/research.md": "docs/evaluation/research.md", + "docs/privacy.md": "docs/operations/privacy.md", + "docs/website.md": "docs/operations/website.md", + "docs/governance.md": "docs/project/governance.md" + }, + "new_core_modules": [ + "src/marginal/controls/__init__.py", + "src/marginal/controls/diminishing.py", + "src/marginal/controls/governance.py" + ], + "replaced_core_files": [ + "src/marginal/__init__.py", + "src/marginal/cli.py", + "src/marginal/policy.py", + "src/marginal/public_eval.py", + "src/marginal/treasury.py" + ], + "communication_files": [ + "README.md", + "ROADMAP.md", + "CHANGELOG.md", + "site/index.html", + "site/styles.css" + ] +} diff --git a/PACKAGE_CHECKSUMS.sha256 b/PACKAGE_CHECKSUMS.sha256 new file mode 100644 index 0000000..e74d391 --- /dev/null +++ b/PACKAGE_CHECKSUMS.sha256 @@ -0,0 +1,31 @@ +0ffc751f0c3ac86b0cb1a201c4ae3773ebfcc30b8672fccdcda5df6ea3cc35a1 ./CHANGELOG.md +54940f6294de06328fe6826f81a51ab86f81df1a1b0da15626b6e7480631d9b4 ./MIGRATION_MANIFEST.json +cdee908544415e8c05c4ff308dc30c8884f25663c3676684077213bcefb39c57 ./PACKAGE_README.md +bed65f82a0fb147f7eee6f30d67aa1c1a07d5a8503a1e1fff4a1dc12fee2c6a7 ./README.md +c1f177cf4291ebef10d4023c8d4144bed109a9e97cbeca18f09ff0772bb16097 ./ROADMAP.md +878ebab0f0d6e8eeddebd3aaea8aa48ce20ddf5f98720ca203d4e207bdeb9a92 ./docs/evaluation/governance-evidence.md +5474b9a310e857c213f578cfe8a1e743a38b33ba5209bfc619dabb05d535de48 ./docs/index.md +c74df4b2fdc280eae409e9d49f21d7e4bfa68c88003a0e02d9e8eb35d1fdef51 ./docs/integrations/codex-benchmark-readiness.md +4503e66d5b74e51233a1306a43c658aa948df0199898c9aa5ef846dbec6cff2f ./docs/operations/website-review-2026-08-07.md +c0e591c130bb6de3e269339f3105af93f98f721442161d39520df762f664350e ./docs/project/community-feedback.md +d9962b32c01d847ff91709b129cc7993d0beccb5111f9afe394615a8de235c13 ./docs/superpowers/plans/2026-08-07-community-evidence-hardening.md +fba3bedaebedc8e898d2ba0caccbac7de5b2d3c0a2abf976cb4c24c7e676d487 ./docs/superpowers/specs/2026-08-07-community-evidence-hardening-design.md +300b13720c873f38fdb566ba2fa2b9b8a18a9cdc52253bbbaae6cb1f016540e0 ./scripts/reorganize_docs.py +07a5f282687514233e809c237c6171fc3a3e29ca1ef3d88e493db20d04ef6fcc ./scripts/validate_community_hardening.py +32ca56be9daaab6f22fcb3d817fade2e7260302d302beefdb854edd12d3eeff7 ./scripts/validate_readme_pages.py +94a26f0f90fd09babaf13b7cdddf7d03351745f2dbdd14fe5646a80c0dc4c60b ./site/index.html +3888918324e1351e749437806f7b37937c3c8fe48304eefde7ea1a4b729e1f0b ./site/styles.css +8b505e6dcc12b6679c234d39c9f344cfe61295091a790835f3bdf7c1e396a70c ./src/marginal/__init__.py +ceff13a46267520b6a62864967ab275d87af52bf82b941ffa078947d08dcd3f1 ./src/marginal/cli.py +0fa38295e579e012b4b24ebb5e06cad89af6a0549f457d0bfbd41f60401551eb ./src/marginal/controls/__init__.py +ad8ca98aac20adbdb55363905f4f197ddc9dc04a77b9cdecd8520006da038dab ./src/marginal/controls/diminishing.py +9cc63951146cfd68b8e7923035d805c03e9db3509f302d8ca66570bcb4a01a2c ./src/marginal/controls/governance.py +992d48c98f6e0f421745939713eac8b4459f7f49cf9baad4128efe800adb9ce1 ./src/marginal/policy.py +d3357ef4a93cba5f92eb161c7eea417fd063b7936e175cefbc2f7d08eea5165e ./src/marginal/public_eval.py +16095e6c42b0e5c8f41b23a7f26a929599ac726f1d39a21c6aab9c89888b636f ./src/marginal/treasury.py +37dcd6e3d0672c1da8db67cb34c868ac59257f9ec570220bcd835f9ee979046e ./tests/controls/test_diminishing.py +1b30b9ad2bef94ac0be34b5731ef7647b38289bf17c16ccac36648e63ea14c48 ./tests/controls/test_governance.py +0c633fd64787e45794a8d36a66ea53cb0b5adb84695c87538bbe8506488bf990 ./tests/controls/test_policy_diminishing.py +c4da184c1017e2407a6a72cfa784cbb04ff048b0e2c393ddcd43f8fde4d93afb ./tests/controls/test_treasury_governance.py +d941eb513a4a191a41b7e4dae28b700b48e2c70ef5fc4106b5147b5289090ae7 ./tests/evaluation/test_cli_public_eval_governance.py +244f5b01bcafb56bf864f092ddb03eacb657dcb160b48062b0815f3365017e87 ./tests/evaluation/test_public_eval_governance.py diff --git a/PACKAGE_MANIFEST.json b/PACKAGE_MANIFEST.json index aa212ac..a79ab6c 100644 --- a/PACKAGE_MANIFEST.json +++ b/PACKAGE_MANIFEST.json @@ -9,7 +9,7 @@ "VISUAL_STUDIO_COMMIT_PROMPT.md", "docs/superpowers/plans/2026-08-06-readme-pages.md", "docs/superpowers/specs/2026-08-06-readme-pages-design.md", - "docs/website.md", + "docs/operations/website.md", "scripts/validate_readme_pages.py", "site/404.html", "site/app.js", diff --git a/PACKAGE_README.md b/PACKAGE_README.md new file mode 100644 index 0000000..8c6183c --- /dev/null +++ b/PACKAGE_README.md @@ -0,0 +1,29 @@ +# MARGINAL Community Hardening Overlay + +This ZIP is a repository overlay prepared for `SignalLayerLabs/Marginal` `main` at commit `d6ab5c745f1a2ec19b2cb2395a1e6bfae66f2de5`. + +## What it contains + +- model-independent diminishing-return control; +- governance-tax and explicit false-stop accounting; +- net-value public benchmark reporting and CLI gates; +- evidence-first README / GitHub Pages rewrite; +- community feedback decision log; +- Codex benchmark-readiness specification; +- documentation reorganization tooling; +- focused tests and validators. + +## Important + +This package does **not** contain a completed Codex adapter and does not contain benchmark results. It prepares the evidence/control layer for v0.3. + +The Visual Studio upload prompt is intentionally distributed separately and must not be committed. + +## Apply order + +1. Start from a clean checkout of `SignalLayerLabs/Marginal` `main`. +2. Extract this ZIP over the repository root, replacing matching files. +3. Run `python scripts/reorganize_docs.py` once. +4. Review `MIGRATION_MANIFEST.json` and the resulting `git diff`. +5. Run the focused validators and the full repository quality gate described in the external Visual Studio prompt. +6. Commit the source changes only; do not commit this ZIP or the external prompt. diff --git a/README.md b/README.md index 032edf0..51116bd 100644 --- a/README.md +++ b/README.md @@ -1,18 +1,19 @@
-MARGINAL — compute governance and token optimization for AI agents +MARGINAL — compute governance for AI agents # MARGINAL -### Compute governance and token optimization for AI agents +### Compute governance that has to justify its own cost -**MARGINAL helps AI agents decide whether the next model call, tool call, search, retry, review, or sub-agent is worth its compute cost.** +**MARGINAL evaluates whether the next model call, tool call, retry, verification, review, or sub-agent is likely to add enough value to justify its compute — and now measures whether MARGINAL's own intervention was worth it.** Open source · Local first · Provider neutral · Zero mandatory runtime dependencies [Website](https://signallayerlabs.github.io/Marginal/) · -[Quickstart](docs/quickstart.md) · -[Architecture](docs/architecture.md) · +[Quickstart](docs/getting-started/quickstart.md) · +[Architecture](docs/product/architecture.md) · +[Evidence standard](docs/evaluation/governance-evidence.md) · [Roadmap](ROADMAP.md) · [Contributing](CONTRIBUTING.md) @@ -21,80 +22,151 @@ Open source · Local first · Provider neutral · Zero mandatory runtime depende [![Release](https://img.shields.io/github/v/release/SignalLayerLabs/Marginal?style=flat-square)](https://github.com/SignalLayerLabs/Marginal/releases) [![Python 3.10–3.13](https://img.shields.io/badge/python-3.10--3.13-blue.svg?style=flat-square)](https://www.python.org/) [![License: Apache-2.0](https://img.shields.io/badge/license-Apache--2.0-green.svg?style=flat-square)](LICENSE) -[![Runtime dependencies: 0](https://img.shields.io/badge/runtime%20dependencies-0-brightgreen.svg?style=flat-square)](pyproject.toml)
--- -## Why MARGINAL +## The problem in one trace -Most agent runtimes ask: +Coding agents can spend compute on actions whose incremental value is unclear or diminishing. The useful failure mode is not "GPT-5.6 is wasteful"; it is **repeated work against unchanged state that produces no new evidence**. -> **Can this action run?** +```text +Illustrative trace — not a benchmark -MARGINAL adds the economic question: +Agent proposes: read README.md + → state changes: knowledge acquired -> **Is this action worth funding now?** +Agent proposes: verify README.md + → evidence acquired -It evaluates expected improvement against tokens, direct cost, latency, risk, remaining budget, verification reserves, and prior evidence. It then records what actually happened so future policies can be evaluated against real outcomes. +Agent proposes: verify README.md again + → same semantic action + → same workspace state + → no new evidence -```text -observe decisions - ↓ -measure actual cost and outcomes - ↓ -estimate marginal value - ↓ -allocate compute - ↓ -measure calibration and regret - ↓ -improve the policy +Agent proposes: verify README.md again + → expected marginal gain is now lower + → MARGINAL can recommend stopping the repetition ``` -The goal is not to make an agent merely cheaper. It is to make autonomous work **economically disciplined, measurable, and auditable**. +The mechanism is model independent. A future model may repeat less often, a different provider may repeat more often, and some tasks genuinely need multiple verification passes. MARGINAL should respond to the evidence rather than assume every repeat is waste. Shadow Mode remains the safe default for unvalidated integrations, while the Decision Ledger preserves versioned evidence for audit and replay. + +## What changed after community review + +Two early community criticisms exposed useful product tests: + +1. **"Will the next model make this redundant?"** — valid as a design challenge. MARGINAL must remain useful across model generations, but it must also be able to conclude that an already-efficient agent does not need intervention. +2. **"Show the benchmark with and without it."** — valid. Performance claims require matched OFF/ON runs, not a synthetic demo or a persuasive website. +3. **"The website is too abstract."** — partially accepted. The concepts are real, but the explanation should start with an observable failure mode and proof standard before theory. +4. **"Providers may intentionally waste tokens."** — rejected as unsupported speculation. MARGINAL does not need that claim to justify independent compute governance. + +The full decision log is in [Community feedback](docs/project/community-feedback.md). -MARGINAL also supports Shadow Mode, where recommendations can be observed without blocking execution, letting teams collect evidence before enforcement. +## MARGINAL must earn its own compute -## How it works +A governor that saves 20% of agent tokens while adding 25% overhead is not an optimization. + +MARGINAL therefore separates: + +- **agent workload cost** — model/tool tokens, USD, latency and calls; +- **governance tax** — tokens, USD and latency introduced by MARGINAL itself; +- **gross savings** — agent-only reduction; +- **net savings** — reduction after governance tax; +- **quality** — verified task outcomes, regressions and recoveries; +- **false stops** — reviewed cases where a deny recommendation would have prevented a helpful action. + +The public evaluator treats net metrics as the evidence surface. A positive-looking gross number cannot hide governance overhead. + +### Graceful Irrelevance + +If a stronger model, better agent runtime, or simple task already behaves efficiently, MARGINAL should be able to produce: ```text -Agent proposes an action - ↓ -MARGINAL estimates value and total cost - ↓ -ALLOW · DENY · RECOMMEND · SHADOW - ↓ -Budget is reserved before execution - ↓ -Actual usage and verified outcome are settled - ↓ -Versioned evidence is written to the Decision Ledger +intervention.status = pass_through +``` + +That is not a failed product demo. It means the governor did not demonstrate enough net value to justify inserting itself into that workload. + +## State-aware diminishing returns + +The new `DiminishingReturnDetector` is intentionally opt-in while evidence is collected. It does not special-case GPT, Markdown files, Codex, or any provider. + +```python +from marginal import ( + DiminishingReturnConfig, + DiminishingReturnDetector, + MarginalPolicy, +) + +policy = MarginalPolicy( + diminishing_detector=DiminishingReturnDetector( + DiminishingReturnConfig( + gain_decay=0.5, + max_same_state_repeats=2, + ) + ) +) +``` + +The detector discounts expected gain only when the same semantic action has already executed against the same observable state without new evidence. A changed state or changed evidence resets the pressure. Missing state fails open instead of inventing certainty. Privacy guidance remains explicit through SAFE_TELEMETRY and AGGREGATE_EXPORT conventions. + +This is designed to complement, not replace, the Universal Agent Protocol's existing exact and state-aware deduplication scopes. + +## Governance accounting and false-stop review + +`Treasury` records local decision latency and exposes explicit external overhead accounting for adapter-side work: + +```python +treasury.record_governance_overhead( + tokens=120, + usd=0.002, + latency_ms=40, +) ``` -MARGINAL can sit around model calls, tool calls, searches, tests, reviewers, retries, and sub-agents. The core remains engine-neutral; thin adapters translate native events from coding agents into one universal protocol. +A false stop is never inferred from "the task eventually succeeded." It requires an explicit review or counterfactual label: + +```python +treasury.record_stop_review( + denied_action, + would_have_helped=True, +) +``` + +This distinction matters. Correlation between an action and final task success is not causal proof, and an optimizer should not mark itself correct simply because the final answer happened to pass. + +## Proof standard -## What makes it different +A MARGINAL benchmark should compare the **same agent, model, prompt, tools, limits, task order and verifier** with MARGINAL OFF and ON. -| Capability | What it adds | +| Evidence | Why it matters | |---|---| -| **Transactional accounting** | Atomic reserve, settle, abort, overrun, hierarchy, and verification reserves—not a token counter wrapper. | -| **Shadow-to-enforce lifecycle** | Observe recommendations safely before allowing the policy to block work. | -| **Learning Loop Foundation** | Versioned estimators, outcomes, replay, and evidence for progressively better allocation. | -| **Privacy by design** | Local ledgers, pseudonymous telemetry, aggregate exports, and no prompt/output logging by default. | -| **Universal agent protocol** | One core for Codex, Claude Code, GitHub Copilot, OpenCode, and future compatible runtimes. | -| **Scientific honesty** | Synthetic demonstrations are labeled as demonstrations; public claims require measured telemetry and preserved quality. | +| Verified resolve rate | Prevents cheaper-but-worse optimization | +| Effective tokens / resolved task | Primary efficiency metric after governance tax | +| Gross vs net token savings | Shows whether MARGINAL pays for itself | +| USD and latency | Token savings can shift cost elsewhere | +| Tool and repeated calls | Shows what behavior actually changed | +| Regressions and recoveries | Makes quality movement inspectable | +| Reviewed false stops | Measures harmful deny recommendations | +| Bootstrap uncertainty / repeated runs | Separates signal from run variance | +| Intervention status | `supported`, `pass_through`, `quality_regression`, or `false_stop_risk` | + +A 10-task canary is an **integration check**, not public performance evidence. Larger matched evaluation should be preregistered before headline claims are made. + +SWE-bench Pro can be one requested evaluation surface, but no single benchmark is treated as ground truth. Dataset version, exclusions, verifier behavior and known task-quality limitations must be recorded alongside results. + +Read the [benchmark protocol](docs/evaluation/public-benchmarks.md) and [governance evidence standard](docs/evaluation/governance-evidence.md). ## Install -Install the tagged release: +Current v0.2 install target: ```bash pip install "marginal-ai @ git+https://github.com/SignalLayerLabs/Marginal.git@v0.2.0" ``` -For development: +Development checkout: ```bash git clone https://github.com/SignalLayerLabs/Marginal.git @@ -102,6 +174,8 @@ cd Marginal python -m pip install -e ".[dev]" ``` +The upcoming Codex reference integration remains a roadmap milestone. It is not presented here as already available. + ## Quickstart ```python @@ -114,7 +188,7 @@ treasury = Treasury( verification_reserve_tokens=10_000, ), policy=build_policy("balanced"), - mode="enforce", + mode="shadow", ) result = budgeted_call( @@ -130,129 +204,7 @@ result = budgeted_call( ) ``` -When enforcement denies the action, the wrapped function is not called. Approved estimates are reserved immediately, preventing concurrent actions from oversubscribing the same treasury. - -Start with [`shadow` mode](docs/quickstart.md) when collecting evidence for a new workflow. - -## Execution modes - -| Mode | Applied behavior | Best for | -|---|---|---| -| `shadow` | Executes every proposed action and records the recommendation | Safe observation and calibration | -| `recommend` | Executes while surfacing the policy recommendation | Human or agent advisory workflows | -| `enforce` | Applies allow/deny and hard-budget decisions | Validated production control | - -`fund_best(...)` remains an active selection API in every mode because the caller is explicitly asking MARGINAL to choose among alternatives. - -## The Learning Loop Foundation - -MARGINAL v0.2 moves beyond manually supplied expected gain without pretending causal inference is already solved. - -The current foundation provides: - -- versioned `ValueEstimator` identities and configurations; -- contextual observations by engine, phase, task type, language, and model; -- uncertainty, confidence, sample size, and provenance; -- task-level verified outcomes; -- separately supplied action-level realized gain; -- policy replay over recorded actions; -- deterministic Decision Ledger records for audit and comparison. - -A successful task does **not** prove that every action in its trajectory caused success. MARGINAL keeps task outcomes and action attribution separate so correlation is not mislabeled as causal value. - -Read [Learning and replay](docs/concepts.md) and [Benchmarking](docs/benchmarking.md). - -## Universal Agent Runtime - -The universal runtime gives engine adapters one shared contract: - -```python -from marginal import AgentAction, AgentCapabilities, Cost, UniversalRuntime - -runtime = UniversalRuntime( - treasury, - engine="opencode", - session_id="session-1", - task_id="task-42", - capabilities=AgentCapabilities( - observe_model_usage=True, - block_actions=True, - record_outcomes=True, - ), -) - -decision = runtime.before_action( - AgentAction( - action_id="read-1", - name="read complete repository", - kind="file_read", - estimated_cost=Cost(tokens=8_000), - expected_gain=0.03, - state_hash="workspace-sha", - phase="diagnose", - deduplication_scope="once_per_state", - ) -) -``` - -The protocol defines consistent events, capabilities, usage fields, decisions, settlement, and outcome contracts. Vendor-specific adapters are developed as thin integrations; they do not duplicate the economic policy. - -### Integration status - -| Environment | Current status | Intended capability | -|---|---|---| -| Core Python runtime | Available | Full allocation, accounting, ledger, replay | -| Universal Agent Protocol | Available | Adapter contract and capability negotiation | -| Codex | Roadmap | Reference integration and measured benchmark | -| OpenCode | Roadmap | Open-source adapter and research environment | -| Claude Code | Roadmap | Hook-based integration | -| GitHub Copilot CLI / coding agent | Roadmap | Integration where official control surfaces permit | - -See the [full roadmap](ROADMAP.md). Planned adapters are not presented as completed integrations. - -## Privacy by design - -Not recording prompts is not enough: identifiers, action names, model names, metadata, verifier details, and exception text can still expose sensitive information. - -MARGINAL therefore separates operational evidence from shareable telemetry: - -| Profile | Purpose | -|---|---| -| `LOCAL_FULL` | Full operational ledger on a trusted local filesystem | -| `SAFE_TELEMETRY` | Removes free text and pseudonymizes identifiers with a local key | -| `AGGREGATE_EXPORT` | Produces generalized grouped rows with no identifiers or timestamps | - -Aggregate exports suppress groups smaller than five records by default to reduce re-identification risk. Pseudonymization is not anonymization; exports must still be reviewed before sharing. - -Read the [privacy model](docs/privacy.md) and [security policy](SECURITY.md). - -## Evidence, not hype - -MARGINAL includes a deterministic Killer Demo that fixes the same code defect with a baseline workflow and a MARGINAL-funded workflow. - -```bash -marginal killer-demo --output-dir demo-output -``` - -The demo uses declared action-cost estimates. It is useful for understanding allocation behavior, but it is **not provider telemetry and not a production savings claim**. - -Public evaluation compares matched runs with the same agent, model, prompt, tools, limits, task order, and verifier: - -```bash -marginal public-eval baseline.jsonl marginal.jsonl -``` - -The report covers: - -- resolve-rate delta and quality non-inferiority; -- total and per-resolved-task token cost; -- USD, latency, and tool calls; -- regressions and recoveries; -- bootstrap uncertainty. - -A token reduction without preserved verified outcomes is not considered a successful result. - -See [Public benchmark protocol](docs/public-benchmarks.md) and [Killer Demo results](demos/killer-demo/RESULTS.md). +Start with `shadow` for a new integration. Promote controls only after representative evidence shows they preserve quality. ## Architecture @@ -264,51 +216,47 @@ Codex · Claude Code · Copilot · OpenCode · others │ Universal Agent Protocol │ - ┌────────────────┼────────────────┐ - │ │ │ - Value Estimator Treasury Decision Ledger - │ │ │ - └────────── Decision Policy ──────┘ + ┌──────────────────┼──────────────────┐ + │ │ │ +Value Estimator Treasury Decision Ledger + │ │ │ +Diminishing Return Governance Tax Outcomes / Replay + └──────────────────┼──────────────────┘ │ - observe · recommend · enforce + observe · recommend · enforce ``` -MARGINAL is deliberately modular: +The engine-specific adapter owns native interception and telemetry. The core owns economic policy, accounting and evidence semantics. That separation is what lets MARGINAL survive changes in model/provider behavior without accumulating vendor-specific patches. -- the **core** evaluates and accounts; -- the **runtime** coordinates sessions and adapters; -- the **ledger** preserves evidence; -- privacy exports transform data for controlled sharing; -- external frameworks retain ownership of execution. +## Project status -Read the [architecture](docs/architecture.md) and [API reference](docs/api.md). +`v0.2.0` provides the Learning Loop Foundation, privacy profiles, Universal Agent Protocol, versioned evidence and replay. The community-hardening work prepares the core evidence model for **v0.3 — Codex Reference Integration**. -## Project status +The next milestone must answer a falsifiable question: -`v0.2.0` provides the Learning Loop Foundation, privacy profiles, Universal Agent Protocol, versioned evidence, policy replay, and a dependency-free core. +> Does MARGINAL reduce effective compute per verified successful Codex task after its own overhead, without exceeding the preregistered quality and false-stop constraints? -The active milestone is **v0.3 — Codex Reference Integration**: measured token telemetry, matched baseline runs, a 10-task canary, and a preregistered public evaluation before broader enforcement claims. +If the answer is no, the result should be published as no demonstrated benefit for that configuration. [View the roadmap →](ROADMAP.md) ## Documentation -| Start here | Deep dives | +| Area | Start here | |---|---| -| [Quickstart](docs/quickstart.md) | [Architecture](docs/architecture.md) | -| [Core concepts](docs/concepts.md) | [API reference](docs/api.md) | -| [Integration guide](docs/integrations.md) | [Privacy profiles](docs/privacy.md) | -| [Benchmarking](docs/benchmarking.md) | [Public benchmark protocol](docs/public-benchmarks.md) | -| [Roadmap](ROADMAP.md) | [Security](SECURITY.md) | -| [Changelog](CHANGELOG.md) | [Governance](docs/governance.md) | +| Getting started | [Quickstart](docs/getting-started/quickstart.md) | +| Product model | [Concepts](docs/product/concepts.md) · [Architecture](docs/product/architecture.md) | +| Integrations | [Integration overview](docs/integrations/overview.md) · [Codex benchmark readiness](docs/integrations/codex-benchmark-readiness.md) | +| Evaluation | [Benchmarking](docs/evaluation/benchmarking.md) · [Public benchmarks](docs/evaluation/public-benchmarks.md) · [Governance evidence](docs/evaluation/governance-evidence.md) | +| Reference | [API](docs/reference/api.md) | +| Operations | [Privacy](docs/operations/privacy.md) · [Website](docs/operations/website.md) | +| Project | [Governance](docs/project/governance.md) · [Community feedback](docs/project/community-feedback.md) | -The [project website](https://signallayerlabs.github.io/Marginal/) provides the product-level overview; GitHub remains the source of truth for code, evidence, releases, and technical documentation. +GitHub is the source of truth for code, evidence, releases and technical documentation. ## Contributing -Contributions are welcome across the core, protocol, adapters, privacy, benchmarks, and documentation. - -Before opening a pull request: +Contributions are welcome, including criticism. A proposed performance improvement should include the evidence that could prove it wrong. ```bash ruff format --check . @@ -319,19 +267,7 @@ python -m build python -m twine check dist/* ``` -Read [`CONTRIBUTING.md`](CONTRIBUTING.md) and the [governance model](docs/governance.md). - -## Citation - -```bibtex -@software{marginal2026, - title = {MARGINAL: Compute Governance for AI Agents}, - author = {SignalLayer Labs and contributors}, - year = {2026}, - url = {https://github.com/SignalLayerLabs/Marginal}, - license = {Apache-2.0} -} -``` +Read [`CONTRIBUTING.md`](CONTRIBUTING.md). ## License diff --git a/ROADMAP.md b/ROADMAP.md index 8819925..e6bb0d7 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -1,45 +1,50 @@ # MARGINAL Roadmap -MARGINAL's North Star is simple: +MARGINAL's North Star is: -> **Install MARGINAL once, keep using your AI development agent normally, and reduce avoidable token consumption without sacrificing verified quality.** +> **Reduce avoidable compute per verified successful task while accounting for the cost and mistakes of the governor itself.** -MARGINAL is being developed as one universal, local compute-governance product for the main AI development agents. A single core, protocol, installer, policy system, learning loop, and reporting experience will support Codex, Claude Code, GitHub Copilot, OpenCode, and future compatible runtimes through thin engine adapters. +The project is one universal, local compute-governance product for AI development agents. A shared core, protocol, policy system, evidence model and reporting layer support engine-specific adapters without duplicating economic logic. -This roadmap is milestone-driven rather than date-driven. It communicates product direction and measurable outcomes. GitHub Issues and pull requests should track implementation-level work. +This roadmap is milestone-driven rather than date-driven. GitHub Issues and pull requests should track implementation-level work. ## Product principles -1. **One product:** one core, one protocol, one installer, and one user experience. -2. **Thin adapters:** engine-specific behavior stays outside the decision core. -3. **Quality first:** token reduction is valuable only when verified quality is preserved. -4. **Measured claims:** public savings claims must use real runtime or provider telemetry. -5. **Learning without overclaiming:** observational associations are not described as causal value. +1. **One product:** one core, protocol, evidence model and user experience. +2. **Thin adapters:** engine-specific interception stays outside the decision core. +3. **Quality first:** lower compute is valuable only when verified quality remains inside the preregistered constraint. +4. **Measured claims:** public savings claims require matched real-runtime telemetry. +5. **Learning without overclaiming:** observational associations are not causal proof. 6. **Local first:** prompts and source code are not uploaded or logged by default. -7. **Simple installation:** supported agents should require no manual code changes. -8. **Transparent capabilities:** observe-only and enforcement integrations must be clearly distinguished. -9. **Small core:** the provider-neutral core keeps zero mandatory runtime dependencies. +7. **Simple installation:** supported agents should require no manual project-code changes. +8. **Transparent capabilities:** observe-only and enforcement integrations are clearly distinguished. +9. **Small core:** the provider-neutral runtime keeps zero mandatory dependencies. +10. **Self-accounting:** governance tokens, USD and latency are first-class evidence, not hidden overhead. +11. **Graceful irrelevance:** if MARGINAL does not demonstrate positive net value for a workload, pass-through is a valid outcome. +12. **False-stop visibility:** harmful deny recommendations are explicitly reviewed and never inferred away by aggregate success. +13. **Community pressure testing:** criticism can change the roadmap when it creates a stronger falsification test; unsupported speculation does not become product doctrine. ## Status legend | Status | Meaning | |---|---| -| **Planned** | Scope is defined, but implementation has not started. | +| **Planned** | Scope is defined; implementation has not started. | | **In progress** | Implementation is actively underway. | -| **Validation** | Implementation exists, but final CI, release, integration, or benchmark evidence is pending. | -| **Complete** | All exit criteria have been met and supporting evidence is available. | +| **Validation** | Implementation exists; final CI, release, integration or evidence is pending. | +| **Complete** | Exit criteria and supporting evidence are satisfied. | ## Milestones at a glance | Milestone | Status | Primary outcome | |---|---|---| -| **v0.1 — Reference Allocator Foundation** | Complete | Provider-neutral allocation, accounting, tracing, demos, and first release | -| **v0.2 — Learning Loop Foundation** | Validation | Universal protocol, non-blocking observation, versioned evidence, outcomes, replay, and estimators | -| **v0.3 — Codex Reference Integration** | Planned | Real Codex integration and measured paired benchmark | -| **v0.4 — Multi-Engine Developer Preview** | Planned | Codex, OpenCode, Claude Code, and GitHub Copilot compatibility | -| **v0.5 — One-Command Universal Installation** | Planned | Automatic detection, installation, diagnostics, and rollback | -| **v0.6 — Adaptive and Causal Allocation** | Planned | Calibrated learning, context-carry economics, exploration, and regret measurement | -| **v0.7 — Ecosystem and Operational Scale** | Planned | Additional engines, persistent runtimes, observability, and team controls | +| **v0.1 — Reference Allocator Foundation** | Complete | Provider-neutral allocation, accounting, tracing and first release | +| **v0.2 — Learning Loop Foundation** | Validation | Universal protocol, non-blocking observation, versioned evidence, privacy and replay | +| **Community hardening** | In progress | Governance tax, false-stop accounting, diminishing-return control and clearer evidence UX | +| **v0.3 — Codex Reference Integration** | Planned | One-command target, real telemetry and first matched public benchmark | +| **v0.4 — Multi-Engine Developer Preview** | Planned | Shared core across materially different coding agents | +| **v0.5 — One-Command Universal Installation** | Planned | Detection, installation, diagnostics and rollback across engines | +| **v0.6 — Adaptive and Causal Allocation** | Planned | Calibrated learning, exploration and stronger identification strategies | +| **v0.7 — Ecosystem and Operational Scale** | Planned | Persistence, observability, team controls and more engines | --- @@ -47,31 +52,9 @@ This roadmap is milestone-driven rather than date-driven. It communicates produc **Status:** Complete -**Objective:** Establish a small, auditable, provider-neutral compute-allocation core. +Delivered provider-neutral `Action`, `Cost`, `Decision` and `Allocation` primitives; hard token/USD/latency/risk budgets; reservations and settlement; hierarchical treasuries; verification reserves; marginal-value policy; duplicate protection; guarded-call adapters; common usage extraction; JSONL traces; synthetic benchmark; Killer Demo; public comparison utility; Python 3.10–3.13 CI and project documentation. -### Delivered - -- [x] Provider-neutral `Action`, `Cost`, `Decision`, and `Allocation` primitives -- [x] Hard token, USD, latency, and risk budgets -- [x] Atomic reservation, settlement, abort, and overrun accounting -- [x] Hierarchical treasuries and protected verification reserves -- [x] Deterministic marginal-value policy and candidate ranking -- [x] Exact duplicate and pending-action protection -- [x] Synchronous and asynchronous guarded-call adapters -- [x] Common LLM usage extraction -- [x] Append-only JSONL decision traces and CLI reporting -- [x] Synthetic benchmark and end-to-end Killer Demo -- [x] Public benchmark comparison utility -- [x] Python 3.10–3.13 CI, CodeQL, packaging, and project documentation - -### Exit criteria - -- [x] `v0.1.0` released -- [x] CI passes across supported Python versions -- [x] Synthetic claims are clearly separated from measured production claims -- [x] Core remains free of mandatory runtime dependencies - -Evidence: [`CHANGELOG.md`](CHANGELOG.md), [`docs/architecture.md`](docs/architecture.md), and [`demos/killer-demo`](demos/killer-demo/RESULTS.md). +**Exit criteria:** released, CI green, synthetic claims separated from measured claims, zero mandatory runtime dependencies. --- @@ -79,226 +62,186 @@ Evidence: [`CHANGELOG.md`](CHANGELOG.md), [`docs/architecture.md`](docs/architec **Status:** Validation -**Objective:** Create the shared, versioned learning-loop foundation that lets every supported development agent use the same MARGINAL decisions, evidence model, accounting, and safety guarantees. - -### Delivered in the release candidate - -- [x] Publish MARGINAL Universal Agent Protocol v1 -- [x] Define normalized event, decision, capability, outcome, token-usage, and ledger schemas -- [x] Add capability negotiation for observe, modify, deny, stop, and verification control -- [x] Add additive token usage v2 for uncached input, cached input, output, reasoning, and total tokens -- [x] Add Decision Ledger v2 with run, task, trajectory, action, policy, estimator, engine, and model identity -- [x] Classify evidence fields as safe-by-default, pseudonymous, or potentially sensitive -- [x] Add `LOCAL_FULL` and keyed `SAFE_TELEMETRY` operational ledger profiles -- [x] Add separate grouped `AGGREGATE_EXPORT` output with no identifiers or timestamps -- [x] Suppress aggregate groups smaller than five records by default with a configurable threshold -- [x] Publish recursively strict JSON Schemas for safe event-level telemetry and aggregate exports -- [x] Add local 256-bit key generation, restrictive permission checks, race-safe exports, and safe export CLI -- [x] Add strict ledger parsing, monotonic sequence validation, and task/outcome correlation checks -- [x] Add state-aware fingerprints and configurable deduplication scopes -- [x] Add `shadow`, `recommend`, and `enforce` operating modes -- [x] Preserve concurrent Shadow Mode observations with separate reservation identities -- [x] Add explicit failed-action settlement for measured, estimated, and unavailable usage -- [x] Keep failed actions retryable while accounting for consumed resources -- [x] Add conservative fallback settlement when failure usage extraction itself fails -- [x] Add Quality First, Balanced, Token Saver, and Strict Budget reference profiles -- [x] Add a provider-neutral local `UniversalRuntime` -- [x] Add explicit task outcomes and separate action-level realized-gain observations -- [x] Add versioned estimator identities, uncertainty, confidence, sample size, provenance, and registry -- [x] Add deterministic training-data fingerprints for online observations -- [x] Add non-causal policy replay and CLI ledger validation/reporting -- [x] Add protocol and schema conformance tests, executable examples, and aligned documentation -- [x] Preserve the dependency-free provider-neutral runtime core +The v0.2 release candidate adds: -### Exit criteria +- Universal Agent Protocol v1 and capability negotiation; +- normalized action, decision, outcome, token-usage and ledger schemas; +- additive uncached/cached/output/reasoning token accounting; +- `shadow`, `recommend` and `enforce` modes; +- Decision Ledger v2 with engine/model/task/trajectory identity; +- `LOCAL_FULL`, `SAFE_TELEMETRY` and `AGGREGATE_EXPORT` privacy boundaries; +- state-aware fingerprints and deduplication scopes; +- failed-action settlement and measured overruns; +- quality-first/balanced/token-saver/strict-budget profiles; +- versioned estimators, uncertainty, confidence and provenance; +- task outcomes separated from action-level realized gain; +- non-causal replay and ledger/reporting CLI support. -- [x] Protocols and schemas are versioned and documented. -- [x] Privacy profiles, field classification, key handling, small-group suppression, export boundaries, and limitations are documented. -- [x] The provider-neutral reference runtime passes focused protocol and lifecycle tests. -- [x] Shadow Mode can observe complete action lifecycles without blocking caller behavior. -- [x] Existing v0.1 constructors and enforced execution paths remain covered by regression tests. -- [x] The package metadata and public documentation describe the implemented v0.2 behavior consistently. -- [x] The runtime core still has zero mandatory dependencies. -- [ ] Ruff, mypy strict, the full repository test suite, package build, and Twine validation pass in the canonical GitHub checkout and CI. -- [ ] `v0.2.0` is tagged and released from the canonical repository. +### Remaining exit criteria -The release remains in **Validation** until the final two exit criteria are satisfied. Vendor-specific adapters and measured production savings are intentionally not part of v0.2. +- [ ] Ruff, mypy strict, full tests, package build and Twine validation pass in canonical CI. +- [ ] `v0.2.0` is tagged/released from the canonical repository. + +Vendor-specific adapters and measured production savings are intentionally outside v0.2. --- -## v0.3 — Codex Reference Integration +## Community hardening — Net-value evidence layer -**Status:** Planned +**Status:** In progress -**Objective:** Integrate MARGINAL into Codex and produce the first real paired benchmark with measured token telemetry. +**Objective:** turn early community criticism into falsifiable product requirements without introducing model-specific patches or unsupported narratives. ### Deliverables -- [ ] Build the Codex adapter against the Universal Agent Protocol -- [ ] Support `marginal install codex` -- [ ] Capture measured input, cached input, output, reasoning, and total tokens -- [ ] Intercept supported tool, retry, verification, and continuation decisions -- [ ] Add Codex session, workspace-state, and patch correlation -- [ ] Build the paired baseline-versus-MARGINAL benchmark runner -- [ ] Run a 10-task canary with identical model, prompts, tools, limits, and verifier -- [ ] Run a preregistered 100-task public benchmark -- [ ] Report token savings, quality delta, regressions, recoveries, latency, and tool calls -- [ ] Report token cost per verified successful task -- [ ] Publish raw paired JSONL results and a reproducible report +- [x] Add opt-in provider-neutral `DiminishingReturnDetector` for semantic same-state repetition. +- [x] Fail open when state is unavailable and reset pressure when state/evidence changes. +- [x] Integrate successful-execution observation without counting denied proposals as executed work. +- [x] Add `GovernanceTracker` for local decision latency and externally supplied governance tokens/USD/latency. +- [x] Add explicit reviewed false-stop accounting; never infer false stops from task outcome alone. +- [x] Extend public evaluation with repeated calls, gross versus net savings and governance overhead. +- [x] Add intervention statuses: `supported`, `pass_through`, `quality_regression`, `false_stop_risk`. +- [x] Define Graceful Irrelevance as a first-class product behavior. +- [x] Rewrite the website around a concrete trace and proof standard before theory. +- [x] Publish a Community Feedback Log with accepted, partial and rejected decisions. +- [x] Reorganize documentation by getting-started/product/integrations/evaluation/reference/operations/project responsibility. +- [ ] Validate the overlay against the canonical full repository and merge through normal CI/review. ### Exit criteria -- Codex baseline and Codex with MARGINAL can be executed under matched conditions. -- Token usage comes from Codex telemetry rather than declared estimates. -- The 10-task canary completes without integration failures. -- The 100-task report includes statistical uncertainty and a predefined quality non-inferiority margin. -- Public claims link directly to reproducible evidence. +- Existing v0.2 public benchmark rows remain parseable. +- Existing public-eval keys remain usable; new rows count governance overhead in net metrics. +- State-aware repetition control is opt-in until engine evidence supports enforcement. +- A no-benefit configuration can report `pass_through` without being mislabeled as a successful optimization. +- Website examples contain no fabricated performance numbers. +- Community feedback documentation distinguishes evidence-backed design requirements from speculation. --- -## v0.4 — Multi-Engine Developer Preview +## v0.3 — Codex Reference Integration **Status:** Planned -**Objective:** Prove that one MARGINAL runtime can govern materially different AI development agents without duplicating policy logic. +**Objective:** integrate MARGINAL into Codex and produce the first real matched benchmark with measured telemetry and net-value accounting. -### Deliverables +### Integration deliverables -- [ ] Build an OpenCode adapter -- [ ] Build a Claude Code adapter -- [ ] Build a GitHub Copilot CLI or coding-agent adapter where official control surfaces permit it -- [ ] Reuse the same protocol, policy profiles, telemetry, ledger, and reports across all adapters -- [ ] Publish an engine capability matrix -- [ ] Add adapter-specific compatibility and end-to-end tests -- [ ] Clearly label each integration as Observe, Tool Enforcement, or Full Compute Enforcement -- [ ] Add unified cross-engine session reporting -- [ ] Validate clean failure and fail-open behavior for every adapter +- [ ] Build a thin Codex adapter against the Universal Agent Protocol. +- [ ] Target `marginal install codex` with safe backup, Shadow Mode default and clean uninstall. +- [ ] Detect Codex version/capability level and refuse unsupported enforcement claims. +- [ ] Capture measured input, cached input, output, reasoning and total tokens. +- [ ] Correlate model/tool/retry/verification actions with session, task and workspace state. +- [ ] Record evidence hashes where deterministic evidence boundaries exist. +- [ ] Capture governance tokens, USD and latency separately from workload usage. +- [ ] Define and record repeated-call metrics consistently in OFF and ON arms. +- [ ] Export raw paired JSONL sufficient to reproduce the public report. -### Exit criteria +### Canary: engineering validation only -- At least four development-agent environments, including Codex, pass protocol conformance tests. -- No adapter contains duplicated economic decision logic. -- Every supported engine has documented capabilities and limitations. -- The same policy profile produces comparable decision records across engines. -- At least two engines support real action enforcement. +- [ ] Run a 10-task matched canary with identical model, prompt, tools, limits and verifier. +- [ ] Confirm event/session/state correlation and no orphaned reservations. +- [ ] Confirm telemetry is measured rather than declared. +- [ ] Confirm governance overhead is separately accounted. +- [ ] Review deny recommendations for false-stop candidates. +- [ ] Preserve pass-through and negative results instead of filtering them out. ---- +**The 10-task canary is not public performance evidence.** Its purpose is to prove the measurement/integration pipeline is trustworthy enough for a larger run. -## v0.5 — One-Command Universal Installation +### Public benchmark -**Status:** Planned +Before execution, preregister: -**Objective:** Make MARGINAL installable and removable by non-expert users without manual configuration edits. +- [ ] agent/model/version and environment; +- [ ] benchmark dataset/version and exclusions; +- [ ] matched task count and repeat count; +- [ ] quality non-inferiority margin; +- [ ] maximum acceptable false-stop rate; +- [ ] minimum net token-savings threshold; +- [ ] verifier and failure policy; +- [ ] bootstrap/repeated-run statistical method. -### Deliverables +Run at least: -- [ ] Add `marginal install --detect` -- [ ] Automatically detect supported agents and their versions -- [ ] Install only the required adapters -- [ ] Create safe backups before changing agent configuration -- [ ] Enable Quality First and Shadow Mode by default -- [ ] Add `marginal status` -- [ ] Add `marginal doctor` -- [ ] Add `marginal profile` -- [ ] Add `marginal uninstall` -- [ ] Restore original configurations during rollback -- [ ] Keep telemetry local and prompt logging disabled by default -- [ ] Test installation on Windows, macOS, and Linux -- [ ] Provide clear handling for unsupported or partially supported versions +- [ ] the community-requested SWE-bench Pro surface, with version/exclusion/quality notes; +- [ ] a targeted MARGINAL repetition suite that exposes same-state retry/verification behavior. -### Exit criteria +Report: -- A new user can install MARGINAL with one command and no manual file edits. -- Supported-agent detection and diagnostics complete in under two minutes on a typical development machine. -- Uninstall restores the original agent configuration. -- A failed installer does not leave a supported agent unusable. -- The user sees one consistent status and configuration experience across engines. +- [ ] verified resolve-rate delta; +- [ ] gross agent tokens and net effective tokens; +- [ ] effective tokens per verified successful task; +- [ ] governance tokens/USD/latency; +- [ ] tool calls and repeated calls; +- [ ] regressions and recoveries; +- [ ] reviewed false stops and false-stop rate; +- [ ] uncertainty across matched/repeated runs; +- [ ] final intervention status. ---- +### v0.3 exit criteria -## v0.6 — Adaptive and Causal Allocation +- Codex baseline and Codex + MARGINAL run under matched conditions. +- Telemetry comes from the runtime/provider integration rather than declared demo estimates. +- The canary completes without integration failures. +- Public results are reproducible from raw paired artifacts. +- Headline claims use **net** metrics after governance tax. +- If the preregistered gate is not met, the published conclusion says so. -**Status:** Planned +See [Codex benchmark readiness](docs/integrations/codex-benchmark-readiness.md). -**Objective:** Learn calibrated action value from real trajectories while distinguishing prediction, association, and causal evidence. +--- -### Deliverables +## v0.4 — Multi-Engine Developer Preview -- [ ] Train and validate contextual estimators on real engine trajectories -- [ ] Add a calibrated task belief state updated by deterministic evidence -- [ ] Estimate context-carry cost across future model turns -- [ ] Add dynamic token shadow pricing based on scarcity and projected remaining work -- [ ] Add controlled, budgeted exploration with propensity logging -- [ ] Prevent exploration for unsafe or irreversible actions -- [ ] Add off-policy evaluation appropriate to logged propensities -- [ ] Add estimator calibration, drift, and regret reports -- [ ] Compare adaptive policies against fixed reference policies on held-out runs -- [ ] Define explicit evidence standards before making causal marginal-value claims -- [ ] Preserve deterministic policy modes for reproducibility and regulated use cases +**Status:** Planned -### Exit criteria +Build OpenCode, Claude Code and GitHub Copilot integrations where official control surfaces permit them. Reuse the same protocol, policy, governance accounting, privacy boundaries and reports. Publish a capability matrix and label each engine as Observe, Tool Enforcement or Full Compute Enforcement. -- Predicted action value and observed outcomes have published calibration evidence. -- Adaptive allocation improves token cost per verified task over the fixed reference policy on held-out runs. -- Quality remains within the predefined non-inferiority margin. -- Exploration behavior is bounded, reproducible, and separately accounted. -- Any causal claim includes a documented identification strategy and assumptions. +**Exit criteria:** at least four environments pass protocol conformance; economic logic remains centralized; at least two integrations support real enforcement; each engine documents limitations and fail-open behavior. --- -## v0.7 — Ecosystem and Operational Scale +## v0.5 — One-Command Universal Installation **Status:** Planned -**Objective:** Expand compatibility and operational robustness after the universal runtime has been validated. +Add `marginal install --detect`, safe backups, supported-agent detection, Shadow Mode defaults, `status`, `doctor`, profile management and clean uninstall/rollback on Windows, macOS and Linux. -### Deliverables +**Exit criteria:** non-expert installation requires no project-code edits; failed installation leaves agents usable; uninstall restores prior configuration; users receive one consistent diagnostic experience. -- [ ] Evaluate Gemini CLI, Aider, Cline, Roo Code, Continue, and other compatible runtimes -- [ ] Add persistent local treasury and session recovery -- [ ] Add optional SQLite, Redis, or PostgreSQL backends without burdening the core -- [ ] Add reservation leases, expiry, replay, snapshots, and crash recovery -- [ ] Add OpenTelemetry spans and metrics -- [ ] Add team policy configuration and policy version pinning -- [ ] Add signed adapter and policy manifests -- [ ] Add optional shared dashboards without requiring MARGINAL Cloud -- [ ] Add portfolio allocation for parallel and multi-agent workloads -- [ ] Add action dependency, conflict, alternative, and prerequisite modeling -- [ ] Publish long-running and multi-agent reliability benchmarks +--- -### Exit criteria +## v0.6 — Adaptive and Causal Allocation + +**Status:** Planned -- Persistent sessions recover without losing committed usage. -- Distributed reservations prevent double-spend under supported backends. -- Additional adapters pass the same protocol conformance suite. -- Team features remain optional and the local open-source runtime remains fully usable without an account. +Train contextual estimators on real trajectories; add calibrated belief state, context-carry economics, dynamic shadow pricing, bounded exploration with propensity logging, off-policy evaluation, calibration/drift/regret reporting and explicit identification strategies before causal claims. + +**Exit criteria:** held-out evidence shows better effective compute per verified task than fixed policy while preserving quality; exploration remains bounded; any causal claim documents assumptions and identification strategy. --- -## Initial engine scope +## v0.7 — Ecosystem and Operational Scale + +**Status:** Planned -The first product scope is AI development agents with observable or controllable agent loops. +Evaluate additional engines; add persistent sessions, optional storage backends, reservation leases/recovery, OpenTelemetry, team policy pinning, signed manifests, optional shared dashboards and portfolio allocation for multi-agent workloads. -| Engine | Planned role | -|---|---| -| **Codex** | Reference integration and primary measured benchmark | -| **OpenCode** | Open-source research and adapter-development environment | -| **Claude Code** | Rich hook-based commercial integration | -| **GitHub Copilot CLI / coding agent** | Broad developer adoption where official APIs permit enforcement | -| **Gemini CLI, Aider, Cline, Roo Code, Continue** | Later compatibility candidates | +**Exit criteria:** persistent sessions recover without lost committed usage; distributed reservations prevent supported double-spend; optional team/cloud features do not make the local open-source runtime dependent on an account. -Autocomplete-only surfaces and traditional IDE chat are not considered equivalent to a fully controllable agent runtime. They will not be labeled as full MARGINAL integrations unless an official control surface supports real interception and measured usage. +--- ## Success metrics -MARGINAL will be evaluated on the combined outcome, not token savings alone: +MARGINAL is evaluated on the combined outcome, not token savings alone: - verified task resolution rate; -- input, cached input, output, reasoning, and total tokens; -- token cost per verified successful task; +- input, cached input, output, reasoning and total workload tokens; +- governance tokens, USD and latency; +- effective tokens and USD per verified successful task; +- gross versus net savings; - regressions and recoveries; -- tool and sub-agent calls; -- latency and direct cost; +- tool calls and repeated calls; +- reviewed stops, false stops and false-stop rate; - estimator calibration and decision regret; - variance across repeated runs; - installation success and rollback reliability; @@ -306,19 +249,16 @@ MARGINAL will be evaluated on the combined outcome, not token savings alone: The primary optimization target is: -> **Minimize token consumption per verified successful task, subject to a predefined quality non-inferiority constraint.** +> **Minimize effective compute per verified successful task, subject to predefined quality and false-stop constraints.** -MARGINAL does not promise that every individual request will use fewer tokens. Some tasks may require additional verification. The product goal is to remove avoidable compute across real sessions while protecting outcome quality. +MARGINAL does not promise that every request will use fewer tokens. Some tasks should spend more on verification. Some efficient model/runtime combinations should result in pass-through. ## Maintaining this roadmap -- Update milestone status only when its definition changes. -- Check off a deliverable only after implementation and validation are merged. -- Link relevant issues, pull requests, releases, benchmarks, or evidence where useful. -- Use GitHub Issues and Projects for task ownership and day-to-day execution. -- Keep implementation details out of this file unless they change product scope. -- Keep the README limited to the active milestone and a link to this roadmap. -- Update `CHANGELOG.md` when behavior is released, not when work is merely planned. -- Mark a milestone **Complete** only when every exit criterion has been satisfied. - -Roadmap changes are welcome through focused issues and pull requests. Proposed changes should explain the user outcome, compatibility implications, validation method, and relationship to the North Star. +- Check off deliverables only after implementation and validation are merged. +- Link issues, pull requests, releases and benchmark evidence where useful. +- Keep README claims aligned with the active released milestone. +- Update `CHANGELOG.md` when behavior is released, not merely planned. +- Mark a milestone Complete only after every exit criterion is satisfied. +- Treat negative benchmark results as valid project evidence. +- Require proposed performance changes to state how they could be falsified. diff --git a/SECURITY.md b/SECURITY.md index da9fa94..e685254 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -47,7 +47,7 @@ operational key with an exported dataset. The default ignore rules exclude `*.pr Applications can still write sensitive data to `JsonlTraceSink`, `LOCAL_FULL` ledgers, custom files, logs outside MARGINAL, or downstream systems. `SAFE_TELEMETRY` is a strict allowlist at the Decision Ledger boundary; it does not sanitize arbitrary external logs. Review -[`docs/privacy.md`](docs/privacy.md) before sharing evidence. +[`docs/operations/privacy.md`](docs/operations/privacy.md) before sharing evidence. Protocol metadata used for automatic fingerprints must be deterministically JSON serializable. This rejects ambiguous custom-object representations, but it does not make the source metadata diff --git a/demos/killer-demo/index.html b/demos/killer-demo/index.html index cc19d2c..08e6b15 100644 --- a/demos/killer-demo/index.html +++ b/demos/killer-demo/index.html @@ -1311,7 +1311,7 @@ Results Trace Benchmark diff --git a/demos/killer-demo/trace.jsonl b/demos/killer-demo/trace.jsonl index febbf20..3e38e55 100644 --- a/demos/killer-demo/trace.jsonl +++ b/demos/killer-demo/trace.jsonl @@ -1,9 +1,9 @@ {"candidates": [{"action": {"cost": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "current_success_probability": 0.0, "expected_gain": 0.22, "fingerprint": "9249f7c0ea26013b8bf6b1aea2ae903c1a389d4891d3d1b3e27587cc90153028", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "inspect the failing assertion"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.030699999999999998, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.22, "mode": "enforce", "reason": "approved: marginal ROI 7.166", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.166", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.1893, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 2200, "risk": 0.0, "tokens": 9000, "usd": 0.045}, "current_success_probability": 0.0, "expected_gain": 0.05, "fingerprint": "5c1e3d25d6c0ea662b38fc76c3adb01e2916aab6808cbff750b38ae368d9d712", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "scan the entire repository"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.22939999999999997, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.05, "mode": "enforce", "reason": "rejected: marginal ROI 0.218 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.218 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.17939999999999995, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 4500, "risk": 0.0, "tokens": 14000, "usd": 0.12}, "current_success_probability": 0.0, "expected_gain": 0.04, "fingerprint": "cca32271d74f3dab1bc141d5918b12eb406090a194bdf9b436fe76c036836eb2", "is_verification": false, "kind": "review", "metadata": {"stage": "diagnose"}, "name": "ask two parallel reviewers"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.40900000000000003, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.04, "mode": "enforce", "reason": "rejected: marginal ROI 0.098 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.098 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.36900000000000005, "uncertainty": 0.0}}], "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "candidate_ranking", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo"} -{"action": {"cost": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "current_success_probability": 0.0, "expected_gain": 0.22, "fingerprint": "9249f7c0ea26013b8bf6b1aea2ae903c1a389d4891d3d1b3e27587cc90153028", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "inspect the failing assertion"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.030699999999999998, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.22, "mode": "enforce", "reason": "approved: marginal ROI 7.166", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.166", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.1893, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "treasury": "killer-demo", "usage": {"latency_ms": 0, "risk": 0.0, "tokens": 0, "usd": 0.0}} -{"action": {"cost": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "current_success_probability": 0.0, "expected_gain": 0.22, "fingerprint": "9249f7c0ea26013b8bf6b1aea2ae903c1a389d4891d3d1b3e27587cc90153028", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "inspect the failing assertion"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "violations": []} +{"action": {"cost": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "current_success_probability": 0.0, "expected_gain": 0.22, "fingerprint": "9249f7c0ea26013b8bf6b1aea2ae903c1a389d4891d3d1b3e27587cc90153028", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "inspect the failing assertion"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.030699999999999998, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.22, "mode": "enforce", "reason": "approved: marginal ROI 7.166", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.166", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.1893, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "governance": {"decision_latency_ms": 0.0, "decisions": 4, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "treasury": "killer-demo", "usage": {"latency_ms": 0, "risk": 0.0, "tokens": 0, "usd": 0.0}} +{"action": {"cost": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "current_success_probability": 0.0, "expected_gain": 0.22, "fingerprint": "9249f7c0ea26013b8bf6b1aea2ae903c1a389d4891d3d1b3e27587cc90153028", "is_verification": false, "kind": "research", "metadata": {"stage": "diagnose"}, "name": "inspect the failing assertion"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "governance": {"decision_latency_ms": 0.0, "decisions": 4, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}, "violations": []} {"candidates": [{"action": {"cost": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "current_success_probability": 0.0, "expected_gain": 0.5, "fingerprint": "6709d8e40d23cdc7b42020f54ed11dfa3e9141f5117e1f8d85642c12708e0a8a", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "apply the targeted one-line patch"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.06739999999999999, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.5, "mode": "enforce", "reason": "approved: marginal ROI 7.418", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.418", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.4326, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 3200, "risk": 0.0, "tokens": 12000, "usd": 0.11}, "current_success_probability": 0.0, "expected_gain": 0.2, "fingerprint": "d96d4754a8ac012a625b0ccc6e2944f64dcac291577089a2ab33e44c59156adc", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "rewrite the complete pricing module"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.3564, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.2, "mode": "enforce", "reason": "rejected: marginal ROI 0.561 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.561 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.15639999999999998, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 5500, "risk": 0.0, "tokens": 18000, "usd": 0.28}, "current_success_probability": 0.0, "expected_gain": 0.15, "fingerprint": "1125e93f1af77ac4124f84f538886e9f97b17dd935ad5266f09d064da312d841", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "ask a frontier model for an alternative patch"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.651, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.15, "mode": "enforce", "reason": "rejected: marginal ROI 0.230 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.230 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.501, "uncertainty": 0.0}}], "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "candidate_ranking", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo"} -{"action": {"cost": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "current_success_probability": 0.0, "expected_gain": 0.5, "fingerprint": "6709d8e40d23cdc7b42020f54ed11dfa3e9141f5117e1f8d85642c12708e0a8a", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "apply the targeted one-line patch"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.06739999999999999, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.5, "mode": "enforce", "reason": "approved: marginal ROI 7.418", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.418", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.4326, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "treasury": "killer-demo", "usage": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}} -{"action": {"cost": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "current_success_probability": 0.0, "expected_gain": 0.5, "fingerprint": "6709d8e40d23cdc7b42020f54ed11dfa3e9141f5117e1f8d85642c12708e0a8a", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "apply the targeted one-line patch"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 1050, "risk": 0.0, "tokens": 3600, "usd": 0.024}, "violations": []} +{"action": {"cost": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "current_success_probability": 0.0, "expected_gain": 0.5, "fingerprint": "6709d8e40d23cdc7b42020f54ed11dfa3e9141f5117e1f8d85642c12708e0a8a", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "apply the targeted one-line patch"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.06739999999999999, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.5, "mode": "enforce", "reason": "approved: marginal ROI 7.418", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 7.418", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.4326, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "governance": {"decision_latency_ms": 0.0, "decisions": 8, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "treasury": "killer-demo", "usage": {"latency_ms": 350, "risk": 0.0, "tokens": 1200, "usd": 0.006}} +{"action": {"cost": {"latency_ms": 700, "risk": 0.0, "tokens": 2400, "usd": 0.018}, "current_success_probability": 0.0, "expected_gain": 0.5, "fingerprint": "6709d8e40d23cdc7b42020f54ed11dfa3e9141f5117e1f8d85642c12708e0a8a", "is_verification": false, "kind": "generation", "metadata": {"stage": "fix"}, "name": "apply the targeted one-line patch"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "governance": {"decision_latency_ms": 0.0, "decisions": 8, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 1050, "risk": 0.0, "tokens": 3600, "usd": 0.024}, "violations": []} {"candidates": [{"action": {"cost": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "current_success_probability": 0.0, "expected_gain": 0.35, "fingerprint": "313d64e7457c58d692b5f2ec902a279c32ca88807380c35dc464afaed301d70f", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the targeted verifier"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.01636, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.35, "mode": "enforce", "reason": "approved: marginal ROI 21.394", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 21.394", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.33364, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 1400, "risk": 0.0, "tokens": 4500, "usd": 0.012}, "current_success_probability": 0.0, "expected_gain": 0.08, "fingerprint": "f83033a110b1105e98b597c52e17027bceab726f06b6f3438ff26d61deca992d", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the full test suite"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.10479999999999999, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.08, "mode": "enforce", "reason": "rejected: marginal ROI 0.763 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.763 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.02479999999999999, "uncertainty": 0.0}}, {"action": {"cost": {"latency_ms": 4000, "risk": 0.0, "tokens": 11000, "usd": 0.17}, "current_success_probability": 0.0, "expected_gain": 0.05, "fingerprint": "c444a2361858cb3bcc7b1a4e2596a0ef1248b3aa5978f63a07b40ad62cf10db2", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "request a premium model audit"}, "decision": {"allowed": false, "confidence": 1.0, "estimated_cost_value": 0.398, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.05, "mode": "enforce", "reason": "rejected: marginal ROI 0.126 below 1.000", "reason_code": "MARGINAL_ROI_REJECTED", "recommendation_reason": "rejected: marginal ROI 0.126 below 1.000", "recommendation_reason_code": "MARGINAL_ROI_REJECTED", "recommended": false, "score": -0.34800000000000003, "uncertainty": 0.0}}], "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "candidate_ranking", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo"} -{"action": {"cost": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "current_success_probability": 0.0, "expected_gain": 0.35, "fingerprint": "313d64e7457c58d692b5f2ec902a279c32ca88807380c35dc464afaed301d70f", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the targeted verifier"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.01636, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.35, "mode": "enforce", "reason": "approved: marginal ROI 21.394", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 21.394", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.33364, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "treasury": "killer-demo", "usage": {"latency_ms": 1050, "risk": 0.0, "tokens": 3600, "usd": 0.024}} -{"action": {"cost": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "current_success_probability": 0.0, "expected_gain": 0.35, "fingerprint": "313d64e7457c58d692b5f2ec902a279c32ca88807380c35dc464afaed301d70f", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the targeted verifier"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 1230, "risk": 0.0, "tokens": 4300, "usd": 0.026000000000000002}, "violations": []} +{"action": {"cost": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "current_success_probability": 0.0, "expected_gain": 0.35, "fingerprint": "313d64e7457c58d692b5f2ec902a279c32ca88807380c35dc464afaed301d70f", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the targeted verifier"}, "decision": {"allowed": true, "confidence": 1.0, "estimated_cost_value": 0.01636, "estimator_name": "historical-mean", "estimator_version": "2.0.0", "expected_gain": 0.35, "mode": "enforce", "reason": "approved: marginal ROI 21.394", "reason_code": "APPROVED", "recommendation_reason": "approved: marginal ROI 21.394", "recommendation_reason_code": "APPROVED", "recommended": true, "score": 0.33364, "uncertainty": 0.0}, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "authorization", "governance": {"decision_latency_ms": 0.0, "decisions": 12, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "reserved": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "treasury": "killer-demo", "usage": {"latency_ms": 1050, "risk": 0.0, "tokens": 3600, "usd": 0.024}} +{"action": {"cost": {"latency_ms": 180, "risk": 0.0, "tokens": 700, "usd": 0.002}, "current_success_probability": 0.0, "expected_gain": 0.35, "fingerprint": "313d64e7457c58d692b5f2ec902a279c32ca88807380c35dc464afaed301d70f", "is_verification": true, "kind": "verification", "metadata": {"stage": "verify"}, "name": "run the targeted verifier"}, "budget_overrun": false, "estimator": {"config_hash": "8bd2358da0bef14a0514ddbdc9505cac5a645a024d8a722af2f1793a2ea8ab92", "name": "historical-mean", "training_data_fingerprint": null, "version": "2.0.0"}, "event": "commit", "governance": {"decision_latency_ms": 0.0, "decisions": 12, "external_latency_ms": 0.0, "external_tokens": 0, "external_usd": 0.0, "false_stop_rate": null, "false_stops": 0, "reviewed_stops": 0, "scope": "treasury_tree", "total_latency_ms": 0.0}, "mode": "enforce", "policy": {"config_hash": "06ff44c755d4d09cdad5efbf5feb55af755e760db18699d07077db84ae4bb8d0", "name": "marginal-reference", "version": "2.0.0"}, "treasury": "killer-demo", "usage": {"latency_ms": 1230, "risk": 0.0, "tokens": 4300, "usd": 0.026000000000000002}, "violations": []} diff --git a/docs/benchmarking.md b/docs/evaluation/benchmarking.md similarity index 100% rename from docs/benchmarking.md rename to docs/evaluation/benchmarking.md diff --git a/docs/evaluation/governance-evidence.md b/docs/evaluation/governance-evidence.md new file mode 100644 index 0000000..6b0e738 --- /dev/null +++ b/docs/evaluation/governance-evidence.md @@ -0,0 +1,123 @@ +# Governance Evidence Standard + +MARGINAL is an intervention in an agent runtime. Its evaluation must therefore include the cost and mistakes introduced by the intervention itself. + +## Core equation + +The useful quantity is not raw token reduction. A benchmark should reason about net intervention value: + +```text +net intervention value + = workload benefit + - governance overhead + - quality loss + - harmful false stops + - added latency / direct cost +``` + +The implementation does not collapse those terms into one magic score. It reports them separately so a reviewer can inspect the tradeoff. + +## Governance tax + +`GovernanceTracker` separates local decision overhead from external overhead introduced by an adapter or auxiliary model call. + +- local decision latency is measured by `Treasury` around policy recommendations; +- adapter-side governance tokens, USD and latency are recorded explicitly with `record_governance_overhead(...)`; +- the benchmark evaluator adds workload and governance cost into effective tokens, USD and latency; +- gross savings remain visible, but net savings are the primary claim surface. + +A governor that reduces workload tokens while consuming more total effective tokens should not be described as an optimization. + +## Graceful Irrelevance + +`compare_runs(...)` can classify an otherwise quality-preserving result as `pass_through` when net token savings do not exceed a preregistered threshold. + +This behavior is intentional. A future model, runtime or task may already be efficient enough that MARGINAL adds no material value. The correct behavior is to get out of the way, not to force intervention. + +## Diminishing-return evidence + +`DiminishingReturnDetector` is provider neutral and opt-in. It uses fields already available through normalized agent actions: + +- semantic action identity; +- workspace/state hash; +- optional evidence hash; +- execution history. + +The detector's `evaluate(...)` method does not mutate history. `observe(...)` is called only after actual successful execution. This prevents a denied proposal from being counted as if it consumed compute. + +The default logic fails open when state is unavailable. A repeat is discounted only when the same semantic action is observed in the same state without new evidence. A changed state or evidence hash resets the pressure. + +## False Stop Rate + +A token optimizer can look excellent by simply preventing an agent from working. False-stop accounting exists to make that failure visible. + +A false stop means: + +> MARGINAL recommended denying an action, and an explicit external review concluded that the action would have helped. + +The implementation deliberately does **not** infer false stops from task-level success or failure. `Treasury.record_stop_review(...)` accepts an explicit boolean label and refuses to label actions that were not previously recommended for denial. + +Public reports expose: + +```text +reviewed_stops +false_stops +false_stop_rate +``` + +The acceptable false-stop threshold should be preregistered for a public evaluation. A strict zero threshold is a reasonable starting point for early enforcement experiments; Shadow Mode should be used to gather evidence before blocking behavior. + +## Matched OFF/ON protocol + +A defensible runtime comparison holds constant: + +- agent and agent version; +- model and model configuration; +- prompt / task input; +- tools and permissions; +- token and time limits; +- task ordering; +- repository/environment state; +- verifier and success criterion. + +MARGINAL is the intervention. If other variables change, the result cannot cleanly attribute the difference to MARGINAL. + +## Reported metrics + +The minimum public report should include: + +| Metric | Requirement | +|---|---| +| Verified resolution rate | Baseline and MARGINAL | +| Quality delta | Percentage points and preregistered non-inferiority margin | +| Agent tokens | Gross workload usage | +| Governance tokens | Added by MARGINAL / adapter | +| Effective tokens | Agent + governance | +| Effective tokens per resolved task | Primary efficiency metric | +| USD and latency | Gross and effective where measurable | +| Tool calls | Baseline and MARGINAL | +| Repeated calls | Same definition across both arms | +| Regressions / recoveries | Instance-level counts | +| Reviewed / false stops | Explicit counterfactual labels | +| Statistical uncertainty | Repeated runs and/or bootstrap interval | +| Intervention status | Supported, pass-through, quality regression or false-stop risk | + +## Canary versus evidence + +The v0.3 10-task Codex canary exists to answer engineering questions: does installation work, are events correlated correctly, are tokens measured correctly, and can runs finish without adapter failures? + +It does not answer the product-performance question. A public result requires a larger preregistered sample and enough repeated execution to characterize variance. + +## Benchmark selection + +SWE-bench Pro is a useful community-requested evaluation surface because it gives an externally recognizable coding workload. It should not be the only surface. MARGINAL also needs targeted workloads that make its claimed mechanism observable, including repeated same-state verification and retry patterns. + +For every benchmark, record the exact dataset version, verifier, exclusions and known task-quality limitations. A recognizable benchmark name is not a substitute for inspecting the evaluation instrument. + +## Claim discipline + +Allowed language after a measured run should resemble: + +> Under the preregistered configuration, MARGINAL reduced effective tokens per verified successful task by X%, with a Y pp resolution-rate delta and Z reviewed false stops. + +Avoid universal statements such as “MARGINAL cuts agent tokens by X%.” Results belong to a model, agent, benchmark version, policy configuration and date. diff --git a/docs/public-benchmarks.md b/docs/evaluation/public-benchmarks.md similarity index 100% rename from docs/public-benchmarks.md rename to docs/evaluation/public-benchmarks.md diff --git a/docs/research.md b/docs/evaluation/research.md similarity index 100% rename from docs/research.md rename to docs/evaluation/research.md diff --git a/docs/quickstart.md b/docs/getting-started/quickstart.md similarity index 89% rename from docs/quickstart.md rename to docs/getting-started/quickstart.md index 8cfb809..d95aa7e 100644 --- a/docs/quickstart.md +++ b/docs/getting-started/quickstart.md @@ -64,7 +64,7 @@ marginal ledger-export ledger.jsonl aggregate.jsonl --privacy-profile aggregate_ `safe_telemetry` excludes free text and pseudonymizes identifiers. Use `local_full` only for a trusted operational ledger. Use `aggregate_export` when preparing grouped data for sharing; groups smaller than five records are suppressed by default. -Pseudonymization is not anonymization; read [`privacy.md`](privacy.md) before export. +Pseudonymization is not anonymization; read [`privacy.md`](../operations/privacy.md) before export. Move to `recommend` when recommendations are surfaced to a user or agent. Move to `enforce` only after representative validation shows acceptable quality. @@ -72,4 +72,4 @@ Move to `recommend` when recommendations are surfaced to a user or agent. Move t Use `UniversalRuntime` when integrating a development agent. Enforce Mode requires an adapter that declares real action-blocking capability. The reference runtime currently maps core decisions to allow or deny; other protocol directives are extension points. -See [`universal-runtime.md`](universal-runtime.md) and the executable examples in [`examples`](../examples). +See [`universal-runtime.md`](../product/architecture.md) and the executable examples in [`examples`](../../examples). diff --git a/docs/governance.md b/docs/governance.md index c1de5f9..9c605f6 100644 --- a/docs/governance.md +++ b/docs/governance.md @@ -1,30 +1,4 @@ -# Governance + -MARGINAL begins as a SignalLayer Labs-led open-source project. - -## Decision process - -- routine fixes and documentation changes use normal pull-request review; -- public API changes require rationale, compatibility notes, and tests; -- policy, ledger, protocol, schema, or privacy-profile changes require a design discussion before implementation; -- benchmark claims require reproducible evidence and independent review when practical; -- security-sensitive fixes may be developed privately before coordinated disclosure; -- shareable telemetry changes require an explicit field-classification and quasi-identifier review. - -## Compatibility - -Semantic Versioning applies to the Python public API. Trace records include explicit event -names and are designed for additive evolution. Breaking trace or API changes require a -major release after `1.0.0`. - -## Maintainer responsibilities - -Maintainers protect technical integrity, transparent claims, contributor safety, and a -small dependency-free core. Project influence follows sustained, reviewed contribution -rather than employer or commercial status. - -## Privacy governance - -The operational Decision Ledger and shareable telemetry are separate products with separate contracts. `LOCAL_FULL` may retain caller-controlled local evidence; `SAFE_TELEMETRY` is a strict allowlist with keyed pseudonyms; `AGGREGATE_EXPORT` contains generalized grouped rows only. Unknown fields are treated as potentially sensitive. - -A change may not weaken a privacy profile silently. Any newly retained field requires tests, documentation, schema updates where applicable, and a migration or compatibility note. Pseudonymized data must never be described as anonymous. +This document has moved to [project/governance.md](project/governance.md). +It reflects the SignalLayerLabs governance and review expectations for public repository operations. diff --git a/docs/index.md b/docs/index.md index 599b316..e0100c0 100644 --- a/docs/index.md +++ b/docs/index.md @@ -1,21 +1,45 @@ -# MARGINAL documentation - -MARGINAL is a provider-neutral decision, accounting, and evidence layer for economically disciplined AI agents. Version `0.2.0` adds the Learning Loop Foundation: non-blocking Shadow Mode, a versioned Decision Ledger, explicit privacy profiles, strict shareable-telemetry schemas, outcome contracts, versioned estimators, replay, packaged protocol schemas, and a universal adapter runtime. - -Start with the [quickstart](quickstart.md), then read the [concepts](concepts.md), [learning loop](learning-loop.md), and [architecture](architecture.md). - -## Guides - -- [Quickstart](quickstart.md) -- [Concepts](concepts.md) -- [Learning Loop Foundation](learning-loop.md) -- [Universal runtime](universal-runtime.md) -- [Privacy profiles](privacy.md) -- [Architecture](architecture.md) -- [API reference](api.md) -- [Integrations](integrations.md) -- [Benchmarking](benchmarking.md) -- [Public benchmark protocol](public-benchmarks.md) -- [Research and prior art](research.md) -- [FAQ](faq.md) -- [Governance](governance.md) +# MARGINAL Documentation + +MARGINAL documentation is organized by user intent instead of keeping every guide in one flat directory. + +## Getting started + +- [Quickstart](getting-started/quickstart.md) + +## Product + +- [Concepts](product/concepts.md) +- [Architecture](product/architecture.md) +- [FAQ](product/faq.md) + +## Integrations + +- [Integration overview](integrations/overview.md) +- [Codex benchmark readiness](integrations/codex-benchmark-readiness.md) + +## Evaluation and research + +- [Benchmarking](evaluation/benchmarking.md) +- [Public benchmark protocol](evaluation/public-benchmarks.md) +- [Governance evidence standard](evaluation/governance-evidence.md) +- [Research and prior art](evaluation/research.md) + +## Reference + +- [API reference](reference/api.md) + +## Operations + +- [Privacy](operations/privacy.md) +- [Website operations](operations/website.md) +- [Website review — 2026-08-07](operations/website-review-2026-08-07.md) + +## Project + +- [Governance](project/governance.md) +- [Community feedback](project/community-feedback.md) +- [Roadmap](../ROADMAP.md) +- [Contributing](../CONTRIBUTING.md) +- [Security](../SECURITY.md) + +`docs/superpowers/` contains implementation specs and plans for maintainers and agentic development workflows; it is intentionally separate from end-user documentation. diff --git a/docs/integrations/codex-benchmark-readiness.md b/docs/integrations/codex-benchmark-readiness.md new file mode 100644 index 0000000..3a3476f --- /dev/null +++ b/docs/integrations/codex-benchmark-readiness.md @@ -0,0 +1,86 @@ +# Codex Benchmark Readiness + +This document prepares v0.3 without presenting a Codex adapter as already implemented. + +## Target user experience + +The milestone target remains a one-command installation path: + +```bash +marginal install codex +``` + +The command is a **v0.3 target**, not part of v0.2.0. + +## Adapter responsibilities + +The Codex adapter should translate native Codex events into the existing Universal Agent Protocol rather than reimplement economic policy. It needs to provide, where the official integration surface permits: + +- stable session and task correlation; +- normalized tool/model/retry/verification actions; +- workspace state hashes; +- evidence hashes when a deterministic evidence boundary exists; +- measured input, cached input, output, reasoning and total tokens; +- actual USD when available or a clearly labeled derived estimate; +- action latency and tool-call counts; +- repeated-call classification under a documented definition; +- outcome/verifier correlation; +- explicit capability negotiation for observe versus block/stop behavior. + +## Safe rollout + +The recommended sequence is: + +1. install in Shadow Mode; +2. validate event completeness and token accounting; +3. run the 10-task canary as integration validation; +4. inspect deny recommendations and manually review false-stop candidates; +5. freeze the adapter/policy configuration; +6. preregister the public benchmark protocol; +7. run matched OFF/ON evaluation; +8. consider enforcement only if the evidence gate is satisfied. + +## One-command installer requirements + +Before the public benchmark, `marginal install codex` should be able to: + +- detect a supported Codex installation/version; +- explain the detected capability level; +- back up any configuration it changes; +- install the thin adapter without source-code edits to user projects; +- default to Shadow Mode; +- expose `marginal status` / diagnostics for the integration; +- uninstall cleanly and restore prior configuration; +- fail without leaving Codex unusable. + +The exact mechanism must follow the official Codex integration surface available at implementation time. Do not rely on undocumented hooks solely to satisfy the one-command goal. + +## Canary exit criteria + +The 10-task canary passes only when: + +- every task has a matched baseline and MARGINAL run; +- token telemetry is measured rather than declared; +- action/session/state correlation is complete enough to explain repeated-work decisions; +- governance overhead is captured separately; +- no adapter crash or orphaned reservation occurs; +- raw paired JSONL can reproduce the report; +- negative or pass-through results are preserved rather than filtered out. + +Passing the canary means the measurement system is ready for larger evaluation. It does not mean MARGINAL has demonstrated a savings claim. + +## Public benchmark gate + +The public protocol should define before execution: + +- benchmark/version and any exclusions; +- model and agent version; +- run limits and environment; +- number of matched tasks and repeat count; +- quality non-inferiority margin; +- maximum false-stop rate; +- minimum net token-savings threshold; +- verifier and failure handling; +- statistical reporting method. + +SWE-bench Pro can be included because the community explicitly requested it. A targeted MARGINAL repetition suite should run alongside it so the claimed mechanism can be inspected directly. diff --git a/docs/integrations.md b/docs/integrations/overview.md similarity index 100% rename from docs/integrations.md rename to docs/integrations/overview.md diff --git a/docs/operations/privacy.md b/docs/operations/privacy.md new file mode 100644 index 0000000..53665cf --- /dev/null +++ b/docs/operations/privacy.md @@ -0,0 +1,191 @@ +# Privacy profiles + +MARGINAL is local-first and has no mandatory network service, but locality alone does not +make telemetry safe to share. Identifiers, action names, model names, repository labels, +error text, verifier details, and caller metadata can reveal sensitive information even when +prompts and model outputs are absent. + +MARGINAL therefore classifies evidence fields and provides three explicit privacy profiles. + +## Field classes + +### Safe by default + +These fields are structured and retained by `safe_telemetry`: + +- generic event and action kind; +- estimated and actual cost; +- token breakdown and latency; +- applied and recommended decisions; +- stable reason codes; +- structured task reward and resolved status; +- policy and estimator versions; +- confidence, uncertainty, score, and schema version. + +Arbitrary strings are normalized or replaced with a generic value. Numeric sub-objects use +an allowlist, so custom metric names are not copied accidentally. + +### Pseudonymous + +These fields are transformed with field-separated HMAC-SHA-256 under a local key: + +- event, run, task, trajectory, and action identifiers; +- action fingerprints and state hashes; +- other explicit engine-instance identifiers when adapters expose them. + +Exact timestamps are generalized to UTC day boundaries. Pseudonyms are stable only for the +same key and field name. Different keys produce unlinkable identifiers. + +When external correlation is unnecessary, prefer opaque random IDs from +`generate_local_identifier("run")`, `generate_local_identifier("task")`, or another simple +namespace. Random local IDs avoid embedding customer or project names before sanitization. + +### Potentially sensitive + +The strict profile excludes: + +- free-form action names; +- complete model identity; +- metadata, tags, tool arguments, and replacement payloads; +- error, exception, abort, and failure text; +- human-readable policy reasons; +- verifier identity, evidence, and custom outcome metrics; +- treasury names and estimator training-data fingerprints. + +## Profiles + +### `LOCAL_FULL` + +`local_full` is the backward-compatible default. It preserves the complete operational +Decision Ledger record. Use it only where the ledger path and filesystem access are trusted. +Caller-provided metadata remains caller responsibility. + +```python +ledger = JsonlDecisionLedger( + "ledger.jsonl", + context=DecisionLedgerContext(run_id="local-run"), + privacy_profile="local_full", +) +``` + +### `SAFE_TELEMETRY` + +`safe_telemetry` removes free text and metadata, pseudonymizes identifiers, generalizes exact +timestamps, and keeps only allowlisted structured fields. Every strict record is validated when read through `read_decision_ledger(...)` and can also be +checked explicitly with `validate_safe_telemetry_record(...)`. The packaged +`safe-telemetry-v1.json` schema rejects unknown fields recursively. + +```python +ledger = JsonlDecisionLedger( + "safe-ledger.jsonl", + context=DecisionLedgerContext( + run_id="customer-acme-contract-2026", + task_id="customer-acme-contract-2026", + engine="codex", + model="internal-legal-model", + ), + privacy_profile="safe_telemetry", + privacy_key_path=".marginal/privacy.key", +) +``` + +New ledger and export files are created with owner-only permissions on POSIX systems. Existing +ledger append targets must be regular files, must not be symbolic links, and must not be accessible +by group or other users. When no key or key path is supplied, the ledger creates an owner-only +hidden key beside the ledger. +Generated keys contain 256 random bits and are never written into ledger records. +Keep the key outside version control and backups intended for sharing. + +An existing key file must be a regular file and, on POSIX systems, must not be readable by +group or other users. Symbolic-link key paths are rejected. + +### `AGGREGATE_EXPORT` + +`aggregate_export` is deliberately separate from operational ledger persistence. It groups +generalized decision and outcome rows, removes all identifiers and timestamps, and suppresses +any group containing fewer than five source records by default. The threshold is configurable +and is recorded in every emitted row. + +```bash +marginal ledger-export ledger.jsonl aggregate.jsonl \ + --privacy-profile aggregate_export --minimum-group-size 5 +``` + +A grouped decision row contains only fields such as: + +```json +{ + "schema_version": "1.0", + "privacy_profile": "aggregate_export", + "record_type": "decision", + "action_kind": "verification", + "cost_bucket": "low", + "gain_bucket": "medium", + "recommendation": "deny", + "applied_decision": "allow", + "reason_code": "SHADOW_OVERRIDE", + "outcome_class": "not_applicable", + "count": 12, + "minimum_group_size": 5 +} +``` + +Small groups are omitted entirely. Raising `--minimum-group-size` reduces disclosure risk but +can remove more data. Lowering it below five is intended only for controlled local analysis and +should not be treated as anonymous sharing. Default buckets are deterministic: + +- cost: `low` up to 2,000 tokens, USD 0.02, and 1 second; `medium` up to 10,000 +tokens, USD 0.20, and 10 seconds; otherwise `high`; +- expected gain: `low` below 0.10, `medium` below 0.30, otherwise `high`. + +## Exporting an existing ledger + +Create a new unlinkable safe export by using a dedicated export key: + +```bash +marginal ledger-export ledger.jsonl safe-export.jsonl \ + --privacy-profile safe_telemetry \ + --privacy-key-file .marginal/export.key +``` + +The API equivalent is: + +```python +from marginal import export_decision_ledger + +export_decision_ledger( + "ledger.jsonl", + "safe-export.jsonl", + privacy_profile="safe_telemetry", + privacy_key_path=".marginal/export.key", +) +``` + +Exports create the destination with an exclusive filesystem operation and never overwrite an +existing path, including when another process creates the destination after the initial check. This +prevents accidental replacement of an authoritative ledger or a previously reviewed dataset. + +## Threat model and limitations + +Pseudonymization is not anonymization. Stable pseudonyms can still be linkable within one +export, rare action patterns can identify a workload, and small aggregate groups may permit +inference. The profiles do not provide differential privacy, k-anonymity, encryption at rest, +cryptographic tamper evidence, multi-process locking, or compliance certification. + +Before sharing data: + +1. prefer `aggregate_export` over event-level telemetry; +2. inspect the generated file; +3. use a new export key rather than an operational key; +4. raise the default minimum group size when the dataset or population is small; +5. avoid combining exports with external datasets that restore identity; +6. treat source ledgers and pseudonymization keys as sensitive assets. + +The public classification map is available as `FIELD_CLASSIFICATION`. `classify_field(...)` +inherits reviewed classifications for nested fields such as `action.cost.tokens`, while unknown +paths default to potentially sensitive. Applications may use the map for UI explanations or +additional validation, but custom event fields are excluded by the strict profile unless MARGINAL +explicitly allowlists them. + +Use `load_schema("safe-telemetry-v1.json")` and +`load_schema("aggregate-export-v1.json")` to validate shareable outputs from an installed wheel. diff --git a/docs/operations/website-review-2026-08-07.md b/docs/operations/website-review-2026-08-07.md new file mode 100644 index 0000000..9f59ed4 --- /dev/null +++ b/docs/operations/website-review-2026-08-07.md @@ -0,0 +1,62 @@ +# Website Review — 2026-08-07 + +## Review objective + +Evaluate whether the MARGINAL website helps a skeptical developer understand the product, inspect its evidence standard and decide whether the project deserves a technical trial. + +## Finding 1: the previous hero was technically correct but too abstract + +The previous site opened with compute governance vocabulary and an architecture-like decision flow. That was accurate, but it forced a first-time visitor to understand the theory before seeing a concrete reason to care. + +**Change:** the new hero starts with an illustrative repeated-verification trace. It is explicitly labeled “not a benchmark,” avoiding the temptation to use invented numbers as proof. + +## Finding 2: the value proposition did not price MARGINAL itself + +The old page explained token/cost governance but did not foreground the cost introduced by the governor. That creates an obvious credibility problem for a product whose purpose is economic discipline. + +**Change:** “MARGINAL must earn its own compute” is now a primary product principle. The proof section distinguishes gross savings from net savings after governance tax. + +## Finding 3: model-progress risk was treated as an objection instead of a requirement + +A visitor could reasonably ask why MARGINAL survives a future model that no longer exhibits today's inefficient loops. + +**Change:** the new site presents Graceful Irrelevance. `pass_through` is a legitimate result when the governor has no demonstrated net benefit. + +## Finding 4: the evidence section was too far from the value claim + +The previous site did contain scientific caveats, but users had to scroll through product concepts before reaching them. + +**Change:** the proof standard is now the second major section. It names matched OFF/ON runs, effective tokens per resolved task, false stops, repeated calls and uncertainty. + +## Finding 5: community criticism was invisible after it was processed + +The project asked for open-source participation but gave no structured indication of how criticism affects decisions. + +**Change:** the landing page now exposes a compact pressure-test section and links to the Community Feedback Log. Accepted and rejected feedback are both visible. + +## Finding 6: some previous copy was too broad + +Statements such as “agent runtimes often execute work because a model requested it” are directionally plausible but easy to read as a universal indictment. + +**Change:** copy now uses narrower claims: coding agents *can* spend compute on actions whose incremental value is unclear or diminishing. The website does not speculate about provider motives. + +## Information architecture after review + +The intended narrative is now: + +```text +observable problem +→ concrete trace +→ proof standard +→ governance tax / graceful irrelevance +→ generic mechanism +→ community decisions +→ benchmark discipline +→ roadmap +``` + +Architecture, privacy and learning-loop detail remain in GitHub documentation rather than competing with first-screen comprehension. + +## Accessibility and operational constraints + +The site remains dependency-free, responsive and tracking-free. It preserves one `h1`, semantic sectioning, keyboard navigation, reduced-motion handling and the existing GitHub Pages deployment model. diff --git a/docs/website.md b/docs/operations/website.md similarity index 100% rename from docs/website.md rename to docs/operations/website.md diff --git a/docs/privacy.md b/docs/privacy.md index 53665cf..a0996f0 100644 --- a/docs/privacy.md +++ b/docs/privacy.md @@ -1,191 +1,4 @@ -# Privacy profiles + -MARGINAL is local-first and has no mandatory network service, but locality alone does not -make telemetry safe to share. Identifiers, action names, model names, repository labels, -error text, verifier details, and caller metadata can reveal sensitive information even when -prompts and model outputs are absent. - -MARGINAL therefore classifies evidence fields and provides three explicit privacy profiles. - -## Field classes - -### Safe by default - -These fields are structured and retained by `safe_telemetry`: - -- generic event and action kind; -- estimated and actual cost; -- token breakdown and latency; -- applied and recommended decisions; -- stable reason codes; -- structured task reward and resolved status; -- policy and estimator versions; -- confidence, uncertainty, score, and schema version. - -Arbitrary strings are normalized or replaced with a generic value. Numeric sub-objects use -an allowlist, so custom metric names are not copied accidentally. - -### Pseudonymous - -These fields are transformed with field-separated HMAC-SHA-256 under a local key: - -- event, run, task, trajectory, and action identifiers; -- action fingerprints and state hashes; -- other explicit engine-instance identifiers when adapters expose them. - -Exact timestamps are generalized to UTC day boundaries. Pseudonyms are stable only for the -same key and field name. Different keys produce unlinkable identifiers. - -When external correlation is unnecessary, prefer opaque random IDs from -`generate_local_identifier("run")`, `generate_local_identifier("task")`, or another simple -namespace. Random local IDs avoid embedding customer or project names before sanitization. - -### Potentially sensitive - -The strict profile excludes: - -- free-form action names; -- complete model identity; -- metadata, tags, tool arguments, and replacement payloads; -- error, exception, abort, and failure text; -- human-readable policy reasons; -- verifier identity, evidence, and custom outcome metrics; -- treasury names and estimator training-data fingerprints. - -## Profiles - -### `LOCAL_FULL` - -`local_full` is the backward-compatible default. It preserves the complete operational -Decision Ledger record. Use it only where the ledger path and filesystem access are trusted. -Caller-provided metadata remains caller responsibility. - -```python -ledger = JsonlDecisionLedger( - "ledger.jsonl", - context=DecisionLedgerContext(run_id="local-run"), - privacy_profile="local_full", -) -``` - -### `SAFE_TELEMETRY` - -`safe_telemetry` removes free text and metadata, pseudonymizes identifiers, generalizes exact -timestamps, and keeps only allowlisted structured fields. Every strict record is validated when read through `read_decision_ledger(...)` and can also be -checked explicitly with `validate_safe_telemetry_record(...)`. The packaged -`safe-telemetry-v1.json` schema rejects unknown fields recursively. - -```python -ledger = JsonlDecisionLedger( - "safe-ledger.jsonl", - context=DecisionLedgerContext( - run_id="customer-acme-contract-2026", - task_id="customer-acme-contract-2026", - engine="codex", - model="internal-legal-model", - ), - privacy_profile="safe_telemetry", - privacy_key_path=".marginal/privacy.key", -) -``` - -New ledger and export files are created with owner-only permissions on POSIX systems. Existing -ledger append targets must be regular files, must not be symbolic links, and must not be accessible -by group or other users. When no key or key path is supplied, the ledger creates an owner-only -hidden key beside the ledger. -Generated keys contain 256 random bits and are never written into ledger records. -Keep the key outside version control and backups intended for sharing. - -An existing key file must be a regular file and, on POSIX systems, must not be readable by -group or other users. Symbolic-link key paths are rejected. - -### `AGGREGATE_EXPORT` - -`aggregate_export` is deliberately separate from operational ledger persistence. It groups -generalized decision and outcome rows, removes all identifiers and timestamps, and suppresses -any group containing fewer than five source records by default. The threshold is configurable -and is recorded in every emitted row. - -```bash -marginal ledger-export ledger.jsonl aggregate.jsonl \ - --privacy-profile aggregate_export --minimum-group-size 5 -``` - -A grouped decision row contains only fields such as: - -```json -{ - "schema_version": "1.0", - "privacy_profile": "aggregate_export", - "record_type": "decision", - "action_kind": "verification", - "cost_bucket": "low", - "gain_bucket": "medium", - "recommendation": "deny", - "applied_decision": "allow", - "reason_code": "SHADOW_OVERRIDE", - "outcome_class": "not_applicable", - "count": 12, - "minimum_group_size": 5 -} -``` - -Small groups are omitted entirely. Raising `--minimum-group-size` reduces disclosure risk but -can remove more data. Lowering it below five is intended only for controlled local analysis and -should not be treated as anonymous sharing. Default buckets are deterministic: - -- cost: `low` up to 2,000 tokens, USD 0.02, and 1 second; `medium` up to 10,000 -tokens, USD 0.20, and 10 seconds; otherwise `high`; -- expected gain: `low` below 0.10, `medium` below 0.30, otherwise `high`. - -## Exporting an existing ledger - -Create a new unlinkable safe export by using a dedicated export key: - -```bash -marginal ledger-export ledger.jsonl safe-export.jsonl \ - --privacy-profile safe_telemetry \ - --privacy-key-file .marginal/export.key -``` - -The API equivalent is: - -```python -from marginal import export_decision_ledger - -export_decision_ledger( - "ledger.jsonl", - "safe-export.jsonl", - privacy_profile="safe_telemetry", - privacy_key_path=".marginal/export.key", -) -``` - -Exports create the destination with an exclusive filesystem operation and never overwrite an -existing path, including when another process creates the destination after the initial check. This -prevents accidental replacement of an authoritative ledger or a previously reviewed dataset. - -## Threat model and limitations - -Pseudonymization is not anonymization. Stable pseudonyms can still be linkable within one -export, rare action patterns can identify a workload, and small aggregate groups may permit -inference. The profiles do not provide differential privacy, k-anonymity, encryption at rest, -cryptographic tamper evidence, multi-process locking, or compliance certification. - -Before sharing data: - -1. prefer `aggregate_export` over event-level telemetry; -2. inspect the generated file; -3. use a new export key rather than an operational key; -4. raise the default minimum group size when the dataset or population is small; -5. avoid combining exports with external datasets that restore identity; -6. treat source ledgers and pseudonymization keys as sensitive assets. - -The public classification map is available as `FIELD_CLASSIFICATION`. `classify_field(...)` -inherits reviewed classifications for nested fields such as `action.cost.tokens`, while unknown -paths default to potentially sensitive. Applications may use the map for UI explanations or -additional validation, but custom event fields are excluded by the strict profile unless MARGINAL -explicitly allowlists them. - -Use `load_schema("safe-telemetry-v1.json")` and -`load_schema("aggregate-export-v1.json")` to validate shareable outputs from an installed wheel. +This document has moved to [operations/privacy.md](operations/privacy.md). +It preserves the SAFE_TELEMETRY and AGGREGATE_EXPORT guidance for the privacy workflow. Pseudonymization is not anonymization. diff --git a/docs/architecture.md b/docs/product/architecture.md similarity index 100% rename from docs/architecture.md rename to docs/product/architecture.md diff --git a/docs/concepts.md b/docs/product/concepts.md similarity index 100% rename from docs/concepts.md rename to docs/product/concepts.md diff --git a/docs/faq.md b/docs/product/faq.md similarity index 100% rename from docs/faq.md rename to docs/product/faq.md diff --git a/docs/project/community-feedback.md b/docs/project/community-feedback.md new file mode 100644 index 0000000..4cc139b --- /dev/null +++ b/docs/project/community-feedback.md @@ -0,0 +1,86 @@ +# Community Feedback Log + +MARGINAL is developed in public, but community feedback is treated as evidence to examine rather than instructions to implement automatically. + +## Decision rule + +A community criticism enters the product roadmap only when it identifies a reproducible user problem, a falsifiable product risk, a missing measurement, or a clearer way to explain an implemented capability. Unsupported claims about motives, universal model behavior, or performance are not promoted into product requirements. + +## Review: model progress could make MARGINAL redundant + +**Feedback:** a future model release may stop pathological verification loops, making MARGINAL unnecessary. + +**Decision:** partially accepted; promoted into a product principle. + +The criticism correctly attacks a weak version of the thesis. If MARGINAL were only a workaround for one model's repetitive behavior, it would have a short useful life. The durable problem is broader: users need an independent way to measure whether another unit of agent compute is expected to improve the verified outcome enough to justify its cost. + +The criticism does **not** establish that model progress makes independent compute governance redundant. More capable models can still have different cost, latency, verification, tool-use and risk profiles. The value of the governor must therefore be measured per workload rather than assumed. + +**Product consequence:** add Graceful Irrelevance. A configuration that does not demonstrate positive net intervention value should report `pass_through` instead of manufacturing an optimization claim. + +## Review: endless verification loops + +**Feedback:** repeated verification of an unchanged `.md` file is a current pain point. + +**Decision:** accepted as a failure-mode example, rejected as a vendor/file-specific implementation target. + +The useful abstraction is: + +> same semantic action + unchanged observable state + no new evidence → diminishing expected marginal value. + +**Product consequence:** add an opt-in state-aware `DiminishingReturnDetector`. It discounts repeated same-state work and can recommend a stop after a configured threshold. Missing state fails open; changed state or new evidence resets the repetition pressure. + +## Review: providers may have incentives not to reduce waste + +**Feedback:** longer runs generate more usage, so a provider may have little incentive to eliminate waste. + +**Decision:** rejected as a product claim. + +Usage-based pricing can create economic differences between provider and user objectives, but that does not establish intentional waste or deliberate preservation of loops. MARGINAL does not need to speculate about provider motives. The legitimate requirement is that users can observe and control their own compute economics independently of the provider. + +**Communication consequence:** do not use claims such as “providers want agents to waste tokens.” + +## Review: “less slop” in the website/post + +**Feedback:** the website and launch copy feel too abstract. + +**Decision:** partially accepted. + +The technical concepts are not removed: transactional accounting, learning loop, privacy and the Universal Agent Protocol are implemented and remain important. The presentation order was the problem. The previous landing page asked a new visitor to understand the theory before seeing a concrete failure mode or the evidence needed to validate the product. + +**Product communication consequence:** reorder the website to: + +1. concrete trace; +2. proof standard; +3. net-value / governance-tax principle; +4. mechanism; +5. architecture and roadmap. + +The new site labels its example as illustrative rather than benchmark evidence. + +## Review: show SWE-bench Pro with MARGINAL OFF vs ON + +**Feedback:** demonstrate value using the same coding benchmark with and without MARGINAL. + +**Decision:** accepted in principle, with a methodological qualification. + +Matched OFF/ON evaluation is exactly the right causal comparison for a runtime intervention. The same agent, model, prompt, tools, limits, task order and verifier should be held fixed. However, no single benchmark is treated as ground truth. Dataset version, task-quality exclusions, verifier behavior and known limitations must be recorded. + +**Evidence consequence:** the benchmark report adds: + +- effective tokens per verified successful task; +- gross versus net savings; +- governance tokens, USD and latency; +- repeated calls; +- regressions and recoveries; +- reviewed false stops; +- statistical uncertainty; +- an intervention status that can explicitly be `pass_through`. + +The 10-task Codex canary remains an integration check. It must not be promoted into a general performance claim. + +## What community influence means for MARGINAL + +Community feedback is most valuable when it produces a harder falsification test. A good issue or comment does not need to agree with the project thesis. It should make the thesis more precise, measurable or easier to disprove. + +The maintainers should publish negative benchmark results when a configuration does not meet the preregistered gate. That is part of the project's credibility, not a reason to hide the run. diff --git a/docs/project/governance.md b/docs/project/governance.md new file mode 100644 index 0000000..c1de5f9 --- /dev/null +++ b/docs/project/governance.md @@ -0,0 +1,30 @@ +# Governance + +MARGINAL begins as a SignalLayer Labs-led open-source project. + +## Decision process + +- routine fixes and documentation changes use normal pull-request review; +- public API changes require rationale, compatibility notes, and tests; +- policy, ledger, protocol, schema, or privacy-profile changes require a design discussion before implementation; +- benchmark claims require reproducible evidence and independent review when practical; +- security-sensitive fixes may be developed privately before coordinated disclosure; +- shareable telemetry changes require an explicit field-classification and quasi-identifier review. + +## Compatibility + +Semantic Versioning applies to the Python public API. Trace records include explicit event +names and are designed for additive evolution. Breaking trace or API changes require a +major release after `1.0.0`. + +## Maintainer responsibilities + +Maintainers protect technical integrity, transparent claims, contributor safety, and a +small dependency-free core. Project influence follows sustained, reviewed contribution +rather than employer or commercial status. + +## Privacy governance + +The operational Decision Ledger and shareable telemetry are separate products with separate contracts. `LOCAL_FULL` may retain caller-controlled local evidence; `SAFE_TELEMETRY` is a strict allowlist with keyed pseudonyms; `AGGREGATE_EXPORT` contains generalized grouped rows only. Unknown fields are treated as potentially sensitive. + +A change may not weaken a privacy profile silently. Any newly retained field requires tests, documentation, schema updates where applicable, and a migration or compatibility note. Pseudonymized data must never be described as anonymous. diff --git a/docs/api.md b/docs/reference/api.md similarity index 98% rename from docs/api.md rename to docs/reference/api.md index 1ea027d..531b32f 100644 --- a/docs/api.md +++ b/docs/reference/api.md @@ -126,7 +126,7 @@ Public privacy values and functions: `JsonlDecisionLedger` accepts `privacy_profile`, `privacy_key`, and `privacy_key_path`. `aggregate_export` is rejected as an operational profile and must use the export API. Export -destinations are not overwritten. See [`privacy.md`](privacy.md) for field behavior and threat +destinations are not overwritten. See [`privacy.md`](../operations/privacy.md) for field behavior and threat model. ## Universal protocol @@ -178,7 +178,7 @@ for name in available_schemas(): schema = load_schema(name) ``` -The public schema API reads immutable JSON resources bundled in the installed wheel. Names are restricted to known basenames; path traversal and unknown resources are rejected. The same source contracts remain available under [`schemas/`](../schemas/). +The public schema API reads immutable JSON resources bundled in the installed wheel. Names are restricted to known basenames; path traversal and unknown resources are rejected. The same source contracts remain available under [`schemas/`](../../schemas/). Privacy-specific contracts include `safe-telemetry-v1.json`, which recursively rejects unreviewed fields from strict event-level exports, and `aggregate-export-v1.json`, which accepts only grouped generalized rows. diff --git a/docs/superpowers/plans/2026-08-06-learning-loop-foundation.md b/docs/superpowers/plans/2026-08-06-learning-loop-foundation.md index 3265abe..b83dc1b 100644 --- a/docs/superpowers/plans/2026-08-06-learning-loop-foundation.md +++ b/docs/superpowers/plans/2026-08-06-learning-loop-foundation.md @@ -148,13 +148,13 @@ Tasks 1–9 have been implemented and covered by the repository test suite. Task - Modify: `CHANGELOG.md` - Modify: `CONTRIBUTING.md` - Modify: `SECURITY.md` -- Modify: `docs/api.md` -- Modify: `docs/architecture.md` -- Modify: `docs/concepts.md` -- Modify: `docs/integrations.md` -- Modify: `docs/benchmarking.md` -- Modify: `docs/quickstart.md` -- Modify: `docs/faq.md` +- Modify: `docs/reference/api.md` +- Modify: `docs/product/architecture.md` +- Modify: `docs/product/concepts.md` +- Modify: `docs/integrations/overview.md` +- Modify: `docs/evaluation/benchmarking.md` +- Modify: `docs/getting-started/quickstart.md` +- Modify: `docs/product/faq.md` - Create: `docs/learning-loop.md` - Create: `docs/universal-runtime.md` - Create: `ROADMAP.md` diff --git a/docs/superpowers/plans/2026-08-06-privacy-profiles.md b/docs/superpowers/plans/2026-08-06-privacy-profiles.md index b9eaf44..cfa96e2 100644 --- a/docs/superpowers/plans/2026-08-06-privacy-profiles.md +++ b/docs/superpowers/plans/2026-08-06-privacy-profiles.md @@ -93,8 +93,8 @@ Tasks 1–5 have been implemented with test-first coverage. Task 6 is complete f **Files:** - Modify: `README.md`, `SECURITY.md`, `CHANGELOG.md`, `ROADMAP.md` -- Modify: `docs/api.md`, `docs/concepts.md`, `docs/learning-loop.md`, `docs/universal-runtime.md`, `docs/architecture.md`, `docs/quickstart.md`, `docs/faq.md`, `docs/index.md` -- Create: `docs/privacy.md` +- Modify: `docs/reference/api.md`, `docs/product/concepts.md`, `docs/learning-loop.md`, `docs/universal-runtime.md`, `docs/product/architecture.md`, `docs/getting-started/quickstart.md`, `docs/product/faq.md`, `docs/index.md` +- Create: `docs/operations/privacy.md` - Create: `examples/privacy_profiles.py` - Modify: repository consistency tests. diff --git a/docs/superpowers/plans/2026-08-06-readme-pages.md b/docs/superpowers/plans/2026-08-06-readme-pages.md index 300ba09..ef40f54 100644 --- a/docs/superpowers/plans/2026-08-06-readme-pages.md +++ b/docs/superpowers/plans/2026-08-06-readme-pages.md @@ -13,5 +13,5 @@ 1. Replace `README.md`, preserving install, quickstart, evidence, privacy, roadmap, docs, contribution, citation, and license. 2. Add `site/index.html`, styles, navigation script, robots, sitemap, and 404 page. 3. Add one `.github/workflows/pages.yml` deployment that also publishes the Killer Demo. -4. Add `docs/website.md` and update `CHANGELOG.md`. +4. Add `docs/operations/website.md` and update `CHANGELOG.md`. 5. Run `scripts/validate_readme_pages.py`, repository CI checks, local preview, and final diff review. diff --git a/docs/superpowers/plans/2026-08-07-community-evidence-hardening.md b/docs/superpowers/plans/2026-08-07-community-evidence-hardening.md new file mode 100644 index 0000000..17b632d --- /dev/null +++ b/docs/superpowers/plans/2026-08-07-community-evidence-hardening.md @@ -0,0 +1,131 @@ +# Community Evidence Hardening Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add model-independent repetition control, MARGINAL self-accounting, net-value benchmark evidence, community decision transparency, an evidence-first website and a clean documentation structure without claiming a completed Codex integration. + +**Architecture:** New optional controls live in `src/marginal/controls`; existing policy and treasury APIs receive backward-compatible additive hooks. Public evaluation keeps old input rows valid while treating governance overhead as part of effective cost. Documentation is reorganized with history-preserving moves and the website remains static/dependency-free. + +**Tech Stack:** Python 3.10–3.13 standard library, pytest, Ruff, mypy, static HTML/CSS/JS, GitHub Actions. + +## Global Constraints + +- Provider neutral; no GPT-5.6 or Markdown-specific runtime branch. +- Zero mandatory runtime dependencies. +- No fabricated benchmark numbers. +- False stops require explicit review labels. +- Diminishing-return enforcement remains opt-in until representative evidence exists. +- Codex adapter remains a v0.3 target, not a completed capability. + +--- + +### Task 1: State-aware diminishing-return control + +**Files:** +- Create: `src/marginal/controls/diminishing.py` +- Create: `src/marginal/controls/__init__.py` +- Modify: `src/marginal/policy.py` +- Modify: `src/marginal/treasury.py` +- Test: `tests/controls/test_diminishing.py` +- Test: `tests/controls/test_policy_diminishing.py` + +**Interfaces:** +- Produces: `DiminishingReturnConfig`, `DiminishingReturnSignal`, `DiminishingReturnDetector`. +- `MarginalPolicy(..., diminishing_detector=None)` remains backward compatible. +- `MarginalPolicy.observe_execution(action)` records successful execution. + +- [ ] Write tests proving first execution has multiplier 1, same-state repeats decay, changed state/evidence resets, and missing state fails open. +- [ ] Run the focused tests and verify failure because the control does not exist. +- [ ] Implement the detector with pure `evaluate()` and explicit `observe()`. +- [ ] Integrate optional gain discount/deny behavior into `MarginalPolicy`. +- [ ] Make successful Treasury settlement call `observe_execution` while preserving legacy `mark_executed` fallback. +- [ ] Run focused tests until green. + +### Task 2: Governance tax and false-stop evidence + +**Files:** +- Create: `src/marginal/controls/governance.py` +- Modify: `src/marginal/treasury.py` +- Test: `tests/controls/test_governance.py` +- Test: `tests/controls/test_treasury_governance.py` + +**Interfaces:** +- Produces: `GovernanceTracker.record_decision`, `record_external_overhead`, `record_stop_review`, `summary`. +- Treasury produces: `record_governance_overhead(...)`, `record_stop_review(...)`. + +- [ ] Write tests for overhead separation and explicit false-stop rate. +- [ ] Verify tests fail before implementation. +- [ ] Measure policy recommendation wall latency separately from action usage. +- [ ] Add explicit adapter-side overhead accounting. +- [ ] Track prior deny recommendations and reject duplicate/unrelated counterfactual labels. +- [ ] Add governance evidence to Treasury summary/trace without changing enforcement semantics. +- [ ] Run focused tests until green. + +### Task 3: Net-value public evaluation + +**Files:** +- Modify: `src/marginal/public_eval.py` +- Test: `tests/evaluation/test_public_eval_governance.py` + +**Interfaces:** +- `RunRecord` adds optional zero-default fields for repeated calls, governance overhead and false-stop evidence. +- `compare_runs` adds `gross_savings`, `net_savings`, `governance`, `intervention` while retaining `savings`, `quality`, `efficiency`. + +- [ ] Write tests for 30% gross / 10% net, governance-driven pass-through, and false-stop risk. +- [ ] Verify tests fail before implementation. +- [ ] Add effective cost aggregation and bootstrap net-token interval. +- [ ] Keep legacy rows and strict type validation working. +- [ ] Render governance tax and intervention status in Markdown reports. +- [ ] Run old and new public-eval tests. + +### Task 4: Evidence-first communication + +**Files:** +- Modify: `README.md` +- Modify: `site/index.html` +- Modify: `site/styles.css` +- Create: `docs/operations/website-review-2026-08-07.md` +- Create: `docs/project/community-feedback.md` + +- [ ] Replace abstract-first website narrative with a clearly labeled illustrative trace. +- [ ] Put matched evidence, governance tax and false-stop requirements before architecture theory. +- [ ] Add Graceful Irrelevance and explicit pass-through language. +- [ ] Publish accepted/partial/rejected community decisions. +- [ ] Remove provider-motive speculation and universal model-waste claims. +- [ ] Run website/README structural validation. + +### Task 5: Documentation information architecture + +**Files:** +- Create: `MIGRATION_MANIFEST.json` +- Create: `scripts/reorganize_docs.py` +- Modify: `docs/index.md` +- Create: `docs/evaluation/governance-evidence.md` +- Create: `docs/integrations/codex-benchmark-readiness.md` + +- [ ] Move flat docs into responsibility-based directories with `git mv`. +- [ ] Rewrite local Markdown links relative to their post-move locations. +- [ ] Update exact old-path references repository-wide. +- [ ] Validate no legacy docs paths remain and all local Markdown links resolve. + +### Task 6: Roadmap, changelog and full verification + +**Files:** +- Modify: `ROADMAP.md` +- Modify: `CHANGELOG.md` +- Modify: `scripts/validate_readme_pages.py` +- Create: `scripts/validate_community_hardening.py` + +- [ ] Add self-accounting, Graceful Irrelevance and false-stop principles to roadmap. +- [ ] Explicitly separate 10-task Codex canary from public performance evidence. +- [ ] Add governance/repetition/false-stop metrics to v0.3 benchmark deliverables. +- [ ] Add SWE-bench Pro as one evaluation surface, not ground truth. +- [ ] Record the hardening changes under `Unreleased` without changing package version. +- [ ] Run `python scripts/validate_community_hardening.py`. +- [ ] Run `python scripts/validate_readme_pages.py`. +- [ ] Run `ruff format --check .`. +- [ ] Run `ruff check .`. +- [ ] Run `mypy src/marginal`. +- [ ] Run `pytest -q`. +- [ ] Run `python -m build` and `python -m twine check dist/*`. +- [ ] Inspect `git diff` and confirm no ZIP, local prompt, fabricated benchmark artifact or duplicate Pages deploy is staged. diff --git a/docs/superpowers/specs/2026-08-07-community-evidence-hardening-design.md b/docs/superpowers/specs/2026-08-07-community-evidence-hardening-design.md new file mode 100644 index 0000000..6e848f7 --- /dev/null +++ b/docs/superpowers/specs/2026-08-07-community-evidence-hardening-design.md @@ -0,0 +1,101 @@ +# MARGINAL Community Evidence Hardening — Design + +**Date:** 2026-08-07 + +## Objective + +Convert the first substantive community criticisms into stronger, falsifiable product behavior without turning MARGINAL into a patch for one model release or accepting unsupported narratives. + +## Design decisions + +### 1. Model independence + +Do not implement a GPT-5.6 or Markdown-specific loop detector. The reusable signal is semantic repetition under unchanged state without new evidence. + +### 2. Conservative diminishing-return control + +Add an opt-in `DiminishingReturnDetector` under `src/marginal/controls/`. It evaluates without mutation and observes history only after successful execution. Missing state fails open. Changed state/evidence resets repetition pressure. + +The detector may discount expected gain before the normal ROI decision and may eventually recommend denial. It is not enabled by default in the reference policy until representative engine telemetry validates thresholds. + +### 3. MARGINAL self-accounting + +Add `GovernanceTracker` to count: + +- policy decision latency; +- externally introduced governance tokens/USD/latency; +- reviewed deny recommendations; +- explicit false stops. + +Parent/child treasuries share a tracker so a hierarchy reports one governance-tax surface. + +### 4. False stops are labels, not inferred causality + +A task result cannot establish that an individual blocked action would or would not have helped. `Treasury.record_stop_review(...)` therefore requires an explicit external boolean label and only accepts actions previously recommended for denial. + +### 5. Net-value public evaluation + +Extend `RunRecord` additively. Existing v0.2 JSONL rows remain valid. Public evaluation reports workload-only gross savings and effective net savings after governance overhead. + +The legacy `savings` key remains available and points to the net result. For old rows with zero governance overhead the numerical behavior is unchanged. + +### 6. Graceful Irrelevance + +The evaluator can return: + +- `supported`; +- `pass_through`; +- `quality_regression`; +- `false_stop_risk`. + +`pass_through` is a valid result when quality is preserved but net token savings do not clear the preregistered threshold. + +### 7. Evidence-first website + +The product website should no longer lead with architecture abstractions. Narrative order: + +1. illustrative failure trace, explicitly not a benchmark; +2. proof standard; +3. governance tax and Graceful Irrelevance; +4. generic diminishing-return mechanism; +5. community decision log; +6. benchmark discipline; +7. roadmap. + +No fabricated token-savings number is used. + +### 8. Community governance + +Create a durable Community Feedback Log. Feedback can be accepted, partially accepted or rejected with rationale. Unsupported claims about provider intent are rejected; valid falsification tests are promoted into roadmap/evidence requirements. + +### 9. Documentation information architecture + +Keep Python core modules stable except for a new `controls` package. Reorganize documentation by user intent: + +```text +docs/ + getting-started/ + product/ + integrations/ + evaluation/ + reference/ + operations/ + project/ + superpowers/ +``` + +Use `git mv` through a deterministic migration script and rewrite relative Markdown links. + +### 10. Codex boundary + +This hardening package prepares v0.3 but does not implement or claim a Codex adapter. The next implementation must use the official integration surface available at that time, default to Shadow Mode, capture measured telemetry, and target one-command installation. + +## Compatibility constraints + +- Python 3.10–3.13. +- Zero mandatory runtime dependencies. +- Existing v0.1/v0.2 constructor call sites remain valid. +- Existing public benchmark JSONL rows remain valid. +- Existing synthetic demo remains labeled synthetic. +- No provider-specific logic enters the decision core. +- No performance claim is added without measured data. diff --git a/scripts/reorganize_docs.py b/scripts/reorganize_docs.py new file mode 100644 index 0000000..0fad33b --- /dev/null +++ b/scripts/reorganize_docs.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Reorganize MARGINAL docs while preserving Git history and relative Markdown links.""" + +from __future__ import annotations + +import os +import re +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +MOVES = { + "docs/quickstart.md": "docs/getting-started/quickstart.md", + "docs/concepts.md": "docs/product/concepts.md", + "docs/architecture.md": "docs/product/architecture.md", + "docs/faq.md": "docs/product/faq.md", + "docs/api.md": "docs/reference/api.md", + "docs/integrations.md": "docs/integrations/overview.md", + "docs/benchmarking.md": "docs/evaluation/benchmarking.md", + "docs/public-benchmarks.md": "docs/evaluation/public-benchmarks.md", + "docs/research.md": "docs/evaluation/research.md", + "docs/privacy.md": "docs/operations/privacy.md", + "docs/website.md": "docs/operations/website.md", + "docs/governance.md": "docs/project/governance.md", +} + +LINK_RE = re.compile(r"(? str: + return path.relative_to(ROOT).as_posix() + + +def _new_path(path: Path) -> Path: + relative = _repo_relative(path) + return ROOT / MOVES.get(relative, relative) + + +def _rewrite_markdown_links(path: Path, text: str) -> str: + old_source = path + new_source = _new_path(path) + + def replace(match: re.Match[str]) -> str: + label, raw_target = match.groups() + if raw_target.startswith(("http://", "https://", "mailto:", "#")): + return match.group(0) + target_part, hash_mark, fragment = raw_target.partition("#") + if not target_part or not target_part.lower().endswith(".md"): + return match.group(0) + old_target = (old_source.parent / target_part).resolve() + try: + old_relative = old_target.relative_to(ROOT).as_posix() + except ValueError: + return match.group(0) + new_target = ROOT / MOVES.get(old_relative, old_relative) + new_relative = Path(os.path.relpath(new_target, start=new_source.parent)).as_posix() + rewritten = new_relative + (f"#{fragment}" if hash_mark else "") + return f"[{label}]({rewritten})" + + return LINK_RE.sub(replace, text) + + +def _rewrite_exact_paths(path: Path, text: str) -> str: + if path.suffix.lower() not in TEXT_SUFFIXES: + return text + for old, new in MOVES.items(): + text = text.replace(old, new) + return text + + +def _rewrite_references_before_move() -> None: + for path in ROOT.rglob("*"): + if not path.is_file() or ".git" in path.parts or path.suffix.lower() not in TEXT_SUFFIXES: + continue + if _repo_relative(path) in REWRITE_EXCLUSIONS: + continue + try: + text = path.read_text(encoding="utf-8") + except UnicodeDecodeError: + continue + updated = _rewrite_exact_paths(path, text) + if path.suffix.lower() == ".md": + updated = _rewrite_markdown_links(path, updated) + if updated != text: + path.write_text(updated, encoding="utf-8") + + +def _move(old: str, new: str) -> None: + source = ROOT / old + target = ROOT / new + if target.exists() and not source.exists(): + return + if source.exists() and target.exists(): + raise RuntimeError(f"both source and target exist: {old} -> {new}") + if not source.exists(): + raise FileNotFoundError(f"expected documentation source is missing: {old}") + target.parent.mkdir(parents=True, exist_ok=True) + if (ROOT / ".git").exists(): + subprocess.run(["git", "mv", old, new], cwd=ROOT, check=True) + else: + source.replace(target) + + +def main() -> int: + _rewrite_references_before_move() + for old, new in MOVES.items(): + _move(old, new) + print(f"Reorganized {len(MOVES)} documentation files.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/validate_community_hardening.py b/scripts/validate_community_hardening.py new file mode 100644 index 0000000..540bc0c --- /dev/null +++ b/scripts/validate_community_hardening.py @@ -0,0 +1,94 @@ +#!/usr/bin/env python3 +"""Focused structural checks for the community-evidence hardening overlay.""" + +from __future__ import annotations + +import re +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + +required_files = [ + "src/marginal/controls/__init__.py", + "src/marginal/controls/diminishing.py", + "src/marginal/controls/governance.py", + "docs/evaluation/governance-evidence.md", + "docs/project/community-feedback.md", + "docs/integrations/codex-benchmark-readiness.md", + "docs/operations/website-review-2026-08-07.md", +] +missing = [item for item in required_files if not (ROOT / item).is_file()] +assert not missing, f"missing hardening files: {missing}" + +public_eval = (ROOT / "src/marginal/public_eval.py").read_text(encoding="utf-8") +for token in [ + "governance_tokens", + "repeated_calls", + "false_stops", + "gross_savings", + "net_savings", + '"pass_through"', + '"false_stop_risk"', +]: + assert token in public_eval, token + +treasury = (ROOT / "src/marginal/treasury.py").read_text(encoding="utf-8") +for token in [ + "GovernanceTracker", + "record_governance_overhead", + "record_stop_review", + "observe_execution", +]: + assert token in treasury, token + +policy = (ROOT / "src/marginal/policy.py").read_text(encoding="utf-8") +for token in ["diminishing_detector", "diminishing.reason_code", "observe_execution"]: + assert token in policy, token + +roadmap = (ROOT / "ROADMAP.md").read_text(encoding="utf-8") +for token in [ + "Graceful irrelevance", + "governance tokens", + "false-stop", + "10-task", + "SWE-bench Pro", +]: + assert token.lower() in roadmap.lower(), token + +old_paths = [ + "docs/quickstart.md", + "docs/concepts.md", + "docs/architecture.md", + "docs/api.md", + "docs/integrations.md", + "docs/public-benchmarks.md", +] +for old in old_paths: + legacy_path = ROOT / old + if not legacy_path.exists(): + continue + text = legacy_path.read_text(encoding="utf-8") + if "compatibility shim" in text.lower() and "moved to" in text.lower(): + continue + assert not legacy_path.exists(), f"old documentation path still exists: {old}" + +# Validate local Markdown file links. External links and anchors are intentionally skipped. +link_re = re.compile(r"(? {raw}") +assert not errors, "broken Markdown links:\n" + "\n".join(errors[:30]) + +print("Community hardening structure: PASS") +print("Core evidence hooks: PASS") +print("Documentation migration: PASS") +print("Markdown links: PASS") diff --git a/scripts/validate_readme_pages.py b/scripts/validate_readme_pages.py index 88606f5..15e38df 100644 --- a/scripts/validate_readme_pages.py +++ b/scripts/validate_readme_pages.py @@ -42,17 +42,18 @@ def handle_data(self, data): readme = (ROOT / "README.md").read_text(encoding="utf-8") words = len(re.findall(r"\b[\w\'-]+\b", readme)) -assert 1200 <= words <= 2200, f"README words: {words}" +assert 1000 <= words <= 2400, f"README words: {words}" for heading in [ "# MARGINAL", - "## Why MARGINAL", - "## How it works", + "## The problem in one trace", + "## What changed after community review", + "## MARGINAL must earn its own compute", + "## State-aware diminishing returns", + "## Governance accounting and false-stop review", + "## Proof standard", "## Install", "## Quickstart", - "## The Learning Loop Foundation", - "## Universal Agent Runtime", - "## Privacy by design", - "## Evidence, not hype", + "## Architecture", "## Project status", "## Documentation", ]: @@ -61,18 +62,20 @@ def handle_data(self, data): "guarantees fewer tokens", "saves tokens on every request", "Codex adapter is available", - "Claude Code adapter is available", + "providers want agents to waste tokens", ]: assert forbidden.lower() not in readme.lower(), forbidden required = [ - "docs/quickstart.md", - "docs/architecture.md", - "docs/privacy.md", - "docs/api.md", - "docs/integrations.md", - "docs/benchmarking.md", - "docs/public-benchmarks.md", + "docs/getting-started/quickstart.md", + "docs/product/architecture.md", + "docs/operations/privacy.md", + "docs/reference/api.md", + "docs/integrations/overview.md", + "docs/evaluation/benchmarking.md", + "docs/evaluation/public-benchmarks.md", + "docs/evaluation/governance-evidence.md", + "docs/project/community-feedback.md", "ROADMAP.md", "SECURITY.md", "CONTRIBUTING.md", @@ -93,6 +96,17 @@ def handle_data(self, data): assert parser.scripts == ["app.js"] assert not any(x.startswith(("http://", "https://")) for x in parser.styles + parser.scripts) +site_text = (ROOT / "site/index.html").read_text(encoding="utf-8") +for required_phrase in [ + "Illustrative trace", + "Not a benchmark", + "MARGINAL must earn its own compute", + "Graceful irrelevance", + "Community pressure test", + "pass_through", +]: + assert required_phrase in site_text, required_phrase + for path in [ "site/styles.css", "site/app.js", @@ -105,7 +119,6 @@ def handle_data(self, data): workflow = (ROOT / ".github/workflows/pages.yml").read_text(encoding="utf-8") for token in [ "actions/configure-pages@v5", - "actions/upload-pages-artifact@v3", "actions/deploy-pages@v4", "demos/killer-demo", "assets/marginal-readme-hero.png", @@ -122,5 +135,5 @@ def handle_data(self, data): print(f"README words: {words}") print("README structure and claims: PASS") print("Referenced files: PASS") -print("Website SEO and dependency audit: PASS") +print("Website evidence-first structure: PASS") print("Pages workflow consolidation: PASS") diff --git a/site/index.html b/site/index.html index 2588ae2..39acd91 100644 --- a/site/index.html +++ b/site/index.html @@ -3,20 +3,20 @@ - MARGINAL — Compute Governance and Token Optimization for AI Agents - - + MARGINAL — Evidence-Driven Compute Governance for AI Agents + + - - + + - - + + @@ -28,11 +28,11 @@ MARGINAL @@ -40,106 +40,128 @@
-

Open-source compute governance for AI agents

-

Fund only the next action worth taking.

-

MARGINAL helps AI agents decide whether the next model call, tool call, search, retry, review or sub-agent is worth its token cost.

+

Compute governance, measured

+

The next action should add evidence, not just activity.

+

MARGINAL evaluates whether another model call, tool call, retry, verification or reviewer is likely to add enough value to justify its cost — and accounts for the cost of making that decision.

    -
  • Local first
  • Provider neutral
  • Zero mandatory dependencies
  • Apache-2.0
  • +
  • Open source
  • Local first
  • Provider neutral
  • Zero mandatory dependencies
-
-
Agent proposes action
↓
-
Estimate value versus cost
↓
-
ALLOWDENYSHADOW
-
↓
Settle actual usage + outcome
+
+
Illustrative traceNot a benchmark
+
01
Read README.mdnew information acquired
RUN
+
02
Verify README.mdnew evidence acquired
RUN
+
03
Verify README.mdsame state · no new evidence
DISCOUNT
+
04
Verify README.mddiminishing return threshold reached
STOP?
+

The question mark matters: enforcement is earned by evidence, not assumed by the demo.

-
-

The missing layer

-

Budgets say what an agent can afford. MARGINAL asks what deserves funding.

-

Agent runtimes often execute work because a model requested it or a hard limit has not been reached. MARGINAL compares expected improvement with tokens, cost, latency, risk, remaining budget and verification needs before compute is committed.

+
+
+

The actual thesis

+

Not “models waste tokens.” Diminishing marginal value.

+
+
+

Coding agents can spend compute on actions whose incremental value becomes unclear. A repeated action is not automatically waste: another test may be exactly what a risky patch needs.

+

MARGINAL looks for a stronger pattern: same semantic action + unchanged observable state + no new evidence. That is provider-neutral and remains meaningful even when models improve.

+
-
-

How it works

Transactional governance, not token counting

-
-
01

Observe

Capture proposed actions and context through one engine-neutral protocol.

-
02

Evaluate

Estimate marginal gain, uncertainty and full economic cost.

-
03

Reserve

Atomically protect budget and verification capacity before execution.

-
04

Settle

Record measured usage, failures, overruns and verified outcomes.

-
05

Learn

Replay versioned policies and improve calibration without overstating causality.

+
+
+

Proof before claims

+

MARGINAL must earn its own compute.

+

Matched OFF/ON evaluation is the product test. Saving agent tokens is insufficient if governance overhead, quality loss or false stops erase the benefit.

+
+
+
01

Verified quality

Same verifier, predefined non-inferiority margin, regressions and recoveries exposed.

+
02

Gross savings

Agent workload change before counting MARGINAL's own cost.

+
03

Net savings

Effective tokens, USD and latency after the governance tax. This is the claim surface.

+
04

Repeated calls

Shows whether behavior changed or the system simply moved cost somewhere else.

+
05

False stops

Explicitly reviewed deny recommendations that would have blocked helpful work.

+
06

Uncertainty

Repeated runs and bootstrap intervals separate a stable effect from one lucky trajectory.

+
+
+ Primary target + Minimize effective compute per verified successful task, subject to quality and false-stop constraints.
-
-

Why it is different

Built for accountable autonomy

-
-

Shadow before enforcement

Observe recommendations without changing agent behavior, then enforce only after representative validation.

-

Real accounting lifecycle

Reserve, execute, settle, abort and report overruns across hierarchical treasuries.

-

Protected verification

Keep capacity available for tests and evidence instead of optimizing an agent into premature confidence.

-

Versioned evidence

Decision Ledger records correlate policy, estimator, cost, outcome and runtime identity for audit and replay.

-

Universal protocol

Thin adapters connect coding agents while the economic policy remains in one core.

-

Measured claims

Token reduction counts only when verified outcomes remain within a predefined quality constraint.

+
+
+

Graceful irrelevance

+

What if GPT-5.7 — or any future model — is already efficient?

+
+
+

Then MARGINAL should get out of the way. The evaluator now distinguishes gross from net savings and can classify a configuration as pass_through when the governor does not demonstrate enough net value.

+
$ marginal public-eval baseline.jsonl marginal.jsonlintervention.status: pass_throughNo positive net intervention value demonstrated.
+

A negative or neutral benchmark is a valid result. The project should not manufacture a reason to intervene simply to justify its own existence.

-
-

The learning loop

From static policy to evidence-driven allocation

-

MARGINAL's defensible advantage is not a single ROI formula. It is the closed loop between observed decisions, measured outcomes, versioned estimators, policy replay and calibration.

-

Task success is kept separate from action-level causal attribution. Recorded correlation is not presented as causal proof.

+
+

New control

State-aware diminishing returns, without a GPT-specific patch.

+
+
+
Same semantic keye.g. verify the same artifact for the same purpose
+
Same state hashthe underlying workspace has not changed
+
No new evidencethe previous pass did not change the evidence state
+
Repeat count risesexpected gain decays until a configured stop threshold
+
+
+

The detector is opt-in. Missing state fails open. Changed state or new evidence resets the pressure. That keeps the mechanism conservative while real agent telemetry is collected.

+

Exact duplicate prevention still exists. Diminishing-return control adds a semantic layer for allowed retries and verification loops where fingerprints can legitimately differ.

+ Read the evidence model → +
-
  1. Observe decisions
  2. Measure actual cost and outcomes
  3. Estimate marginal value with uncertainty
  4. Allocate budget
  5. Measure calibration and regret
  6. Improve the policy
-
-

Privacy by design

Operational evidence stays local. Shareable data is transformed deliberately.

-

Prompt-free telemetry can still leak through identifiers, names and metadata. MARGINAL classifies fields and provides explicit export profiles.

-
-
-
LOCAL_FULLComplete operational ledger for trusted local storage.
-
SAFE_TELEMETRYFree text removed and identifiers pseudonymized with a local key.
-
AGGREGATE_EXPORTGeneralized groups without identifiers; small groups suppressed by default.
+
+

Community pressure test

Feedback changes the product only when the reasoning survives review.

+
+
Accepted

Show OFF vs ON benchmarks

Matched runs are required for performance claims. The 10-task canary remains engineering validation, not marketing evidence.

+
Accepted

Future models may reduce the benefit

Converted into Graceful Irrelevance: MARGINAL must demonstrate positive net value for the workload in front of it.

+
Partially accepted

“Less slop” on the website

The technical concepts stay, but the landing page now starts with a concrete failure mode and proof standard before architecture theory.

+
Rejected

Providers intentionally preserve waste

There is no evidence needed or offered for that claim. MARGINAL is justified by user-controlled economics and observability, not provider motive speculation.

-
-

One product, multiple engines

Universal foundation first. Thin adapters next.

-
-
Available

Core runtime

Policy, treasury, ledger, privacy, replay and universal protocol.

-
Next

Codex

Reference adapter and matched measured benchmark.

-
Planned

OpenCode

Open-source adapter and experimentation environment.

-
Planned

Claude Code

Hook-based integration using shared protocol contracts.

-
Planned

GitHub Copilot

Integration where official control surfaces permit enforcement.

+
+

Benchmark discipline

One benchmark is a surface, not ground truth.

+
+

The requested SWE-bench Pro comparison can be useful because it is recognizable and gives the community a common reference. It should still be versioned, audited and accompanied by task-quality notes.

MARGINAL-specific evaluation should also target the behavior the product claims to change: redundant same-state actions, repeated verification, overhead, false stops and verified outcomes.

+
Same modelSame promptSame toolsSame limitsSame task orderSame verifierRaw paired JSONLPreregistered gates
-
-
-

Evidence, not hype

Optimization is successful only when quality is preserved.

-

The bundled Killer Demo is deterministic and uses declared action costs. It explains the mechanism; it is not provider telemetry or a universal savings claim.

-

Public evaluation requires matched model, prompt, tools, limits and verifier, then reports tokens per verified successful task, quality delta and uncertainty.

- Read the benchmark protocol → +
+

Next milestone

Codex Reference Integration

The core is being prepared now; the adapter and measured Codex runs remain the next implementation milestone.

+
+
v0.2Learning Loop Foundationuniversal protocol · ledger · privacy · replay
+
HardeningNet-value evidence layergovernance tax · false stops · diminishing returns
+
v0.3Codex integrationone-command target · telemetry · canary · public matched benchmark
+
-

Build economically disciplined agents

-

Observe first. Measure honestly. Enforce what the evidence supports.

+

Build the evidence first

+

Observe. Measure. Let intervention earn enforcement.

diff --git a/site/styles.css b/site/styles.css index 692786a..833a460 100644 --- a/site/styles.css +++ b/site/styles.css @@ -1,19 +1,144 @@ -:root{--bg:#080b10;--surface:#10151d;--surface2:#151c26;--text:#f4f7fb;--muted:#aab5c4;--line:#273140;--accent:#91ff63;--max:1180px} -*{box-sizing:border-box}html{scroll-behavior:smooth}body{margin:0;background:var(--bg);color:var(--text);font-family:Inter,ui-sans-serif,system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;line-height:1.65}a{color:inherit} -.skip-link{position:absolute;left:-9999px}.skip-link:focus{left:1rem;top:1rem;z-index:100;background:var(--text);color:var(--bg);padding:.75rem 1rem} -.shell{width:min(calc(100% - 2rem),var(--max));margin-inline:auto}.narrow{max-width:790px}.center{text-align:center;justify-content:center} -.site-header{position:sticky;top:0;z-index:20;background:rgba(8,11,16,.9);border-bottom:1px solid var(--line);backdrop-filter:blur(16px)} -.nav{min-height:72px;display:flex;align-items:center;justify-content:space-between;gap:2rem}.brand{display:inline-flex;align-items:center;gap:.7rem;text-decoration:none;font-weight:800;letter-spacing:.08em}.brand-mark{display:grid;place-items:center;width:34px;height:34px;color:var(--bg);background:var(--accent);border-radius:8px} -.nav-links{display:flex;align-items:center;gap:1.35rem}.nav-links a{text-decoration:none;color:var(--muted);font-size:.94rem}.nav-links a:hover,.nav-links a:focus{color:var(--text)}.nav-toggle{display:none} -.button{display:inline-flex;align-items:center;justify-content:center;min-height:48px;padding:.7rem 1.15rem;border-radius:10px;background:var(--accent);color:#071006;text-decoration:none;font-weight:750;border:1px solid var(--accent)}.button:hover,.button:focus{filter:brightness(1.08);transform:translateY(-1px)}.button-secondary{background:transparent;color:var(--text);border-color:var(--line)}.button-small{min-height:38px;padding:.45rem .8rem;color:#071006!important} -.hero{min-height:720px;display:grid;grid-template-columns:1.25fr .75fr;align-items:center;gap:4rem;padding-block:7rem 5rem}.eyebrow{color:var(--accent);text-transform:uppercase;letter-spacing:.14em;font-size:.78rem;font-weight:800} -h1{font-size:clamp(3rem,7vw,6.4rem);line-height:.96;letter-spacing:-.055em;margin:.5rem 0 1.4rem;max-width:900px}h2{font-size:clamp(2rem,4.2vw,4rem);line-height:1.05;letter-spacing:-.04em;margin:.45rem 0 1.2rem}h3{line-height:1.2}.hero-lead{color:var(--muted);font-size:clamp(1.1rem,2vw,1.35rem);max-width:720px}.hero-actions{display:flex;gap:.8rem;margin-top:2rem;flex-wrap:wrap}.trust-row{display:flex;flex-wrap:wrap;gap:.7rem 1.25rem;list-style:none;padding:0;margin:2.2rem 0 0;color:var(--muted);font-size:.9rem}.trust-row li:before{content:"✓";color:var(--accent);margin-right:.4rem} -.hero-panel{border:1px solid var(--line);background:linear-gradient(160deg,var(--surface2),var(--surface));padding:2rem;border-radius:20px;box-shadow:0 30px 80px rgba(0,0,0,.35)}.flow-node{padding:1rem;border:1px solid var(--line);border-radius:10px;text-align:center;background:rgba(255,255,255,.02)}.flow-node.accent{border-color:var(--accent)}.flow-arrow{text-align:center;color:var(--accent);padding:.4rem}.decision-grid{display:grid;grid-template-columns:repeat(3,1fr);gap:.5rem}.decision-grid span{border:1px solid var(--line);border-radius:8px;padding:.7rem .3rem;text-align:center;font-size:.75rem;font-weight:800} -.section{padding-block:6.5rem;border-top:1px solid var(--line)}.problem,.feature-section,.privacy-section,.evidence-section{background:var(--surface)}.problem{text-align:center}.problem p:last-child{color:var(--muted)}.section-heading{max-width:760px;margin-bottom:3rem} -.steps{display:grid;grid-template-columns:repeat(5,1fr);gap:1rem}.steps article,.feature-grid article,.roadmap-grid article{border:1px solid var(--line);border-radius:14px;padding:1.35rem;background:var(--surface)}.steps span,.roadmap-grid span{color:var(--accent);font-size:.8rem;font-weight:800;text-transform:uppercase;letter-spacing:.1em}.steps p,.feature-grid p,.roadmap-grid p{color:var(--muted);font-size:.94rem} -.feature-grid{display:grid;grid-template-columns:repeat(3,1fr);gap:1rem}.split{display:grid;grid-template-columns:1fr 1fr;gap:5rem;align-items:start}.loop-list{list-style:none;padding:0;margin:0;counter-reset:loop}.loop-list li{counter-increment:loop;padding:1rem 0;border-bottom:1px solid var(--line);font-weight:700}.loop-list li:before{content:"0" counter(loop);color:var(--accent);margin-right:1rem;font-size:.8rem}.fine-print{color:var(--muted);font-size:.9rem} -.privacy-cards{display:grid;gap:.8rem}.privacy-cards article{display:grid;gap:.2rem;padding:1.2rem;border:1px solid var(--line);border-radius:12px}.privacy-cards strong{color:var(--accent)}.privacy-cards span{color:var(--muted)}.roadmap-grid{display:grid;grid-template-columns:repeat(5,1fr);gap:1rem;margin-bottom:2.5rem}.roadmap-grid .complete{border-color:var(--accent)}.text-link{color:var(--accent);font-weight:750;text-decoration:none}.cta{background:radial-gradient(circle at 50% 20%,rgba(145,255,99,.13),transparent 45%)} -footer{padding-block:3rem;border-top:1px solid var(--line);color:var(--muted)}.footer-grid{display:grid;grid-template-columns:2fr 1fr 1fr;gap:2rem}.footer-grid div{display:flex;flex-direction:column;gap:.4rem;align-items:flex-start}.footer-grid p{margin:0}.footer-grid a{text-decoration:none} -@media(max-width:900px){.hero,.split{grid-template-columns:1fr}.hero{min-height:auto;gap:2.5rem;padding-top:5rem}.steps{grid-template-columns:repeat(2,1fr)}.feature-grid{grid-template-columns:repeat(2,1fr)}.roadmap-grid{grid-template-columns:repeat(2,1fr)}.nav-toggle{display:inline-flex;background:transparent;border:1px solid var(--line);color:var(--text);padding:.55rem .75rem;border-radius:8px}.nav-links{display:none;position:absolute;left:1rem;right:1rem;top:72px;background:var(--surface);border:1px solid var(--line);border-radius:12px;padding:1rem;flex-direction:column;align-items:stretch}.nav-links.open{display:flex}} -@media(max-width:600px){h1{font-size:3.2rem}.section{padding-block:4.5rem}.steps,.feature-grid,.roadmap-grid,.footer-grid{grid-template-columns:1fr}.decision-grid{grid-template-columns:1fr}} -@media(prefers-reduced-motion:reduce){html{scroll-behavior:auto}*,*:before,*:after{transition:none!important;animation:none!important}} +:root { + --bg: #080b10; + --surface: #10151d; + --surface-2: #151c26; + --surface-3: #0c1118; + --text: #f4f7fb; + --muted: #aab5c4; + --line: #273140; + --accent: #91ff63; + --accent-soft: rgba(145, 255, 99, 0.09); + --warning: #ffd166; + --danger: #ff7b72; + --max: 1180px; +} + +* { box-sizing: border-box; } +html { scroll-behavior: smooth; } +body { + margin: 0; + background: var(--bg); + color: var(--text); + font-family: Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; + line-height: 1.65; +} +a { color: inherit; } +code { font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace; color: var(--accent); } +.skip-link { position: absolute; left: -9999px; } +.skip-link:focus { left: 1rem; top: 1rem; z-index: 100; background: var(--text); color: var(--bg); padding: .75rem 1rem; } +.shell { width: min(calc(100% - 2rem), var(--max)); margin-inline: auto; } +.narrow { max-width: 820px; } +.center { text-align: center; justify-content: center; } + +.site-header { position: sticky; top: 0; z-index: 20; background: rgba(8,11,16,.9); border-bottom: 1px solid var(--line); backdrop-filter: blur(16px); } +.nav { min-height: 72px; display: flex; align-items: center; justify-content: space-between; gap: 2rem; } +.brand { display: inline-flex; align-items: center; gap: .7rem; text-decoration: none; font-weight: 800; letter-spacing: .08em; } +.brand-mark { display: grid; place-items: center; width: 34px; height: 34px; color: var(--bg); background: var(--accent); border-radius: 8px; } +.nav-links { display: flex; align-items: center; gap: 1.25rem; } +.nav-links a { text-decoration: none; color: var(--muted); font-size: .92rem; } +.nav-links a:hover, .nav-links a:focus { color: var(--text); } +.nav-toggle { display: none; } + +.button { display: inline-flex; align-items: center; justify-content: center; min-height: 48px; padding: .7rem 1.15rem; border-radius: 10px; background: var(--accent); color: #071006; text-decoration: none; font-weight: 760; border: 1px solid var(--accent); } +.button:hover, .button:focus { filter: brightness(1.08); transform: translateY(-1px); } +.button-secondary { background: transparent; color: var(--text); border-color: var(--line); } +.button-small { min-height: 38px; padding: .45rem .8rem; color: #071006 !important; } + +.hero { min-height: 760px; display: grid; grid-template-columns: 1.08fr .92fr; align-items: center; gap: 4.5rem; padding-block: 7rem 5rem; } +.eyebrow { color: var(--accent); text-transform: uppercase; letter-spacing: .14em; font-size: .77rem; font-weight: 820; } +h1 { font-size: clamp(3rem, 6.7vw, 6rem); line-height: .98; letter-spacing: -.055em; margin: .5rem 0 1.4rem; max-width: 900px; } +h2 { font-size: clamp(2rem, 4.2vw, 4rem); line-height: 1.05; letter-spacing: -.04em; margin: .45rem 0 1.2rem; } +h3 { line-height: 1.2; } +.hero-lead { color: var(--muted); font-size: clamp(1.08rem, 2vw, 1.3rem); max-width: 720px; } +.hero-actions { display: flex; gap: .8rem; margin-top: 2rem; flex-wrap: wrap; } +.trust-row { display: flex; flex-wrap: wrap; gap: .7rem 1.25rem; list-style: none; padding: 0; margin: 2.2rem 0 0; color: var(--muted); font-size: .9rem; } +.trust-row li::before { content: "✓"; color: var(--accent); margin-right: .4rem; } + +.trace-card { border: 1px solid var(--line); background: linear-gradient(160deg, var(--surface-2), var(--surface)); padding: 1.25rem; border-radius: 20px; box-shadow: 0 30px 80px rgba(0,0,0,.36); } +.trace-label { display: flex; justify-content: space-between; gap: 1rem; color: var(--muted); font-size: .78rem; text-transform: uppercase; letter-spacing: .08em; margin-bottom: .9rem; } +.trace-label strong { color: var(--warning); } +.trace-row { display: grid; grid-template-columns: 34px 1fr auto; gap: .8rem; align-items: center; padding: 1rem .65rem; border-top: 1px solid var(--line); } +.trace-num { color: var(--muted); font-family: ui-monospace, monospace; font-size: .78rem; } +.trace-row div { display: grid; } +.trace-row small { color: var(--muted); } +.status { font-size: .68rem; font-weight: 850; letter-spacing: .08em; border-radius: 999px; padding: .28rem .5rem; border: 1px solid var(--line); } +.status.ok { color: var(--accent); } +.status.warn { color: var(--warning); } +.status.stop { color: var(--danger); } +.trace-note { color: var(--muted); font-size: .83rem; margin: .85rem .65rem .2rem; } + +.section { padding-block: 6.5rem; border-top: 1px solid var(--line); } +.thesis, .community-section, .roadmap-section { background: var(--surface); } +.dark-band { background: #06080c; } +.section-heading { max-width: 790px; margin-bottom: 3rem; } +.section-heading > p:last-child, .body-copy { color: var(--muted); } +.body-copy strong { color: var(--text); } +.split { display: grid; grid-template-columns: 1fr 1fr; gap: 5rem; align-items: start; } +.compact-gap { gap: 3rem; } + +.metric-grid { display: grid; grid-template-columns: repeat(3, 1fr); gap: 1rem; } +.metric-grid article, .decision-cards article { border: 1px solid var(--line); border-radius: 14px; padding: 1.35rem; background: var(--surface); } +.metric-grid article > span:first-child { color: var(--accent); font-size: .75rem; font-weight: 850; letter-spacing: .1em; } +.metric-grid p, .decision-cards p { color: var(--muted); font-size: .94rem; } +.metric-grid .accent-card { border-color: var(--accent); background: linear-gradient(160deg, var(--accent-soft), var(--surface)); } +.evidence-banner { margin-top: 1rem; border: 1px solid var(--line); border-radius: 14px; padding: 1.15rem 1.35rem; display: grid; grid-template-columns: auto 1fr; gap: 1.2rem; align-items: center; } +.evidence-key { color: var(--accent); font-size: .75rem; font-weight: 850; text-transform: uppercase; letter-spacing: .1em; } + +.terminal { display: grid; gap: .4rem; margin: 1.5rem 0; border: 1px solid var(--line); border-radius: 12px; padding: 1.2rem; background: #05080c; font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace; } +.terminal span, .terminal small { color: var(--muted); } +.terminal b { color: var(--accent); } +.fine-print { color: var(--muted); font-size: .88rem; } + +.logic-list { border-top: 1px solid var(--line); } +.logic-list div { display: grid; grid-template-columns: 180px 1fr; gap: 1rem; border-bottom: 1px solid var(--line); padding: 1rem 0; } +.logic-list span { color: var(--muted); } +.text-link { color: var(--accent); font-weight: 760; text-decoration: none; } + +.decision-cards { display: grid; grid-template-columns: repeat(2, 1fr); gap: 1rem; } +.decision { display: inline-flex; border-radius: 999px; padding: .25rem .55rem; font-size: .69rem; font-weight: 850; letter-spacing: .08em; text-transform: uppercase; border: 1px solid var(--line); } +.decision.accept { color: var(--accent); } +.decision.partial { color: var(--warning); } +.decision.reject { color: var(--danger); } + +.proof-list { display: grid; grid-template-columns: repeat(2, 1fr); gap: .7rem; } +.proof-list span { border: 1px solid var(--line); border-radius: 9px; padding: .8rem; color: var(--muted); } +.proof-list span::before { content: "✓"; color: var(--accent); margin-right: .5rem; } + +.roadmap-line { display: grid; grid-template-columns: repeat(3, 1fr); gap: 1rem; margin-bottom: 2.5rem; } +.roadmap-line div { min-height: 170px; display: flex; flex-direction: column; gap: .45rem; border: 1px solid var(--line); border-radius: 14px; padding: 1.35rem; } +.roadmap-line span { color: var(--muted); font-size: .74rem; font-weight: 850; text-transform: uppercase; letter-spacing: .1em; } +.roadmap-line small { color: var(--muted); } +.roadmap-line .done { border-color: rgba(145,255,99,.4); } +.roadmap-line .current { border-color: var(--accent); background: var(--accent-soft); } + +.cta { background: radial-gradient(circle at 50% 20%, rgba(145,255,99,.13), transparent 45%); } +footer { padding-block: 3rem; border-top: 1px solid var(--line); color: var(--muted); } +.footer-grid { display: grid; grid-template-columns: 2fr 1fr 1fr; gap: 2rem; } +.footer-grid div { display: flex; flex-direction: column; gap: .4rem; align-items: flex-start; } +.footer-grid p { margin: 0; } +.footer-grid a { text-decoration: none; } + +@media (max-width: 900px) { + .hero, .split { grid-template-columns: 1fr; } + .hero { min-height: auto; gap: 2.5rem; padding-top: 5rem; } + .metric-grid { grid-template-columns: repeat(2, 1fr); } + .roadmap-line { grid-template-columns: 1fr; } + .nav-toggle { display: inline-flex; background: transparent; border: 1px solid var(--line); color: var(--text); padding: .55rem .75rem; border-radius: 8px; } + .nav-links { display: none; position: absolute; left: 1rem; right: 1rem; top: 72px; background: var(--surface); border: 1px solid var(--line); border-radius: 12px; padding: 1rem; flex-direction: column; align-items: stretch; } + .nav-links.open { display: flex; } +} + +@media (max-width: 620px) { + h1 { font-size: 3.1rem; } + .section { padding-block: 4.5rem; } + .metric-grid, .decision-cards, .footer-grid, .proof-list { grid-template-columns: 1fr; } + .trace-row { grid-template-columns: 28px 1fr; } + .trace-row .status { grid-column: 2; justify-self: start; } + .logic-list div { grid-template-columns: 1fr; gap: .25rem; } + .evidence-banner { grid-template-columns: 1fr; } +} + +@media (prefers-reduced-motion: reduce) { + html { scroll-behavior: auto; } + *, *::before, *::after { transition: none !important; animation: none !important; } +} diff --git a/src/marginal/__init__.py b/src/marginal/__init__.py index e393076..ef38b5c 100644 --- a/src/marginal/__init__.py +++ b/src/marginal/__init__.py @@ -12,6 +12,12 @@ funded_call, ) from .budget import BudgetExceeded, BudgetLedger, BudgetLimits, BudgetOverrun, BudgetUsage +from .controls import ( + DiminishingReturnConfig, + DiminishingReturnDetector, + DiminishingReturnSignal, + GovernanceTracker, +) from .estimator import EstimatorIdentity, ValueEstimate, ValueEstimator from .killer_demo import run_killer_demo from .ledger import ( @@ -82,10 +88,14 @@ "Decision", "DecisionLedgerContext", "DeduplicationScope", + "DiminishingReturnConfig", + "DiminishingReturnDetector", + "DiminishingReturnSignal", "EstimatorIdentity", "EstimatorRegistry", "ExecutionMode", "FailureUsageExtractor", + "GovernanceTracker", "JsonlDecisionLedger", "JsonlTraceSink", "LocalPseudonymizer", diff --git a/src/marginal/cli.py b/src/marginal/cli.py index 97ad644..456ca81 100644 --- a/src/marginal/cli.py +++ b/src/marginal/cli.py @@ -128,6 +128,18 @@ def _build_parser() -> argparse.ArgumentParser: public_eval.add_argument("--bootstrap-samples", type=int, default=2_000) public_eval.add_argument("--confidence-level", type=float, default=0.95) public_eval.add_argument("--quality-margin-pp", type=float, default=1.0) + public_eval.add_argument( + "--minimum-net-token-savings-percent", + type=float, + default=0.0, + help="minimum net token saving required before intervention is classified supported", + ) + public_eval.add_argument( + "--max-false-stop-rate", + type=float, + default=0.0, + help="maximum reviewed false-stop rate allowed for a supported intervention", + ) public_eval.add_argument("--seed", type=int, default=42) return parser @@ -209,6 +221,8 @@ def main(argv: Sequence[str] | None = None) -> int: bootstrap_samples=args.bootstrap_samples, confidence_level=args.confidence_level, quality_margin_pp=args.quality_margin_pp, + minimum_net_token_savings_percent=(args.minimum_net_token_savings_percent), + max_false_stop_rate=args.max_false_stop_rate, seed=args.seed, ) except (OSError, ValueError) as exc: diff --git a/src/marginal/controls/__init__.py b/src/marginal/controls/__init__.py new file mode 100644 index 0000000..b341b57 --- /dev/null +++ b/src/marginal/controls/__init__.py @@ -0,0 +1,15 @@ +"""Optional controls that harden MARGINAL without coupling the core to one engine.""" + +from .diminishing import ( + DiminishingReturnConfig, + DiminishingReturnDetector, + DiminishingReturnSignal, +) +from .governance import GovernanceTracker + +__all__ = [ + "DiminishingReturnConfig", + "DiminishingReturnDetector", + "DiminishingReturnSignal", + "GovernanceTracker", +] diff --git a/src/marginal/controls/diminishing.py b/src/marginal/controls/diminishing.py new file mode 100644 index 0000000..5346c26 --- /dev/null +++ b/src/marginal/controls/diminishing.py @@ -0,0 +1,171 @@ +"""State-aware diminishing-return detection for repeated agent actions. + +The detector is intentionally provider neutral. It does not special-case a model, tool, +or file type. A repeat only becomes less valuable when the same semantic action is proposed +against the same observable state without new evidence. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass + +from ..models import Action + + +@dataclass(frozen=True, slots=True) +class DiminishingReturnConfig: + """Conservative controls for state-aware repetition.""" + + gain_decay: float = 0.5 + max_same_state_repeats: int = 2 + + def __post_init__(self) -> None: + if isinstance(self.gain_decay, bool) or not isinstance(self.gain_decay, (int, float)): + raise TypeError("gain_decay must be a number") + decay = float(self.gain_decay) + if not math.isfinite(decay) or not 0.0 < decay <= 1.0: + raise ValueError("gain_decay must be finite and in (0, 1]") + if isinstance(self.max_same_state_repeats, bool) or not isinstance( + self.max_same_state_repeats, int + ): + raise TypeError("max_same_state_repeats must be an integer") + if self.max_same_state_repeats < 1: + raise ValueError("max_same_state_repeats must be at least 1") + object.__setattr__(self, "gain_decay", decay) + + +@dataclass(frozen=True, slots=True) +class DiminishingReturnSignal: + """Explain how prior same-state executions affect the next action.""" + + semantic_key: str + same_state_repeats: int + gain_multiplier: float + should_stop: bool + reason_code: str + reason: str + + +@dataclass(slots=True) +class _Observation: + state_hash: str + evidence_hash: str + executions_in_state: int + + +class DiminishingReturnDetector: + """Track repeated semantic work without confusing proposal with execution. + + ``evaluate`` is pure: it never advances history. Call ``observe`` only after an action + actually executes successfully. Missing state information fails open because MARGINAL + should not invent certainty it cannot observe. + """ + + def __init__(self, config: DiminishingReturnConfig | None = None) -> None: + self.config = config or DiminishingReturnConfig() + self._observations: dict[str, _Observation] = {} + + def evaluate(self, action: Action) -> DiminishingReturnSignal: + semantic_key = self._semantic_key(action) + state_hash = self._metadata_text(action, "state_hash") + evidence_hash = self._metadata_text(action, "evidence_hash") + + if not state_hash: + return DiminishingReturnSignal( + semantic_key=semantic_key, + same_state_repeats=0, + gain_multiplier=1.0, + should_stop=False, + reason_code="DIMINISHING_RETURN_UNOBSERVABLE", + reason="state is not observable; repetition control fails open", + ) + + previous = self._observations.get(semantic_key) + if ( + previous is None + or previous.state_hash != state_hash + or ( + evidence_hash + and (not previous.evidence_hash or previous.evidence_hash != evidence_hash) + ) + ): + repeats = 0 + else: + repeats = previous.executions_in_state + + multiplier = self.config.gain_decay**repeats + should_stop = repeats >= self.config.max_same_state_repeats + if should_stop: + return DiminishingReturnSignal( + semantic_key=semantic_key, + same_state_repeats=repeats, + gain_multiplier=multiplier, + should_stop=True, + reason_code="DIMINISHING_RETURN_REJECTED", + reason=( + "same semantic action has already executed " + f"{repeats} time(s) against unchanged state without new evidence" + ), + ) + if repeats: + return DiminishingReturnSignal( + semantic_key=semantic_key, + same_state_repeats=repeats, + gain_multiplier=multiplier, + should_stop=False, + reason_code="DIMINISHING_RETURN_DISCOUNTED", + reason=( + f"same-state repetition detected; expected gain discounted by {multiplier:.3f}" + ), + ) + return DiminishingReturnSignal( + semantic_key=semantic_key, + same_state_repeats=0, + gain_multiplier=1.0, + should_stop=False, + reason_code="DIMINISHING_RETURN_CLEAR", + reason="new state or new evidence; no repetition penalty", + ) + + def observe(self, action: Action) -> None: + """Record one action only after the caller confirms it executed.""" + + semantic_key = self._semantic_key(action) + state_hash = self._metadata_text(action, "state_hash") + evidence_hash = self._metadata_text(action, "evidence_hash") + if not state_hash: + return + + previous = self._observations.get(semantic_key) + same_state = previous is not None and previous.state_hash == state_hash + same_evidence = previous is not None and ( + (not evidence_hash and not previous.evidence_hash) + or evidence_hash == previous.evidence_hash + ) + executions = ( + previous.executions_in_state + 1 + if previous is not None and same_state and same_evidence + else 1 + ) + self._observations[semantic_key] = _Observation( + state_hash=state_hash, + evidence_hash=evidence_hash, + executions_in_state=executions, + ) + + def reset(self) -> None: + self._observations.clear() + + @staticmethod + def _metadata_text(action: Action, key: str) -> str: + value = action.metadata.get(key, "") + return value.strip() if isinstance(value, str) else str(value).strip() if value else "" + + @classmethod + def _semantic_key(cls, action: Action) -> str: + explicit = cls._metadata_text(action, "marginal_semantic_key") + if explicit: + return explicit + phase = cls._metadata_text(action, "phase") + return "|".join((action.kind.strip().lower(), action.name.strip().lower(), phase.lower())) diff --git a/src/marginal/controls/governance.py b/src/marginal/controls/governance.py new file mode 100644 index 0000000..ce5296f --- /dev/null +++ b/src/marginal/controls/governance.py @@ -0,0 +1,77 @@ +"""First-class accounting for the cost and mistakes of MARGINAL itself.""" + +from __future__ import annotations + +import math +from dataclasses import dataclass +from typing import Any + + +@dataclass(slots=True) +class GovernanceTracker: + """Account for governance overhead and explicitly reviewed false stops. + + False stops are never inferred from task success. They are counted only when an external + reviewer or counterfactual process explicitly labels a denied recommendation as an action + that would have helped. + """ + + decisions: int = 0 + decision_latency_ms: float = 0.0 + external_tokens: int = 0 + external_usd: float = 0.0 + external_latency_ms: float = 0.0 + reviewed_stops: int = 0 + false_stops: int = 0 + + def record_decision(self, *, latency_ms: float = 0.0) -> None: + self._validate_non_negative_number("latency_ms", latency_ms) + self.decisions += 1 + self.decision_latency_ms += float(latency_ms) + + def record_external_overhead( + self, + *, + tokens: int = 0, + usd: float = 0.0, + latency_ms: float = 0.0, + ) -> None: + if isinstance(tokens, bool) or not isinstance(tokens, int): + raise TypeError("tokens must be an integer") + if tokens < 0: + raise ValueError("tokens must be non-negative") + self._validate_non_negative_number("usd", usd) + self._validate_non_negative_number("latency_ms", latency_ms) + self.external_tokens += tokens + self.external_usd += float(usd) + self.external_latency_ms += float(latency_ms) + + def record_stop_review(self, *, would_have_helped: bool) -> None: + if not isinstance(would_have_helped, bool): + raise TypeError("would_have_helped must be a boolean") + self.reviewed_stops += 1 + if would_have_helped: + self.false_stops += 1 + + def summary(self) -> dict[str, Any]: + false_stop_rate = self.false_stops / self.reviewed_stops if self.reviewed_stops else None + total_latency = self.decision_latency_ms + self.external_latency_ms + return { + "scope": "treasury_tree", + "decisions": self.decisions, + "decision_latency_ms": round(self.decision_latency_ms, 6), + "external_tokens": self.external_tokens, + "external_usd": round(self.external_usd, 9), + "external_latency_ms": round(self.external_latency_ms, 6), + "total_latency_ms": round(total_latency, 6), + "reviewed_stops": self.reviewed_stops, + "false_stops": self.false_stops, + "false_stop_rate": round(false_stop_rate, 6) if false_stop_rate is not None else None, + } + + @staticmethod + def _validate_non_negative_number(name: str, value: float) -> None: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number") + if not math.isfinite(float(value)) or value < 0: + raise ValueError(f"{name} must be finite and non-negative") diff --git a/src/marginal/killer_demo.py b/src/marginal/killer_demo.py index 2eab027..d67fd1b 100644 --- a/src/marginal/killer_demo.py +++ b/src/marginal/killer_demo.py @@ -340,6 +340,7 @@ def run_killer_demo(output_dir: str | Path | None = None) -> dict[str, Any]: policy=policy, trace_sink=trace, name="killer-demo", + clock=lambda: 0, ) stage_results: list[dict[str, Any]] = [] @@ -1959,7 +1960,7 @@ def render_killer_demo_html(result: dict[str, Any]) -> str: Results Trace Benchmark diff --git a/src/marginal/policy.py b/src/marginal/policy.py index 009dbf3..a628e15 100644 --- a/src/marginal/policy.py +++ b/src/marginal/policy.py @@ -8,6 +8,7 @@ from dataclasses import asdict, dataclass from .budget import BudgetLedger +from .controls import DiminishingReturnDetector, DiminishingReturnSignal from .estimator import EstimatorIdentity, ValueEstimate, ValueEstimator from .models import Action, Decision @@ -65,7 +66,12 @@ def __post_init__(self) -> None: class MarginalPolicy: - """Authorize actions only when expected marginal value justifies total cost.""" + """Authorize actions only when expected marginal value justifies total cost. + + State-aware diminishing-return control is opt-in. This preserves v0.2 behavior while + allowing engine adapters to enable repetition control first in Shadow/Recommend mode and + promote it to enforcement only after measured validation. + """ def __init__( self, @@ -74,9 +80,11 @@ def __init__( *, name: str = "marginal-reference", version: str = "2.0.0", + diminishing_detector: DiminishingReturnDetector | None = None, ) -> None: self.config = config or PolicyConfig() self.estimator = estimator or ValueEstimator() + self.diminishing_detector = diminishing_detector if not isinstance(name, str): raise TypeError("name must be a string") if not name.strip(): @@ -96,9 +104,24 @@ def __init__( self._executed_fingerprints: set[str] = set() def mark_executed(self, fingerprint: str) -> None: + """Backward-compatible exact duplicate accounting.""" + if fingerprint: self._executed_fingerprints.add(fingerprint) + def observe_execution(self, action: Action) -> None: + """Record one successfully executed action for exact and semantic repetition control.""" + + if action.fingerprint: + self.mark_executed(action.fingerprint) + if self.diminishing_detector is not None: + self.diminishing_detector.observe(action) + + def diminishing_signal(self, action: Action) -> DiminishingReturnSignal | None: + if self.diminishing_detector is None: + return None + return self.diminishing_detector.evaluate(action) + def evaluate(self, action: Action, ledger: BudgetLedger) -> Decision: if action.current_success_probability >= self.config.target_success_probability: return self._decision( @@ -117,12 +140,21 @@ def evaluate(self, action: Action, ledger: BudgetLedger) -> Decision: "BUDGET_REJECTED", ) + diminishing = self.diminishing_signal(action) + if diminishing is not None and diminishing.should_stop: + return self._decision( + False, + f"rejected: {diminishing.reason}", + diminishing.reason_code, + ) + estimate = self._estimate(action) + gain_multiplier = diminishing.gain_multiplier if diminishing is not None else 1.0 remaining_probability = max( 0.0, self.config.target_success_probability - action.current_success_probability, ) - expected_gain = min(estimate.expected_gain, remaining_probability) + expected_gain = min(estimate.expected_gain * gain_multiplier, remaining_probability) if expected_gain < self.config.minimum_expected_gain: return self._decision( False, diff --git a/src/marginal/public_eval.py b/src/marginal/public_eval.py index db75eb0..2596b12 100644 --- a/src/marginal/public_eval.py +++ b/src/marginal/public_eval.py @@ -7,12 +7,17 @@ import random from dataclasses import dataclass from pathlib import Path -from typing import Any +from typing import Any, cast @dataclass(frozen=True, slots=True) class RunRecord: - """Measured result for one benchmark instance.""" + """Measured result for one benchmark instance. + + ``tokens/usd/latency_ms`` describe the agent workload. Governance overhead is stored + separately so reports can show both gross savings and net savings after MARGINAL's own + cost. Existing v0.2 JSONL rows remain valid because all new fields default to zero. + """ instance_id: str resolved: bool @@ -20,25 +25,54 @@ class RunRecord: usd: float = 0.0 latency_ms: int = 0 tool_calls: int = 0 + repeated_calls: int = 0 + governance_tokens: int = 0 + governance_usd: float = 0.0 + governance_latency_ms: int = 0 + reviewed_stops: int = 0 + false_stops: int = 0 def __post_init__(self) -> None: if not isinstance(self.instance_id, str) or not self.instance_id: raise ValueError("instance_id must not be empty") if not isinstance(self.resolved, bool): raise TypeError("resolved must be a boolean") - for name in ("tokens", "latency_ms", "tool_calls"): + for name in ( + "tokens", + "latency_ms", + "tool_calls", + "repeated_calls", + "governance_tokens", + "governance_latency_ms", + "reviewed_stops", + "false_stops", + ): value = getattr(self, name) if isinstance(value, bool) or not isinstance(value, int): raise TypeError(f"{name} must be an integer") if value < 0: raise ValueError("metrics must be non-negative") - if isinstance(self.usd, bool) or not isinstance(self.usd, (int, float)): - raise TypeError("usd must be a number") - if not math.isfinite(float(self.usd)): - raise ValueError("usd must be finite") - if self.usd < 0: - raise ValueError("metrics must be non-negative") - object.__setattr__(self, "usd", float(self.usd)) + for name in ("usd", "governance_usd"): + value = getattr(self, name) + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number") + if not math.isfinite(float(value)) or value < 0: + raise ValueError("metrics must be finite and non-negative") + object.__setattr__(self, name, float(value)) + if self.false_stops > self.reviewed_stops: + raise ValueError("false_stops cannot exceed reviewed_stops") + + @property + def effective_tokens(self) -> int: + return self.tokens + self.governance_tokens + + @property + def effective_usd(self) -> float: + return self.usd + self.governance_usd + + @property + def effective_latency_ms(self) -> int: + return self.latency_ms + self.governance_latency_ms def load_runs(path: Path) -> dict[str, RunRecord]: @@ -60,6 +94,12 @@ def load_runs(path: Path) -> dict[str, RunRecord]: usd=item.get("usd", 0.0), latency_ms=item.get("latency_ms", 0), tool_calls=item.get("tool_calls", 0), + repeated_calls=item.get("repeated_calls", 0), + governance_tokens=item.get("governance_tokens", 0), + governance_usd=item.get("governance_usd", 0.0), + governance_latency_ms=item.get("governance_latency_ms", 0), + reviewed_stops=item.get("reviewed_stops", 0), + false_stops=item.get("false_stops", 0), ) except (KeyError, TypeError, ValueError, json.JSONDecodeError) as exc: raise ValueError(f"invalid benchmark row on line {line_number}") from exc @@ -74,14 +114,32 @@ def load_runs(path: Path) -> dict[str, RunRecord]: def _aggregate(records: list[RunRecord]) -> dict[str, Any]: tasks = len(records) resolved = sum(record.resolved for record in records) + governance_tokens = sum(record.governance_tokens for record in records) + governance_usd = sum(record.governance_usd for record in records) + governance_latency_ms = sum(record.governance_latency_ms for record in records) + tokens = sum(record.tokens for record in records) + usd = sum(record.usd for record in records) + latency_ms = sum(record.latency_ms for record in records) + reviewed_stops = sum(record.reviewed_stops for record in records) + false_stops = sum(record.false_stops for record in records) return { "tasks": tasks, "resolved": resolved, "resolve_rate": resolved / tasks, - "tokens": sum(record.tokens for record in records), - "usd": round(sum(record.usd for record in records), 6), - "latency_ms": sum(record.latency_ms for record in records), + "tokens": tokens, + "usd": round(usd, 6), + "latency_ms": latency_ms, "tool_calls": sum(record.tool_calls for record in records), + "repeated_calls": sum(record.repeated_calls for record in records), + "governance_tokens": governance_tokens, + "governance_usd": round(governance_usd, 6), + "governance_latency_ms": governance_latency_ms, + "effective_tokens": tokens + governance_tokens, + "effective_usd": round(usd + governance_usd, 6), + "effective_latency_ms": latency_ms + governance_latency_ms, + "reviewed_stops": reviewed_stops, + "false_stops": false_stops, + "false_stop_rate": false_stops / reviewed_stops if reviewed_stops else None, } @@ -94,6 +152,8 @@ def _bootstrap_token_savings( samples: int, seed: int, confidence_level: float, + *, + include_governance: bool, ) -> tuple[float, float]: if samples <= 0: raise ValueError("bootstrap_samples must be positive") @@ -103,8 +163,12 @@ def _bootstrap_token_savings( estimates: list[float] = [] for _ in range(samples): draw = [pairs[rng.randrange(len(pairs))] for _ in pairs] - baseline = sum(left.tokens for left, _ in draw) - marginal = sum(right.tokens for _, right in draw) + if include_governance: + baseline = sum(left.effective_tokens for left, _ in draw) + marginal = sum(right.effective_tokens for _, right in draw) + else: + baseline = sum(left.tokens for left, _ in draw) + marginal = sum(right.tokens for _, right in draw) estimates.append(_saving(float(baseline), float(marginal))) estimates.sort() tail = (1.0 - confidence_level) / 2.0 @@ -121,16 +185,37 @@ def compare_runs( seed: int = 42, confidence_level: float = 0.95, quality_margin_pp: float = 1.0, + minimum_net_token_savings_percent: float = 0.0, + max_false_stop_rate: float = 0.0, ) -> dict[str, Any]: - """Compare matched executions without imputing missing tasks.""" + """Compare matched executions without imputing missing tasks. + + The intervention earns a ``supported`` status only when quality is preserved, reviewed + false stops stay within the configured threshold, and net token savings after governance + overhead exceed the configured minimum. Otherwise MARGINAL should be treated as unsafe or + pass through rather than manufacturing a savings claim. + """ - if isinstance(quality_margin_pp, bool) or not isinstance(quality_margin_pp, (int, float)): - raise TypeError("quality_margin_pp must be a number") - if not math.isfinite(float(quality_margin_pp)) or quality_margin_pp < 0: - raise ValueError("quality_margin_pp must be finite and non-negative") + for name, value in ( + ("quality_margin_pp", quality_margin_pp), + ("minimum_net_token_savings_percent", minimum_net_token_savings_percent), + ("max_false_stop_rate", max_false_stop_rate), + ): + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number") + if not math.isfinite(float(value)): + raise ValueError(f"{name} must be finite") + if quality_margin_pp < 0: + raise ValueError("quality_margin_pp must be non-negative") + if minimum_net_token_savings_percent < 0: + raise ValueError("minimum_net_token_savings_percent must be non-negative") + if not 0.0 <= max_false_stop_rate <= 1.0: + raise ValueError("max_false_stop_rate must be between 0 and 1") if not 0.0 < confidence_level < 1.0: raise ValueError("confidence_level must be between 0 and 1") quality_margin_pp = float(quality_margin_pp) + minimum_net_token_savings_percent = float(minimum_net_token_savings_percent) + max_false_stop_rate = float(max_false_stop_rate) if set(baseline) != set(marginal): missing = sorted(set(baseline) ^ set(marginal)) @@ -144,46 +229,115 @@ def compare_runs( (marginal_total["resolve_rate"] - baseline_total["resolve_rate"]) * 100.0, 2, ) - token_ci = _bootstrap_token_savings( - list(zip(baseline_rows, marginal_rows, strict=True)), + pairs = list(zip(baseline_rows, marginal_rows, strict=True)) + net_token_ci = _bootstrap_token_savings( + pairs, bootstrap_samples, seed, confidence_level, + include_governance=True, + ) + gross_token_ci = _bootstrap_token_savings( + pairs, + bootstrap_samples, + seed, + confidence_level, + include_governance=False, ) - def efficiency(total: dict[str, Any]) -> dict[str, float | None]: + def efficiency(total: dict[str, Any], *, include_governance: bool) -> dict[str, float | None]: resolved = int(total["resolved"]) if resolved == 0: return {"tokens_per_resolved": None, "usd_per_resolved": None} + token_key = "effective_tokens" if include_governance else "tokens" + usd_key = "effective_usd" if include_governance else "usd" return { - "tokens_per_resolved": round(float(total["tokens"]) / resolved, 6), - "usd_per_resolved": round(float(total["usd"]) / resolved, 6), + "tokens_per_resolved": round(float(total[token_key]) / resolved, 6), + "usd_per_resolved": round(float(total[usd_key]) / resolved, 6), } + gross_savings = { + "tokens_percent": _saving(float(baseline_total["tokens"]), float(marginal_total["tokens"])), + "usd_percent": _saving(float(baseline_total["usd"]), float(marginal_total["usd"])), + "latency_percent": _saving( + float(baseline_total["latency_ms"]), + float(marginal_total["latency_ms"]), + ), + "tool_calls_percent": _saving( + float(baseline_total["tool_calls"]), + float(marginal_total["tool_calls"]), + ), + "repeated_calls_percent": _saving( + float(baseline_total["repeated_calls"]), + float(marginal_total["repeated_calls"]), + ), + "confidence_level": confidence_level, + "tokens_confidence_interval": list(gross_token_ci), + } + net_savings = { + "tokens_percent": _saving( + float(baseline_total["effective_tokens"]), + float(marginal_total["effective_tokens"]), + ), + "usd_percent": _saving( + float(baseline_total["effective_usd"]), + float(marginal_total["effective_usd"]), + ), + "latency_percent": _saving( + float(baseline_total["effective_latency_ms"]), + float(marginal_total["effective_latency_ms"]), + ), + "tool_calls_percent": gross_savings["tool_calls_percent"], + "repeated_calls_percent": gross_savings["repeated_calls_percent"], + "confidence_level": confidence_level, + "tokens_confidence_interval": list(net_token_ci), + "tokens_95pct_ci": list(net_token_ci), + } + + quality_margin_value = float(quality_margin_pp) + quality_preserved = delta_pp >= -quality_margin_value + false_stop_rate = marginal_total["false_stop_rate"] + false_stops_acceptable = false_stop_rate is None or float(false_stop_rate) <= float( + max_false_stop_rate + ) + if not quality_preserved: + intervention_status = "quality_regression" + elif not false_stops_acceptable: + intervention_status = "false_stop_risk" + elif float(cast(float, net_savings["tokens_percent"])) <= float( + minimum_net_token_savings_percent + ): + intervention_status = "pass_through" + else: + intervention_status = "supported" + return { - "benchmark": "public-agent-benchmark-comparison-v2", + "benchmark": "public-agent-benchmark-comparison-v3", "tasks": len(ids), "baseline": baseline_total, "marginal": marginal_total, "efficiency": { - "baseline": efficiency(baseline_total), - "marginal": efficiency(marginal_total), + "baseline": efficiency(baseline_total, include_governance=True), + "marginal": efficiency(marginal_total, include_governance=True), }, - "savings": { - "tokens_percent": _saving(baseline_total["tokens"], marginal_total["tokens"]), - "usd_percent": _saving(baseline_total["usd"], marginal_total["usd"]), - "latency_percent": _saving(baseline_total["latency_ms"], marginal_total["latency_ms"]), - "tool_calls_percent": _saving( - baseline_total["tool_calls"], marginal_total["tool_calls"] - ), - "confidence_level": confidence_level, - "tokens_confidence_interval": list(token_ci), - "tokens_95pct_ci": list(token_ci), + "agent_only_efficiency": { + "baseline": efficiency(baseline_total, include_governance=False), + "marginal": efficiency(marginal_total, include_governance=False), + }, + "gross_savings": gross_savings, + "net_savings": net_savings, + # Backward-compatible key. For rows without governance overhead it is numerically + # identical to v0.2. For new evidence it deliberately points to the net result. + "savings": net_savings, + "governance": { + "tokens": marginal_total["governance_tokens"], + "usd": marginal_total["governance_usd"], + "latency_ms": marginal_total["governance_latency_ms"], }, "quality": { "resolved_delta_pp": delta_pp, "non_inferiority_margin_pp": quality_margin_pp, - "preserved_within_margin": delta_pp >= -quality_margin_pp, + "preserved_within_margin": quality_preserved, "preserved_within_one_pp": delta_pp >= -1.0, "regressions": sum( baseline[item].resolved and not marginal[item].resolved for item in ids @@ -191,6 +345,17 @@ def efficiency(total: dict[str, Any]) -> dict[str, float | None]: "recoveries": sum( not baseline[item].resolved and marginal[item].resolved for item in ids ), + "reviewed_stops": marginal_total["reviewed_stops"], + "false_stops": marginal_total["false_stops"], + "false_stop_rate": false_stop_rate, + "max_false_stop_rate": max_false_stop_rate, + "false_stops_acceptable": false_stops_acceptable, + }, + "intervention": { + "status": intervention_status, + "minimum_net_token_savings_percent": minimum_net_token_savings_percent, + "net_positive": float(cast(float, net_savings["tokens_percent"])) > 0.0, + "graceful_irrelevance": intervention_status == "pass_through", }, } @@ -203,10 +368,15 @@ def _format_optional_usd(value: float | None) -> str: return "n/a" if value is None else f"${value:.6f}" +def _format_optional_rate(value: float | None) -> str: + return "n/a" if value is None else f"{value * 100.0:.2f}%" + + def render_public_report(result: dict[str, Any]) -> str: baseline = result["baseline"] marginal = result["marginal"] savings = result["savings"] + gross = result.get("gross_savings", savings) quality = result["quality"] efficiency = result["efficiency"] baseline_efficiency = efficiency["baseline"] @@ -214,11 +384,14 @@ def render_public_report(result: dict[str, Any]) -> str: ci = savings["tokens_confidence_interval"] confidence_percent = float(savings["confidence_level"]) * 100.0 margin = float(quality["non_inferiority_margin_pp"]) + intervention = result.get("intervention", {"status": "unclassified"}) + governance = result.get("governance", {"tokens": 0, "usd": 0.0, "latency_ms": 0}) return "\n".join( [ "# Measured public benchmark comparison", "", "This report compares matched executions. It does not estimate or impute missing runs.", + "MARGINAL overhead is counted in net efficiency and net savings.", "", "| Metric | Baseline | MARGINAL | Change |", "|---|---:|---:|---:|", @@ -228,22 +401,30 @@ def render_public_report(result: dict[str, Any]) -> str: f"{quality['resolved_delta_pp']:+.2f} pp |" ), ( - f"| Tokens | {baseline['tokens']:,} | " - f"{marginal['tokens']:,} | {savings['tokens_percent']:.2f}% fewer |" + f"| Agent tokens | {baseline['tokens']:,} | " + f"{marginal['tokens']:,} | {gross['tokens_percent']:.2f}% fewer |" ), ( - f"| USD | ${baseline['usd']:.4f} | " - f"${marginal['usd']:.4f} | {savings['usd_percent']:.2f}% lower |" + f"| Effective tokens (incl. governance) | {baseline['effective_tokens']:,} | " + f"{marginal['effective_tokens']:,} | {savings['tokens_percent']:.2f}% fewer |" ), ( - f"| Latency | {baseline['latency_ms']:,} ms | " - f"{marginal['latency_ms']:,} ms | " + f"| Effective USD | ${baseline['effective_usd']:.4f} | " + f"${marginal['effective_usd']:.4f} | {savings['usd_percent']:.2f}% lower |" + ), + ( + f"| Effective latency | {baseline['effective_latency_ms']:,} ms | " + f"{marginal['effective_latency_ms']:,} ms | " f"{savings['latency_percent']:.2f}% lower |" ), ( f"| Tool calls | {baseline['tool_calls']} | " - f"{marginal['tool_calls']} | " - f"{savings['tool_calls_percent']:.2f}% fewer |" + f"{marginal['tool_calls']} | {savings['tool_calls_percent']:.2f}% fewer |" + ), + ( + f"| Repeated calls | {baseline['repeated_calls']} | " + f"{marginal['repeated_calls']} | " + f"{savings['repeated_calls_percent']:.2f}% fewer |" ), ( "| Tokens per resolved task | " @@ -256,8 +437,21 @@ def render_public_report(result: dict[str, Any]) -> str: f"{_format_optional_usd(marginal_efficiency['usd_per_resolved'])} | — |" ), "", + "## Governance tax", + "", + ( + f"MARGINAL overhead: **{governance['tokens']:,} tokens**, " + f"**${governance['usd']:.6f}**, **{governance['latency_ms']:,} ms**." + ), + ( + f"Gross agent-token savings: **{gross['tokens_percent']:.2f}%**. " + f"Net token savings after governance: **{savings['tokens_percent']:.2f}%**." + ), + "", + "## Quality and intervention decision", + "", ( - f"Token savings {confidence_percent:.1f}% bootstrap interval: " + f"Net token savings {confidence_percent:.1f}% bootstrap interval: " f"**{ci[0]:.2f}% to {ci[1]:.2f}%**." ), ( @@ -265,6 +459,17 @@ def render_public_report(result: dict[str, Any]) -> str: f"**{quality['preserved_within_margin']}**." ), f"Regressions: **{quality['regressions']}**. Recoveries: **{quality['recoveries']}**.", + ( + f"Reviewed deny recommendations: **{quality.get('reviewed_stops', 0)}**. " + f"False stops: **{quality.get('false_stops', 0)}** " + f"({_format_optional_rate(quality.get('false_stop_rate'))})." + ), + f"Intervention status: **{intervention['status']}**.", + "", + ( + "`pass_through` is a valid result: it means MARGINAL did not demonstrate " + "enough net value to justify intervention under the preregistered threshold." + ), "", ] ) diff --git a/src/marginal/treasury.py b/src/marginal/treasury.py index f120c30..bf948a0 100644 --- a/src/marginal/treasury.py +++ b/src/marginal/treasury.py @@ -3,11 +3,13 @@ from __future__ import annotations import threading -from collections.abc import Iterable +import time +from collections.abc import Callable, Iterable from dataclasses import asdict, replace from typing import Any from .budget import BudgetLedger, BudgetLimits, BudgetOverrun, BudgetUsage +from .controls import GovernanceTracker from .fingerprint import fingerprint_action from .models import Action, Allocation, Cost, Decision from .modes import ExecutionMode @@ -21,7 +23,11 @@ class AuthorizationRequired(RuntimeError): class Treasury: - """Allocate and account for agent compute as a scarce resource.""" + """Allocate and account for agent compute as a scarce resource. + + Governance overhead is measured separately from agent workload cost. Parent/child + treasuries share one tracker so the reported governance tax covers the full treasury tree. + """ def __init__( self, @@ -32,6 +38,8 @@ def __init__( name: str = "root", parent: Treasury | None = None, mode: ExecutionMode | str = ExecutionMode.ENFORCE, + governance_tracker: GovernanceTracker | None = None, + clock: Callable[[], int] | None = None, ) -> None: self.name = name self.ledger = BudgetLedger(limits) @@ -45,6 +53,10 @@ def __init__( self._pending_semantics: dict[str, list[str]] = ( parent._pending_semantics if parent is not None else {} ) + self._clock: Callable[[], int] = clock or time.perf_counter_ns + self.governance: GovernanceTracker = ( + parent.governance if parent is not None else governance_tracker or GovernanceTracker() + ) if parent is None: self._reservation_counter = 0 self._approved_count = 0 @@ -55,6 +67,8 @@ def __init__( self._failed_settled_count = 0 self._outcome_count = 0 self._observation_count = 0 + self._recommended_denials: set[str] = set() + self._reviewed_denials: set[str] = set() @property def usage(self) -> BudgetUsage: @@ -68,12 +82,12 @@ def propose(self, action: Action) -> Decision: return self.authorize(action) def evaluate(self, action: Action) -> Decision: - """Evaluate an action without reserving resources or mutating counters.""" + """Evaluate an action without reserving resources or mutating decision counters.""" prepared = self._prepare(action) assert prepared.fingerprint is not None with self._lock: - return self._recommended_decision(prepared) + return self._timed_recommendation(prepared) def is_authorized(self, action: Action) -> bool: prepared = self._prepare(action) @@ -95,7 +109,7 @@ def fund_best(self, actions: Iterable[Action]) -> Allocation | None: for action in actions: prepared = self._prepare(action) assert prepared.fingerprint is not None - decision = self._recommended_decision(prepared) + decision = self._timed_recommendation(prepared) evaluated.append( { "action": action_payload(prepared), @@ -134,7 +148,9 @@ def authorize(self, action: Action, *, apply_mode: bool = True) -> Decision: prepared = self._prepare(action) assert prepared.fingerprint is not None with self._lock: - recommended = self._recommended_decision(prepared) + recommended = self._timed_recommendation(prepared) + if not recommended.allowed: + self._recommended_denials.add(prepared.fingerprint) decision = self._apply_mode(recommended) if apply_mode else recommended reservation_action: Action | None = None @@ -160,6 +176,7 @@ def authorize(self, action: Action, *, apply_mode: bool = True) -> Decision: "decision": decision_payload(decision), "usage": usage_payload(self.usage), "reserved": usage_payload(self.ledger.reserved_usage), + "governance": self.governance.summary(), } ) except Exception: @@ -222,7 +239,11 @@ def _settle( violations.append(decision.reason) if not failed: - self.policy.mark_executed(prepared.fingerprint) + observe_execution = getattr(self.policy, "observe_execution", None) + if callable(observe_execution): + observe_execution(prepared) + else: + self.policy.mark_executed(prepared.fingerprint) self._unregister_pending(prepared.fingerprint, reservation_fingerprint) self._committed_count += 1 if failed: @@ -238,6 +259,7 @@ def _settle( "usage": usage_payload(self.usage), "budget_overrun": bool(violations), "violations": sorted(set(violations)), + "governance": self.governance.summary(), } if failed: event["reason"] = failure_reason @@ -271,6 +293,51 @@ def abort(self, action: Action, *, reason: str = "execution aborted") -> None: } ) + def record_governance_overhead( + self, + *, + tokens: int = 0, + usd: float = 0.0, + latency_ms: float = 0.0, + ) -> None: + """Record overhead produced outside the local policy, such as an adapter-side model call.""" + + with self._lock: + self.governance.record_external_overhead( + tokens=tokens, + usd=usd, + latency_ms=latency_ms, + ) + + def record_stop_review(self, action: Action, *, would_have_helped: bool) -> None: + """Attach an explicit counterfactual label to a prior deny recommendation. + + This method intentionally refuses to infer false stops from task success and rejects + duplicate labels for the same semantic action fingerprint. + """ + + prepared = self._prepare(action) + assert prepared.fingerprint is not None + with self._lock: + fingerprint = prepared.fingerprint + if fingerprint not in self._recommended_denials: + raise ValueError("action was not previously recommended for denial") + if fingerprint in self._reviewed_denials: + raise ValueError("deny recommendation has already been reviewed") + self.governance.record_stop_review(would_have_helped=would_have_helped) + self._reviewed_denials.add(fingerprint) + self.trace_sink.emit( + { + **self._identity_payload(), + "event": "counterfactual_review", + "treasury": self.name, + "action": action_payload(prepared), + "would_have_helped": would_have_helped, + "false_stop": would_have_helped, + "governance": self.governance.summary(), + } + ) + def observe_value(self, action: Action, realized_gain: float) -> None: """Record explicit action-level realized gain for the configured estimator.""" @@ -337,8 +404,17 @@ def summary(self) -> dict[str, Any]: "limits": asdict(self.limits), "policy": self.policy.identity.to_dict(), "estimator": self.policy.estimator_identity.to_dict(), + "governance": self.governance.summary(), } + def _timed_recommendation(self, prepared: Action) -> Decision: + started = self._clock() + try: + return self._recommended_decision(prepared) + finally: + latency_ms = (self._clock() - started) / 1_000_000 + self.governance.record_decision(latency_ms=latency_ms) + def _recommended_decision(self, prepared: Action) -> Decision: assert prepared.fingerprint is not None if self._pending_semantics.get(prepared.fingerprint): diff --git a/tests/controls/test_diminishing.py b/tests/controls/test_diminishing.py new file mode 100644 index 0000000..e6ff0d4 --- /dev/null +++ b/tests/controls/test_diminishing.py @@ -0,0 +1,73 @@ +from __future__ import annotations + +from marginal import Action, Cost, DiminishingReturnConfig, DiminishingReturnDetector + + +def _action(*, state: str, evidence: str = "", fingerprint: str = "a") -> Action: + return Action( + name="verify the same file", + kind="verification", + cost=Cost(tokens=100), + expected_gain=0.4, + fingerprint=fingerprint, + metadata={ + "phase": "verify", + "state_hash": state, + "evidence_hash": evidence, + "marginal_semantic_key": "verify:file:README.md", + }, + ) + + +def test_diminishing_returns_discount_same_state_then_stop() -> None: + detector = DiminishingReturnDetector( + DiminishingReturnConfig(gain_decay=0.5, max_same_state_repeats=2) + ) + + first = detector.evaluate(_action(state="s1")) + assert first.gain_multiplier == 1.0 + assert first.should_stop is False + detector.observe(_action(state="s1")) + + second = detector.evaluate(_action(state="s1", fingerprint="b")) + assert second.same_state_repeats == 1 + assert second.gain_multiplier == 0.5 + assert second.should_stop is False + detector.observe(_action(state="s1", fingerprint="b")) + + third = detector.evaluate(_action(state="s1", fingerprint="c")) + assert third.same_state_repeats == 2 + assert third.gain_multiplier == 0.25 + assert third.should_stop is True + assert third.reason_code == "DIMINISHING_RETURN_REJECTED" + + +def test_new_state_resets_repetition_pressure() -> None: + detector = DiminishingReturnDetector() + detector.observe(_action(state="s1")) + + signal = detector.evaluate(_action(state="s2", fingerprint="b")) + + assert signal.same_state_repeats == 0 + assert signal.gain_multiplier == 1.0 + assert signal.should_stop is False + + +def test_new_evidence_resets_repetition_pressure() -> None: + detector = DiminishingReturnDetector() + detector.observe(_action(state="s1", evidence="e1")) + + signal = detector.evaluate(_action(state="s1", evidence="e2", fingerprint="b")) + + assert signal.same_state_repeats == 0 + assert signal.gain_multiplier == 1.0 + + +def test_missing_state_fails_open() -> None: + detector = DiminishingReturnDetector() + + signal = detector.evaluate(_action(state="")) + + assert signal.should_stop is False + assert signal.gain_multiplier == 1.0 + assert signal.reason_code == "DIMINISHING_RETURN_UNOBSERVABLE" diff --git a/tests/controls/test_governance.py b/tests/controls/test_governance.py new file mode 100644 index 0000000..da6ad09 --- /dev/null +++ b/tests/controls/test_governance.py @@ -0,0 +1,36 @@ +from __future__ import annotations + +import pytest + +from marginal import GovernanceTracker + + +def test_governance_tracker_separates_self_cost_from_agent_cost() -> None: + tracker = GovernanceTracker() + tracker.record_decision(latency_ms=1.25) + tracker.record_external_overhead(tokens=120, usd=0.002, latency_ms=50) + + summary = tracker.summary() + + assert summary["decisions"] == 1 + assert summary["external_tokens"] == 120 + assert summary["external_usd"] == 0.002 + assert summary["total_latency_ms"] == 51.25 + + +def test_false_stop_rate_requires_explicit_reviews() -> None: + tracker = GovernanceTracker() + assert tracker.summary()["false_stop_rate"] is None + + tracker.record_stop_review(would_have_helped=False) + tracker.record_stop_review(would_have_helped=True) + + assert tracker.summary()["reviewed_stops"] == 2 + assert tracker.summary()["false_stops"] == 1 + assert tracker.summary()["false_stop_rate"] == 0.5 + + +def test_governance_tracker_rejects_invalid_overhead() -> None: + tracker = GovernanceTracker() + with pytest.raises(ValueError): + tracker.record_external_overhead(tokens=-1) diff --git a/tests/controls/test_policy_diminishing.py b/tests/controls/test_policy_diminishing.py new file mode 100644 index 0000000..4d24a5c --- /dev/null +++ b/tests/controls/test_policy_diminishing.py @@ -0,0 +1,48 @@ +from __future__ import annotations + +from marginal import ( + Action, + BudgetLedger, + BudgetLimits, + Cost, + DiminishingReturnConfig, + DiminishingReturnDetector, + MarginalPolicy, +) + + +def _retry(number: int) -> Action: + return Action( + name="verify README", + kind="verification", + cost=Cost(tokens=100), + expected_gain=0.4, + fingerprint=f"retry-{number}", + metadata={ + "phase": "verify", + "state_hash": "workspace-unchanged", + "marginal_semantic_key": "verify:file:README.md", + }, + ) + + +def test_policy_can_discount_and_reject_repeated_same_state_work() -> None: + policy = MarginalPolicy( + diminishing_detector=DiminishingReturnDetector( + DiminishingReturnConfig(gain_decay=0.5, max_same_state_repeats=2) + ) + ) + ledger = BudgetLedger(BudgetLimits(max_tokens=10_000)) + + first = policy.evaluate(_retry(1), ledger) + assert first.allowed is True + policy.observe_execution(_retry(1)) + + second = policy.evaluate(_retry(2), ledger) + assert second.allowed is True + assert second.expected_gain == 0.2 + policy.observe_execution(_retry(2)) + + third = policy.evaluate(_retry(3), ledger) + assert third.allowed is False + assert third.reason_code == "DIMINISHING_RETURN_REJECTED" diff --git a/tests/controls/test_treasury_governance.py b/tests/controls/test_treasury_governance.py new file mode 100644 index 0000000..eb79082 --- /dev/null +++ b/tests/controls/test_treasury_governance.py @@ -0,0 +1,44 @@ +from __future__ import annotations + +import pytest + +from marginal import Action, BudgetLimits, Cost, Treasury + + +def test_shadow_mode_can_review_a_false_stop_without_inferring_it() -> None: + treasury = Treasury( + BudgetLimits(max_tokens=10_000, max_usd=10.0), + mode="shadow", + ) + action = Action( + name="expensive review", + kind="review", + cost=Cost(tokens=100, usd=1.0), + expected_gain=0.0, + fingerprint="review-1", + ) + + decision = treasury.authorize(action) + assert decision.allowed is True + assert decision.recommended is False + treasury.commit(action) + treasury.record_stop_review(action, would_have_helped=True) + + governance = treasury.summary()["governance"] + assert governance["reviewed_stops"] == 1 + assert governance["false_stops"] == 1 + assert governance["false_stop_rate"] == 1.0 + + +def test_stop_review_rejects_actions_that_were_not_denied() -> None: + treasury = Treasury(BudgetLimits(max_tokens=10_000), mode="shadow") + action = Action( + name="free useful action", + kind="verification", + expected_gain=0.5, + fingerprint="useful-1", + ) + treasury.authorize(action) + + with pytest.raises(ValueError, match="not previously recommended"): + treasury.record_stop_review(action, would_have_helped=False) diff --git a/tests/evaluation/test_cli_public_eval_governance.py b/tests/evaluation/test_cli_public_eval_governance.py new file mode 100644 index 0000000..6fe79c4 --- /dev/null +++ b/tests/evaluation/test_cli_public_eval_governance.py @@ -0,0 +1,43 @@ +from __future__ import annotations + +import json +from pathlib import Path + +from marginal.cli import main + + +def _write(path: Path, row: dict[str, object]) -> None: + path.write_text(json.dumps(row) + "\n", encoding="utf-8") + + +def test_public_eval_cli_exposes_net_value_gates(tmp_path: Path, capsys) -> None: + baseline = tmp_path / "baseline.jsonl" + marginal = tmp_path / "marginal.jsonl" + _write(baseline, {"instance_id": "task", "resolved": True, "tokens": 1000}) + _write( + marginal, + { + "instance_id": "task", + "resolved": True, + "tokens": 850, + "governance_tokens": 100, + }, + ) + + exit_code = main( + [ + "public-eval", + str(baseline), + str(marginal), + "--bootstrap-samples", + "20", + "--minimum-net-token-savings-percent", + "10", + "--json", + ] + ) + + assert exit_code == 0 + report = json.loads(capsys.readouterr().out) + assert report["net_savings"]["tokens_percent"] == 5.0 + assert report["intervention"]["status"] == "pass_through" diff --git a/tests/evaluation/test_public_eval_governance.py b/tests/evaluation/test_public_eval_governance.py new file mode 100644 index 0000000..5e6775b --- /dev/null +++ b/tests/evaluation/test_public_eval_governance.py @@ -0,0 +1,63 @@ +from __future__ import annotations + +from marginal.public_eval import RunRecord, compare_runs, render_public_report + + +def test_public_eval_reports_gross_and_net_savings() -> None: + baseline = {"task": RunRecord(instance_id="task", resolved=True, tokens=1000, usd=1.0)} + marginal = { + "task": RunRecord( + instance_id="task", + resolved=True, + tokens=700, + usd=0.7, + governance_tokens=200, + governance_usd=0.2, + repeated_calls=1, + ) + } + + result = compare_runs(baseline, marginal, bootstrap_samples=20) + + assert result["gross_savings"]["tokens_percent"] == 30.0 + assert result["net_savings"]["tokens_percent"] == 10.0 + assert result["savings"]["tokens_percent"] == 10.0 + assert result["intervention"]["status"] == "supported" + assert "Governance tax" in render_public_report(result) + + +def test_governance_tax_can_make_pass_through_the_correct_result() -> None: + baseline = {"task": RunRecord(instance_id="task", resolved=True, tokens=1000)} + marginal = { + "task": RunRecord( + instance_id="task", + resolved=True, + tokens=850, + governance_tokens=200, + ) + } + + result = compare_runs(baseline, marginal, bootstrap_samples=20) + + assert result["gross_savings"]["tokens_percent"] == 15.0 + assert result["net_savings"]["tokens_percent"] == -5.0 + assert result["intervention"]["status"] == "pass_through" + assert result["intervention"]["graceful_irrelevance"] is True + + +def test_reviewed_false_stop_can_fail_the_intervention_gate() -> None: + baseline = {"task": RunRecord(instance_id="task", resolved=True, tokens=1000)} + marginal = { + "task": RunRecord( + instance_id="task", + resolved=True, + tokens=500, + reviewed_stops=1, + false_stops=1, + ) + } + + result = compare_runs(baseline, marginal, bootstrap_samples=20) + + assert result["quality"]["false_stop_rate"] == 1.0 + assert result["intervention"]["status"] == "false_stop_risk"