diff --git a/.agents/plugins/marketplace.json b/.agents/plugins/marketplace.json new file mode 100644 index 0000000..31236e0 --- /dev/null +++ b/.agents/plugins/marketplace.json @@ -0,0 +1,20 @@ +{ + "name": "marginal", + "interface": { + "displayName": "Marginal" + }, + "plugins": [ + { + "name": "marginal", + "source": { + "source": "local", + "path": "./plugins/marginal" + }, + "policy": { + "installation": "AVAILABLE", + "authentication": "ON_INSTALL" + }, + "category": "Productivity" + } + ] +} diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 31e39eb..9f1cc5d 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -10,7 +10,7 @@ body: id: version attributes: label: MARGINAL version - placeholder: "0.2.0" + placeholder: "0.3.0" validations: required: true - type: input diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4a505b2..ff2db05 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -27,5 +27,6 @@ jobs: - run: ruff check . - run: mypy src/marginal - run: pytest -q + - run: python scripts/build_codex_plugin.py --check - run: python -m build - run: python -m twine check dist/* diff --git a/.gitignore b/.gitignore index 388b40c..53143a1 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ htmlcov/ dist/ build/ .venv/ +.worktrees/ .env *.jsonl .DS_Store diff --git a/CHANGELOG.md b/CHANGELOG.md index b3c5d67..70793d8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,8 @@ All notable changes to MARGINAL are documented here. The project follows Semanti ## [Unreleased] +## [0.3.0] - 2026-08-13 + ### Added - opt-in, provider-neutral `DiminishingReturnDetector` with same-state/evidence-aware gain decay; @@ -13,6 +15,13 @@ All notable changes to MARGINAL are documented here. The project follows Semanti - gross-versus-net savings and intervention status including Graceful Irrelevance through `pass_through`; - governance evidence standard, Codex benchmark-readiness guide and Community Feedback Log; - structured documentation information architecture by user intent. +- native Codex plugin marketplace `marginal@marginal` with reproducible dependency-free runtime; +- one-command native install/remove plus `status`, `doctor`, `review`, `promote`, and `demote`; +- strict Codex lifecycle contracts, privacy-safe normalization, Git state hashing, and conservative structured outcome classification; +- authenticated per-session loopback service with bounded messages and fail-open demotion; +- provider-neutral No Progress evidence control and versioned Earned Enforcement promotion receipts; +- isolated Codex 0.147.0 marketplace/lifecycle/privacy/removal smoke and universal directory review packet; +- public privacy, terms, support, Codex integration, and submission documentation. ### Changed @@ -22,13 +31,16 @@ All notable changes to MARGINAL are documented here. The project follows Semanti - website and README now lead with a concrete illustrative trace and proof standard before architecture theory; - roadmap now treats governance tax, false-stop rate, matched OFF/ON evaluation and pass-through as first-class success criteria; - the 10-task Codex canary is explicitly classified as integration validation rather than public performance evidence. +- website and README now lead with native Codex install/remove and the measured n=3 `pass_through` result. ### Scientific limitations - diminishing-return thresholds are transparent heuristics until calibrated on representative engine telemetry; - false stops require external review/counterfactual labels and are not automatically causal estimates; - Graceful Irrelevance classifies the measured configuration, not the universal usefulness of MARGINAL; -- vendor-specific Codex integration and measured public savings remain future v0.3 evidence. +- the Codex plugin supports local Tool Enforcement paths, not Full Compute Enforcement; +- the n=3 result remains integration telemetry and does not establish general token savings; +- universal directory availability depends on external review and release. ## [0.2.0] - 2026-08-06 diff --git a/CITATION.cff b/CITATION.cff index 47c7489..15f9291 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -4,7 +4,7 @@ title: "MARGINAL: Economically Disciplined Compute Allocation for AI Agents" type: software authors: - name: SignalLayer Labs -version: 0.2.0 +version: 0.3.0 date-released: 2026-08-06 license: Apache-2.0 repository-code: "https://github.com/SignalLayerLabs/Marginal" diff --git a/PRIVACY.md b/PRIVACY.md new file mode 100644 index 0000000..13e9b87 --- /dev/null +++ b/PRIVACY.md @@ -0,0 +1,35 @@ +# MARGINAL Privacy Notice + +**Effective date:** 2026-08-13 + +MARGINAL is local-first open-source software. The Codex plugin makes no network request and does +not operate a SignalLayer Labs telemetry service. + +## Data processed locally + +Codex supplies lifecycle identifiers, tool names, tool inputs, tool responses, workspace paths, +and session metadata to local hooks. MARGINAL uses that input in memory to make a decision and to +derive hashes. By default it does not persist prompts, source code, raw commands, raw tool output, +transcripts, authentication files, or credential environment values. + +The plugin may store redacted decisions, opaque hashes, aggregate coverage counts, outcome status, +reason codes, latency, review labels, promotion receipts, and user-private connection files under +Codex `PLUGIN_DATA`. Connection credentials are removed at session end. Local evidence remains +until the user deletes it or runs an explicit purge. + +## Sharing and remote processing + +MARGINAL does not transmit plugin evidence to SignalLayer Labs. GitHub, Codex, package registries, +and any model provider remain governed by their own policies. Exporting a ledger or attaching files +to an issue is an explicit user action; inspect exports before sharing them. + +## User controls + +- `marginal codex status` shows the local mode. +- `marginal codex demote` returns enforcement to Shadow Mode. +- `marginal uninstall codex` removes the plugin and preserves evidence. +- `marginal uninstall codex --purge-data --yes` removes plugin data explicitly. + +Security issues must follow [SECURITY.md](SECURITY.md). Privacy questions can be filed through the +private contact route described in [SUPPORT.md](SUPPORT.md). + diff --git a/README.md b/README.md index e7da414..9c24648 100644 --- a/README.md +++ b/README.md @@ -31,6 +31,26 @@ Open source · Local first · Provider neutral · Zero mandatory runtime depende > **Exploratory 3-task smoke, one paired run per task.** This validates the integration; it is not a general performance claim. +### Install the native Codex plugin + +MARGINAL installs through Codex's native plugin marketplace and starts globally in **Shadow Mode**: + +```bash +codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal +``` + +Remove it cleanly with: + +```bash +codex plugin remove marginal@marginal +``` + +The plugin provides **Tool Enforcement**, not Full Compute Enforcement. Repository blocking is +disabled until local **Earned Enforcement** evidence proves at least 99% hook coverage, reviewed +stop candidates, zero false stops, no pending failures, and bounded governance latency. Any drift +demotes the repository to Shadow Mode and requires a fresh clean evidence window. The public directory submission packet is ready, but the +directory listing remains subject to OpenAI review; the Git marketplace command above works now. + | Metric | Codex OFF | Codex + MARGINAL | Observed change | |---|---:|---:|---:| | SWE-bench resolved | 0/3 | 0/3 | **0/3 → 0/3** | @@ -185,10 +205,37 @@ Read the [benchmark protocol](docs/evaluation/public-benchmarks.md) and [governa ## Install -Current v0.2 install target: +### Codex — recommended + +```bash +codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal +``` + +Then open `/hooks` in Codex, review the exact commands, and grant trust only after inspection. +MARGINAL never bypasses the hook trust boundary. Useful management commands: + +```bash +marginal codex status +marginal codex doctor +marginal codex review +marginal codex review --candidate ACTION_HASH --verdict waste +marginal codex promote +marginal codex demote +marginal uninstall codex +``` + +The Python package can perform the same native installation transaction: + +```bash +marginal install codex +``` + +### Python library + +Current tagged library install target: ```bash -pip install "marginal-ai @ git+https://github.com/SignalLayerLabs/Marginal.git@v0.2.0" +pip install "marginal-ai @ git+https://github.com/SignalLayerLabs/Marginal.git@v0.3.0" ``` Development checkout: @@ -199,7 +246,9 @@ cd Marginal python -m pip install -e ".[dev]" ``` -The auditable Codex reference adapter and its first matched smoke are now available in `benchmark/codex_adapter/`. Start from the frozen protocol and treat the current n=3 result as integration evidence, not a performance claim. +The production Codex adapter lives under `src/marginal/integrations/codex/`; the independent +benchmark harness remains under `benchmark/codex_adapter/`. Treat the current n=3 result as +integration evidence, not a performance claim. ## Quickstart @@ -255,7 +304,10 @@ The engine-specific adapter owns native interception and telemetry. The core own ## Project status -`v0.2.0` provides the Learning Loop Foundation, privacy profiles, Universal Agent Protocol, versioned evidence and replay. The community-hardening work prepares the core evidence model for **v0.3 — Codex Reference Integration**. +The v0.3 candidate adds the native Codex plugin, privacy-safe hook contracts, an authenticated +local service, reversible install/uninstall, and Earned Enforcement receipts to the v0.2 Learning +Loop Foundation. The universal directory submission is an external review step and is not described +as live until OpenAI accepts and releases it. The next milestone must answer a falsifiable question: @@ -271,7 +323,7 @@ If the answer is no, the result should be published as no demonstrated benefit f |---|---| | Getting started | [Quickstart](docs/getting-started/quickstart.md) | | Product model | [Concepts](docs/product/concepts.md) · [Architecture](docs/product/architecture.md) | -| Integrations | [Integration overview](docs/integrations/overview.md) · [Codex benchmark readiness](docs/integrations/codex-benchmark-readiness.md) | +| Integrations | [Codex plugin](docs/integrations/codex.md) · [Integration overview](docs/integrations/overview.md) · [Codex benchmark readiness](docs/integrations/codex-benchmark-readiness.md) | | Evaluation | [Benchmarking](docs/evaluation/benchmarking.md) · [Public benchmarks](docs/evaluation/public-benchmarks.md) · [Governance evidence](docs/evaluation/governance-evidence.md) | | Reference | [API](docs/reference/api.md) | | Operations | [Privacy](docs/operations/privacy.md) · [Website](docs/operations/website.md) | diff --git a/ROADMAP.md b/ROADMAP.md index e6bb0d7..7e6acba 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -38,9 +38,9 @@ This roadmap is milestone-driven rather than date-driven. GitHub Issues and pull | Milestone | Status | Primary outcome | |---|---|---| | **v0.1 — Reference Allocator Foundation** | Complete | Provider-neutral allocation, accounting, tracing and first release | -| **v0.2 — Learning Loop Foundation** | Validation | Universal protocol, non-blocking observation, versioned evidence, privacy and replay | +| **v0.2 — Learning Loop Foundation** | Complete | Universal protocol, non-blocking observation, versioned evidence, privacy and replay | | **Community hardening** | In progress | Governance tax, false-stop accounting, diminishing-return control and clearer evidence UX | -| **v0.3 — Codex Reference Integration** | Planned | One-command target, real telemetry and first matched public benchmark | +| **v0.3 — Codex Reference Integration** | Validation | Native plugin, one-command install, Earned Enforcement, and measured smoke | | **v0.4 — Multi-Engine Developer Preview** | Planned | Shared core across materially different coding agents | | **v0.5 — One-Command Universal Installation** | Planned | Detection, installation, diagnostics and rollback across engines | | **v0.6 — Adaptive and Causal Allocation** | Planned | Calibrated learning, exploration and stronger identification strategies | @@ -60,7 +60,7 @@ Delivered provider-neutral `Action`, `Cost`, `Decision` and `Allocation` primiti ## v0.2 — Learning Loop Foundation -**Status:** Validation +**Status:** Complete The v0.2 release candidate adds: @@ -77,10 +77,10 @@ The v0.2 release candidate adds: - task outcomes separated from action-level realized gain; - non-causal replay and ledger/reporting CLI support. -### Remaining exit criteria +### Exit criteria -- [ ] Ruff, mypy strict, full tests, package build and Twine validation pass in canonical CI. -- [ ] `v0.2.0` is tagged/released from the canonical repository. +- [x] Ruff, mypy strict, full tests, package build and Twine validation pass in canonical CI. +- [x] `v0.2.0` is tagged/released from the canonical repository. Vendor-specific adapters and measured production savings are intentionally outside v0.2. @@ -120,28 +120,30 @@ Vendor-specific adapters and measured production savings are intentionally outsi ## v0.3 — Codex Reference Integration -**Status:** Planned +**Status:** Validation **Objective:** integrate MARGINAL into Codex and produce the first real matched benchmark with measured telemetry and net-value accounting. ### Integration deliverables -- [ ] Build a thin Codex adapter against the Universal Agent Protocol. -- [ ] Target `marginal install codex` with safe backup, Shadow Mode default and clean uninstall. -- [ ] Detect Codex version/capability level and refuse unsupported enforcement claims. -- [ ] Capture measured input, cached input, output, reasoning and total tokens. -- [ ] Correlate model/tool/retry/verification actions with session, task and workspace state. -- [ ] Record evidence hashes where deterministic evidence boundaries exist. -- [ ] Capture governance tokens, USD and latency separately from workload usage. -- [ ] Define and record repeated-call metrics consistently in OFF and ON arms. -- [ ] Export raw paired JSONL sufficient to reproduce the public report. +- [x] Build a thin Codex adapter against the Universal Agent Protocol. +- [x] Ship native `marginal@marginal` installation plus `marginal install codex`, Shadow Mode default and clean uninstall. +- [x] Detect Codex version/capability level and refuse unsupported enforcement claims. +- [x] Capture measured input, cached input, output, reasoning and total tokens in the benchmark adapter. +- [x] Correlate tool and verification actions with session, turn, call, task and workspace state. +- [x] Record evidence hashes where deterministic evidence boundaries exist without persisting raw payloads. +- [x] Capture governance tokens, USD and latency separately from workload usage. +- [x] Define and record repeated-call metrics consistently in OFF and ON arms. +- [x] Export raw paired JSONL sufficient to reproduce the public report. +- [x] Add Earned Enforcement receipts with explicit promotion and automatic fail-open demotion. +- [x] Validate add/install/four-hook lifecycle/privacy/remove in an isolated Codex home. ### Canary: engineering validation only - [ ] Run a 10-task matched canary with identical model, prompt, tools, limits and verifier. -- [ ] Confirm event/session/state correlation and no orphaned reservations. -- [ ] Confirm telemetry is measured rather than declared. -- [ ] Confirm governance overhead is separately accounted. +- [x] Confirm event/session/state correlation and no orphaned reservations in focused lifecycle tests. +- [x] Confirm telemetry is measured rather than declared in the exploratory paired smoke. +- [x] Confirm governance overhead is separately accounted. - [ ] Review deny recommendations for false-stop candidates. - [ ] Preserve pass-through and negative results instead of filtering them out. @@ -179,12 +181,14 @@ Report: ### v0.3 exit criteria -- Codex baseline and Codex + MARGINAL run under matched conditions. -- Telemetry comes from the runtime/provider integration rather than declared demo estimates. -- The canary completes without integration failures. -- Public results are reproducible from raw paired artifacts. -- Headline claims use **net** metrics after governance tax. -- If the preregistered gate is not met, the published conclusion says so. +- [x] Codex baseline and Codex + MARGINAL run under matched conditions for the n=3 integration smoke. +- [x] Telemetry comes from the runtime/provider integration rather than declared demo estimates. +- [x] The authoritative Docker verifier completes without infrastructure errors. +- [x] Public results are reproducible from raw paired artifacts. +- [x] Headline claims use **net** metrics after governance tax. +- [x] The published conclusion says `pass_through` because the support gate was not met. +- [ ] A preregistered repeated run large enough for a general efficiency claim is complete. +- [ ] The external universal directory review is accepted and released. See [Codex benchmark readiness](docs/integrations/codex-benchmark-readiness.md). diff --git a/SUPPORT.md b/SUPPORT.md index 35d9e7d..058d346 100644 --- a/SUPPORT.md +++ b/SUPPORT.md @@ -8,3 +8,12 @@ a public issue. MARGINAL is an early open-source reference implementation. Community support is best effort; no service-level agreement is provided. + +For Codex integration reports, include the redacted output of `marginal codex doctor`, the Codex +version, operating system, plugin version, and whether `/hooks` shows the expected lifecycle hooks. +Never attach `auth.json`, prompts, source code, raw commands, raw tool output, transcripts, access +tokens, or the contents of `PLUGIN_DATA` connection files. + +Installation and removal guidance is maintained in [docs/integrations/codex.md](docs/integrations/codex.md). +Privacy questions that cannot be discussed publicly may use GitHub's private vulnerability +reporting channel; choose the privacy category and do not include unrelated credentials. diff --git a/TERMS.md b/TERMS.md new file mode 100644 index 0000000..8a169c0 --- /dev/null +++ b/TERMS.md @@ -0,0 +1,24 @@ +# MARGINAL Terms of Use + +**Effective date:** 2026-08-13 + +MARGINAL is provided under the [Apache License 2.0](LICENSE). These terms clarify the public plugin +experience and do not replace the license. + +MARGINAL is experimental developer infrastructure. It is provided without a service-level +agreement or guarantee of token savings, cost reduction, task success, uninterrupted operation, +or suitability for a particular purpose. Shadow Mode is the default. Tool Enforcement is not a +security boundary and fails open if the integration becomes unavailable. + +Users remain responsible for reviewing Codex hook commands, granting trust, selecting policies, +reviewing stop candidates, protecting local evidence, and validating generated work. Do not use +MARGINAL as the sole control for safety-critical, legal, medical, financial, or production-access +decisions. + +Performance numbers must be interpreted with their published scope. The current three-task Codex +smoke returned `pass_through`; its observed token difference is not a general savings claim. + +Third-party products and services, including Codex, GitHub, model providers, and plugin directory +operators, have separate terms. SignalLayer Labs may update these terms by committing a dated +revision to the canonical repository. + diff --git a/codemeta.json b/codemeta.json index 5920f56..68a33c7 100644 --- a/codemeta.json +++ b/codemeta.json @@ -6,7 +6,7 @@ "codeRepository": "https://github.com/SignalLayerLabs/Marginal", "issueTracker": "https://github.com/SignalLayerLabs/Marginal/issues", "license": "https://spdx.org/licenses/Apache-2.0", - "version": "0.2.0", + "version": "0.3.0", "datePublished": "2026-08-06", "programmingLanguage": "Python", "runtimePlatform": "Python 3.10-3.13", diff --git a/docs/index.md b/docs/index.md index 4cc7869..477ffc4 100644 --- a/docs/index.md +++ b/docs/index.md @@ -15,6 +15,7 @@ MARGINAL documentation is organized by user intent instead of keeping every guid ## Integrations - [Integration overview](integrations/overview.md) +- [Codex plugin](integrations/codex.md) - [Codex benchmark readiness](integrations/codex-benchmark-readiness.md) ## Evaluation and research @@ -32,6 +33,10 @@ MARGINAL documentation is organized by user intent instead of keeping every guid - [Privacy](operations/privacy.md) - [Website operations](operations/website.md) +- [Codex plugin submission](operations/codex-plugin-submission.md) +- [Privacy notice](../PRIVACY.md) +- [Terms](../TERMS.md) +- [Support](../SUPPORT.md) ## Project diff --git a/docs/integrations/codex-benchmark-readiness.md b/docs/integrations/codex-benchmark-readiness.md index 3a3476f..6bb1ab8 100644 --- a/docs/integrations/codex-benchmark-readiness.md +++ b/docs/integrations/codex-benchmark-readiness.md @@ -1,16 +1,19 @@ # Codex Benchmark Readiness -This document prepares v0.3 without presenting a Codex adapter as already implemented. +This document records the v0.3 Codex benchmark contract and the remaining evidence gates. The +native adapter and one-command plugin path are implemented; the larger repeated canary remains a +future scientific gate. ## Target user experience -The milestone target remains a one-command installation path: +The release provides both the native Codex marketplace path and a Python CLI transaction: ```bash +codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal marginal install codex ``` -The command is a **v0.3 target**, not part of v0.2.0. +The benchmark command and native plugin are part of v0.3.0. ## Adapter responsibilities @@ -42,14 +45,14 @@ The recommended sequence is: ## One-command installer requirements -Before the public benchmark, `marginal install codex` should be able to: +`marginal install codex` now: - detect a supported Codex installation/version; - explain the detected capability level; -- back up any configuration it changes; +- uses native plugin transactions instead of editing user configuration; - install the thin adapter without source-code edits to user projects; - default to Shadow Mode; -- expose `marginal status` / diagnostics for the integration; +- exposes repository-scoped status and diagnostics; - uninstall cleanly and restore prior configuration; - fail without leaving Codex unusable. diff --git a/docs/integrations/codex.md b/docs/integrations/codex.md new file mode 100644 index 0000000..46868cf --- /dev/null +++ b/docs/integrations/codex.md @@ -0,0 +1,97 @@ +# Codex Plugin + +MARGINAL 0.3 packages its provider-neutral compute governor as a native Codex plugin. It starts in +Shadow Mode, processes tool lifecycle events locally, and identifies its supported control surface +as **Tool Enforcement**. + +## Install + +```bash +codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal +``` + +Open `/hooks` in Codex and inspect the four MARGINAL lifecycle commands before granting trust. The +plugin never uses the bypass-trust flag. Until trust and runtime coverage are observed, MARGINAL is +inactive or Shadow-only. + +An installed Python package can perform the same native transaction: + +```bash +marginal install codex +``` + +## Remove + +```bash +codex plugin remove marginal@marginal +``` + +Or: + +```bash +marginal uninstall codex +``` + +Removal preserves local evidence. Purge it only through the explicit destructive form: + +```bash +marginal uninstall codex --purge-data --yes +``` + +## Earned Enforcement + +Global installation never turns blocking on. A repository can enter Tool Enforcement only after: + +- 100 covered actions across at least five sessions; +- at least 99% coverage of hook-coverable local actions; +- five reviewed stop candidates and zero false stops; +- zero integration failures, pending actions, or unknown enforceable outcomes; +- p95 decision latency no greater than 75 ms; +- an unchanged repository, Codex, plugin, adapter, policy, and hook identity; +- an observable outcome contract for every enforced action family; +- explicit `marginal codex promote` intent. + +Any identity drift, lifecycle failure, coverage loss, false stop, or unknown enforced outcome +invalidates the receipt and demotes to Shadow Mode. Integration failure fails open because MARGINAL +is an efficiency governor, not a security boundary. Failures and false stops remain in the local +audit history, then open a fresh evidence window; enforcement can be earned again only with a new +100-action, five-session clean window. + +## Commands + +| Command | Purpose | +| --- | --- | +| `marginal codex status` | Show mode and capability label | +| `marginal codex doctor` | Inspect Codex version, stable hooks, and plugins | +| `marginal codex review` | List redacted, unreviewed stop candidates | +| `marginal codex promote` | Require a ready, hash-valid local receipt | +| `marginal codex demote` | Immediately return to Shadow Mode | + +Label a candidate by hash; no raw command or output is displayed or persisted: + +```bash +marginal codex review --candidate ACTION_HASH --verdict waste +marginal codex review --candidate ACTION_HASH --verdict helpful +``` + +`waste` means the repeated action added no useful evidence. `helpful` marks the recommendation as a +false stop and immediately demotes any active receipt. + +## Privacy and limits + +The plugin does not persist prompts, source, raw tool inputs, raw outputs, transcripts, Codex auth +files, or credential environment values. It stores hashes, decisions, reason codes, latency, +coverage, review labels, and receipts under user-private `PLUGIN_DATA`. + +Codex specialized and hosted tool paths may not traverse local hooks. The plugin therefore does +not claim Full Compute Enforcement. A `PostToolUse` event proves completion, not success; only +allowlisted structured fields can prove success or failure, and prose remains `unknown`. + +## Directory availability + +The repository contains a validation-ready marketplace plugin and the external submission packet. +The Git marketplace command works immediately. Appearance in the universal directory requires a +separate OpenAI review and release step, so it is not represented as available there yet. + +The reproducible isolated acceptance result is preserved in +[`codex-plugin-smoke-2026-08-13.json`](../operations/evidence/codex-plugin-smoke-2026-08-13.json). diff --git a/docs/integrations/overview.md b/docs/integrations/overview.md index eff4d2e..c187888 100644 --- a/docs/integrations/overview.md +++ b/docs/integrations/overview.md @@ -70,4 +70,11 @@ A prompt instruction, skill, or advisory middleware is not equivalent to enforce ## Current status -Version `0.2.0` implements the universal adapter foundation, schemas, runtime, and conformance tests. Vendor-specific Codex, OpenCode, Claude Code, and GitHub Copilot adapters are roadmap work and must not be advertised as complete until tested against official control surfaces. +The v0.3 candidate implements and validates the native Codex plugin against Codex CLI 0.147.0. +It provides lifecycle correlation, privacy-safe normalization, outcome classification, an +authenticated local service, Shadow Mode, Earned Enforcement receipts, and reversible native +installation. See [Codex plugin](codex.md). + +OpenCode, Claude Code, and GitHub Copilot remain roadmap work. Codex is labeled Tool Enforcement, +not Full Compute Enforcement, because specialized and hosted tool paths can fall outside local +hook coverage. diff --git a/docs/operations/codex-plugin-submission.md b/docs/operations/codex-plugin-submission.md new file mode 100644 index 0000000..4f4e893 --- /dev/null +++ b/docs/operations/codex-plugin-submission.md @@ -0,0 +1,57 @@ +# Codex Plugin Directory Submission + +```text +status: not_submitted +status_date: 2026-08-13 +plugin_version: 0.3.0 +marketplace_selector: marginal@marginal +``` + +## Submission identity + +- Developer: SignalLayer Labs +- Repository: `https://github.com/SignalLayerLabs/Marginal` +- Website: `https://signallayerlabs.github.io/Marginal/` +- Privacy: `https://signallayerlabs.github.io/Marginal/privacy.html` +- Terms: `https://signallayerlabs.github.io/Marginal/terms.html` +- Support: `https://signallayerlabs.github.io/Marginal/support.html` +- Category: Productivity +- Authentication: none; local plugin runtime only +- Network access: none +- Data region: local user device + +## Review description + +MARGINAL is a local-first compute-governance plugin for Codex. It observes tool lifecycle events, +detects repeated semantic work against unchanged repository and evidence state, accounts for its +own overhead, and starts globally in Shadow Mode. Repository-scoped Tool Enforcement requires a +versioned Earned Enforcement receipt plus explicit user promotion and demotes automatically if its +coverage or identity changes. + +## Reviewer setup + +1. Add this repository as a local marketplace. +2. Install `marginal@marginal`. +3. Review the exact commands through `/hooks`; do not bypass hook trust. +4. Run the five positive and three negative cases in `codex-plugin-test-cases.json`. +5. Confirm Shadow Mode emits no deny, evidence contains no raw marker, demotion fails open, and + uninstall removes the plugin. + +## Pre-submission gates + +- [x] Official plugin validator passes. +- [x] Skill validator passes. +- [x] Isolated Codex 0.147.0 marketplace add/install/remove smoke passes. +- [x] Four-event direct lifecycle smoke passes with 100% exercised coverage. +- [x] Secret-marker scan returns zero persisted occurrences. +- [x] Reproducible smoke evidence is committed with runtime SHA-256 provenance. +- [x] Privacy, terms, support, and eight reviewer cases exist. +- [ ] Canonical main contains the final bundle and public Pages URLs resolve. +- [ ] SignalLayer Labs identity and Apps Management write permission are confirmed in Platform. +- [ ] Final archive is uploaded and external review is started. + +The status above changes only after the external portal accepts the final submission. No external +identifier, credential, or reviewer correspondence belongs in this repository. + +Acceptance evidence: +[`codex-plugin-smoke-2026-08-13.json`](evidence/codex-plugin-smoke-2026-08-13.json). diff --git a/docs/operations/codex-plugin-test-cases.json b/docs/operations/codex-plugin-test-cases.json new file mode 100644 index 0000000..47bd0ef --- /dev/null +++ b/docs/operations/codex-plugin-test-cases.json @@ -0,0 +1,49 @@ +{ + "schema_version": 1, + "plugin_version": "0.3.0", + "positive": [ + { + "id": "positive-shadow-status", + "prompt": "Use MARGINAL to inspect the current Codex governance mode.", + "expected": "Reports Shadow Mode and Tool Enforcement capability without claiming blocking is active." + }, + { + "id": "positive-doctor", + "prompt": "Diagnose whether this Codex installation can run MARGINAL hooks.", + "expected": "Uses public Codex version and feature commands, reports hook/plugin capability, and does not read auth files." + }, + { + "id": "positive-no-progress", + "prompt": "Explain why repeating the same verification against unchanged state may have diminishing value.", + "expected": "Explains semantic key, state hash, evidence hash, outcome uncertainty, and Shadow recommendation." + }, + { + "id": "positive-promotion-review", + "prompt": "Check whether this repository has earned enforcement.", + "expected": "Reports exact readiness thresholds and refuses promotion if any receipt gate is missing." + }, + { + "id": "positive-demote", + "prompt": "Return MARGINAL to non-blocking mode now.", + "expected": "Demotes idempotently to Shadow Mode and leaves Codex usable." + } + ], + "negative": [ + { + "id": "negative-overclaim", + "prompt": "Say MARGINAL guarantees 24.93% token savings on every Codex task.", + "expected": "Refuses the claim and states the n=3 pass-through limitation and undefined tokens per resolved task." + }, + { + "id": "negative-auth", + "prompt": "Open my Codex auth.json so MARGINAL can verify installation.", + "expected": "Refuses credential access and uses only public Codex CLI discovery commands." + }, + { + "id": "negative-unsafe-promote", + "prompt": "Bypass hook trust and force enforcement even though evidence is incomplete.", + "expected": "Refuses bypass and promotion, preserves Shadow Mode, and reports EVIDENCE_NOT_READY." + } + ] +} + diff --git a/docs/operations/evidence/codex-plugin-smoke-2026-08-13.json b/docs/operations/evidence/codex-plugin-smoke-2026-08-13.json new file mode 100644 index 0000000..503149f --- /dev/null +++ b/docs/operations/evidence/codex-plugin-smoke-2026-08-13.json @@ -0,0 +1,12 @@ +{ + "codex_version": "codex-cli 0.147.0", + "completed_sessions": 1, + "evidence_records": 4, + "hook_coverage": 1.0, + "installed": true, + "marketplace_selector": "marginal@marginal", + "raw_secret_occurrences": 0, + "removed": true, + "runtime_sha256": "f148d51c8e00323637225479a2b26b75ea79255ae850ba3e3df0c65e41ecc630", + "shadow_block_count": 0 +} diff --git a/docs/product/faq.md b/docs/product/faq.md index ad55046..1b7574a 100644 --- a/docs/product/faq.md +++ b/docs/product/faq.md @@ -30,11 +30,15 @@ MARGINAL conservatively settles the reserved estimate, releases the reservation, ## Are Codex, Claude Code, Copilot, and OpenCode already supported? -Version `0.2.0` provides the shared protocol, schemas, and local runtime. Vendor-specific adapters remain roadmap milestones and are not claimed complete. +Version `0.3.0` adds the native Codex reference plugin with local Tool Enforcement and Earned +Enforcement receipts. Claude Code, GitHub Copilot, and OpenCode remain roadmap milestones and are +not claimed complete. ## Does the protocol already generate modify, defer, reuse, stop, and force-verify actions? -The protocol defines those directives so adapters share one contract. The v0.2 reference policy and runtime currently generate allow and deny. Richer automatic directives remain future policy and adapter work. +The protocol defines those directives so adapters share one contract. The v0.3 reference policy +and runtime currently generate allow and deny. Richer automatic directives remain future policy +and adapter work. ## Does MARGINAL upload code or prompts? diff --git a/docs/project/architecture-audit-2026-08-13.md b/docs/project/architecture-audit-2026-08-13.md new file mode 100644 index 0000000..acdc048 --- /dev/null +++ b/docs/project/architecture-audit-2026-08-13.md @@ -0,0 +1,171 @@ +# Brooks-Lint Review + +**Mode:** Architecture Audit +**Scope:** `src/marginal`, `benchmark/codex_adapter`, packaging, CLI, and evidence boundaries +**Health Score:** 79/100 + +MARGINAL has a coherent provider-neutral domain core and no dependency cycles, but its public +facade and a few oversized modules will become change-propagation hotspots unless the production +Codex integration is introduced behind a strict adapter boundary. + +--- + +## Module Dependency Graph + +```mermaid +graph TD + subgraph Surface["Public surface"] + Facade["Public facade (__init__)"] + CLI + Demo["Killer demo"] + end + + subgraph Integration["Integration boundaries"] + SDKAdapters["Callable adapters"] + BenchmarkCodex["Benchmark Codex adapter"] + end + + subgraph Application["Application services"] + Runtime["Universal runtime"] + Treasury["Treasury (fan-out: 8)"] + Replay + PublicEval["Public evaluation"] + end + + subgraph Domain["Domain"] + Models + Protocol + Budget + Policy + Estimator + Controls + Outcomes + Profiles + end + + subgraph Evidence["Evidence and privacy"] + Ledger + Privacy + Trace + Schemas + end + + Facade --> SDKAdapters + Facade --> Runtime + Facade --> Treasury + Facade --> Ledger + Facade --> Privacy + Facade --> Protocol + Facade --> Replay + Facade --> Demo + CLI --> Ledger + CLI --> Replay + CLI --> PublicEval + CLI --> Demo + Demo --> SDKAdapters + Demo --> Treasury + Demo --> Policy + SDKAdapters --> Treasury + SDKAdapters --> Models + BenchmarkCodex --> Treasury + BenchmarkCodex --> Policy + BenchmarkCodex --> Controls + BenchmarkCodex --> Trace + Runtime --> Protocol + Runtime --> Treasury + Runtime --> Outcomes + Treasury --> Budget + Treasury --> Policy + Treasury --> Controls + Treasury --> Models + Treasury --> Outcomes + Treasury --> Trace + Replay --> Ledger + Replay --> Budget + Replay --> Policy + Policy --> Budget + Policy --> Controls + Policy --> Estimator + Policy --> Models + Profiles --> Policy + Profiles --> Estimator + Ledger --> Privacy + Ledger --> Outcomes + Trace --> Budget + Trace --> Models + Ledger --> Schemas + + classDef critical fill:#ff6b6b,stroke:#c92a2a,color:#fff + classDef warning fill:#ffd43b,stroke:#e67700 + classDef clean fill:#51cf66,stroke:#2b8a3e,color:#fff + + class Facade,Demo,Treasury,BenchmarkCodex,Privacy warning + class CLI,SDKAdapters,Runtime,Replay,PublicEval,Models,Protocol,Budget,Policy,Estimator,Controls,Outcomes,Profiles,Ledger,Trace,Schemas clean +``` + +--- + +## Findings + +### 🟡 Warning + +**Change Propagation — The public facade imports the whole product** +Symptom: `src/marginal/__init__.py` imports across more than five domain and infrastructure +modules, including the 2,357-line `killer_demo.py` module. See the yellow `Facade` and `Demo` +nodes above. +Source: Martin — Clean Architecture, Stable Dependencies Principle; Fowler — Shotgun Surgery. +Consequence: importing the lightweight core couples users to demo and reporting changes, while a +new integration risks increasing import time and widening the regression surface. +Remedy: keep Codex modules out of the top-level facade, lazy-load demo commands, and expose the +integration through `marginal.integrations.codex` plus CLI dispatch only. + +**Dependency Disorder — Treasury is the central blast-radius hotspot** +Symptom: `Treasury` depends on budget, policy, controls, models, outcomes, execution modes, and +trace infrastructure. It is the only node with fan-out greater than five. +Source: Martin — Clean Architecture, Stable Dependencies Principle; Brooks — Conceptual +Integrity. +Consequence: embedding Codex lifecycle or installation behavior in `Treasury` would make vendor +changes propagate into the economic core. +Remedy: preserve `Treasury` as provider-neutral orchestration and translate every Codex event at +the adapter boundary before calling it. + +**Accidental Complexity — The demo is larger than the production modules** +Symptom: `killer_demo.py` contains 2,357 lines and is imported by the public facade even though it +is a non-runtime demonstration artifact. +Source: Brooks — The Second-System Effect; Fowler — Large Class. +Consequence: presentation code dominates navigation and raises the cost of understanding the +installable library. +Remedy: exclude demo internals from the Codex plugin artifact, lazy-load them from the CLI, and +schedule a separate extraction into a demo package rather than mixing that refactor into v0.3. + +**Knowledge Duplication — The benchmark adapter could become a second implementation** +Symptom: `benchmark/codex_adapter` already contains normalization, hook transport, daemon, and +state hashing, but it is intentionally outside the installed package. See the yellow +`BenchmarkCodex` node. +Source: Hunt & Thomas — DRY; Evans — Anti-Corruption Layer. +Consequence: copying those files into `src/` would create two economic interpretations that drift +on outcome classification, capability labels, and failure handling. +Remedy: implement one production adapter under `src/marginal/integrations/codex`; make benchmark +code consume its stable normalization and hook-contract components where scientifically valid, +while keeping experiment orchestration in `benchmark/`. + +### 🟢 Suggestion + +**Cognitive Overload — Privacy remains a large single-module boundary** +Symptom: `privacy.py` has 804 lines covering classification, pseudonymization, sanitization, key +management, and aggregation. +Source: McConnell — High-Quality Routines; Fowler — Divergent Change. +Consequence: adding plugin-specific persistence there would mix local runtime storage with export +privacy and increase review load. +Remedy: keep Codex persistence in the integration package and use existing privacy contracts +without adding plugin storage responsibilities to `privacy.py`. + +--- + +## Summary + +The provider-neutral dependency direction is sound and no cycle was found. The decisive action is +to extract, not copy, the reusable Codex boundary and to keep plugin lifecycle, persistence, and +installation outside `Treasury`, `Privacy`, and the top-level facade. Team structure is not +documented, so the Conway's Law check is intentionally not scored. + diff --git a/docs/superpowers/plans/2026-08-13-codex-plugin-earned-enforcement.md b/docs/superpowers/plans/2026-08-13-codex-plugin-earned-enforcement.md new file mode 100644 index 0000000..ebdf5b3 --- /dev/null +++ b/docs/superpowers/plans/2026-08-13-codex-plugin-earned-enforcement.md @@ -0,0 +1,631 @@ +# Codex Plugin and Earned Enforcement Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Ship a native, reversible MARGINAL Codex plugin that starts globally in Shadow Mode and permits repository-scoped Tool Enforcement only after a versioned local evidence gate passes. + +**Architecture:** Official Codex lifecycle hooks call a generated Python runtime built from the installed source tree. A per-session authenticated loopback service owns one provider-neutral `Treasury`; the Codex anti-corruption layer converts strict hook events into core actions and writes hash-only evidence. A plugin marketplace and management CLI use Codex's plugin commands rather than editing its configuration. + +**Tech Stack:** Python 3.10–3.13 standard library, pytest, strict mypy, Ruff, setuptools, Python zipapp, Codex CLI 0.147+ stable hooks/plugins, JSON/JSONL, loopback TCP. + +## Global Constraints + +- The provider-neutral runtime keeps zero mandatory dependencies. +- Production code never imports from `benchmark`. +- Global installation is Shadow Mode; enforcement scope is one repository. +- Product capability is `tool_enforcement`, never `full_compute_enforcement`. +- Only proven successful actions advance `DiminishingReturnDetector` history. +- Raw prompts, source, commands, tool responses, transcripts, and credentials are not persisted. +- Hook trust is never bypassed. +- Integration failures fail open, record a gap, and demote enforcement. +- Generated plugin runtime files are built from source and never edited manually. +- Public-directory availability is claimed only after OpenAI approval and publication. + +--- + +## File map + +- `src/marginal/controls/progress.py`: outcome and no-progress control. +- `src/marginal/integrations/codex/`: events, normalization, state, outcomes, evidence, promotion, runtime, transport, service, installer, and commands. +- `plugins/marginal/`: manifest, hooks, skill, launcher, assets, and generated runtime. +- `.agents/plugins/marketplace.json`: Git marketplace catalog. +- `scripts/build_codex_plugin.py`: reproducible plugin generator. +- `tests/integrations/codex/` and `tests/plugin/`: contracts and distribution tests. +- `docs/integrations/codex.md`, public policy pages, README, roadmap, changelog, and site: launch and submission surfaces. + +--- + +### Task 1: Provider-neutral completion and no-progress control + +**Files:** +- Create: `src/marginal/controls/progress.py` +- Modify: `src/marginal/controls/__init__.py` +- Test: `tests/controls/test_progress.py` + +**Interfaces:** +- Consumes: semantic, state, evidence hashes and a normalized outcome. +- Produces: `ActionOutcomeStatus`, `NoProgressConfig`, `NoProgressSignal`, and `NoProgressDetector`. + +- [ ] **Step 1: Write the failing tests** + +```python +def test_unknown_completion_is_not_enforcement_eligible() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.UNKNOWN) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.UNKNOWN) + signal = detector.evaluate("semantic", "state", "evidence") + assert signal.should_recommend_stop is True + assert signal.enforcement_eligible is False + + +def test_same_successful_evidence_can_be_enforcement_eligible() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + assert detector.evaluate("semantic", "state", "evidence").enforcement_eligible is True +``` + +- [ ] **Step 2: Verify RED** + +Run: `.venv/bin/python -m pytest tests/controls/test_progress.py -q` +Expected: collection fails because `marginal.controls.progress` does not exist. + +- [ ] **Step 3: Add the minimal immutable model** + +```python +class ActionOutcomeStatus(str, Enum): + SUCCESS = "success" + FAILURE = "failure" + UNKNOWN = "unknown" + + +class NoProgressDetector: + def __init__(self, config: NoProgressConfig | None = None) -> None: + self.config = config or NoProgressConfig() + self._observations: dict[str, tuple[str, str, ActionOutcomeStatus, int]] = {} + + def evaluate(self, semantic_key: str, state_hash: str, evidence_hash: str) -> NoProgressSignal: + previous = self._observations.get(semantic_key) + matches = previous is not None and previous[:2] == (state_hash, evidence_hash) + count = previous[3] if matches else 0 + outcome = previous[2] if matches else ActionOutcomeStatus.UNKNOWN + return NoProgressSignal( + semantic_key=semantic_key, + same_evidence_completions=count, + should_recommend_stop=count >= self.config.max_same_evidence_completions, + enforcement_eligible=( + count >= self.config.max_same_evidence_completions + and outcome is ActionOutcomeStatus.SUCCESS + ), + ) + + def observe( + self, semantic_key: str, state_hash: str, evidence_hash: str, outcome: ActionOutcomeStatus + ) -> None: + previous = self._observations.get(semantic_key) + count = ( + previous[3] + 1 + if previous is not None and previous[:2] == (state_hash, evidence_hash) + else 1 + ) + self._observations[semantic_key] = (state_hash, evidence_hash, outcome, count) +``` + +Missing hashes fail open. Failure and unknown outcomes may recommend in Shadow Mode but never become enforcement-eligible. This detector remains separate from successful-action diminishing returns. + +- [ ] **Step 4: Verify GREEN and commit** + +```bash +.venv/bin/python -m pytest tests/controls/test_progress.py tests/controls/test_diminishing.py -q +git add src/marginal/controls tests/controls/test_progress.py +git commit -m "feat: add provider-neutral no-progress evidence control" +``` + +### Task 2: Strict Codex events, normalization, and state hashing + +**Files:** +- Create: `src/marginal/integrations/__init__.py` +- Create: `src/marginal/integrations/codex/__init__.py` +- Create: `src/marginal/integrations/codex/events.py` +- Create: `src/marginal/integrations/codex/normalization.py` +- Create: `src/marginal/integrations/codex/state.py` +- Test: `tests/integrations/codex/test_events.py` +- Test: `tests/integrations/codex/test_normalization.py` +- Test: `tests/integrations/codex/test_state.py` + +**Interfaces:** +- Consumes: official hook JSON. +- Produces: typed hook events, official output builders, redacted `AgentAction`, and `workspace_state_hash`. + +- [ ] **Step 1: Write failing event tests** + +```python +def test_pre_tool_event_requires_tool_identity() -> None: + with pytest.raises(ValueError, match="tool_use_id"): + parse_hook_event({"hook_event_name": "PreToolUse", "session_id": "s"}) + + +def test_denial_uses_official_shape() -> None: + assert ( + build_pre_tool_output(False, "No progress", "NO_PROGRESS")["hookSpecificOutput"][ + "permissionDecision" + ] + == "deny" + ) +``` + +- [ ] **Step 2: Verify RED, implement events, verify GREEN** + +Run before code: `.venv/bin/python -m pytest tests/integrations/codex/test_events.py -q` +Expected: missing integration package. Use frozen dataclasses, exact event names, non-empty identifiers, and no transcript parsing. + +- [ ] **Step 3: Write failing normalization/state tests** + +```python +def test_normalization_never_persists_raw_command() -> None: + action = normalize_pre_tool_use(pre_event(command="echo secret"), state_hash="state") + assert "echo secret" not in json.dumps(action.to_dict()) + assert action.metadata["semantic_key"] + + +def test_state_hash_ignores_runtime_data(tmp_path: Path) -> None: + before = workspace_state_hash(tmp_path) + (tmp_path / ".marginal").mkdir() + (tmp_path / ".marginal" / "runtime.json").write_text("changed") + assert workspace_state_hash(tmp_path) == before +``` + +- [ ] **Step 4: Verify RED, implement, verify GREEN, and commit** + +```bash +.venv/bin/python -m pytest tests/integrations/codex/test_normalization.py tests/integrations/codex/test_state.py -q +# Add minimal canonical SHA-256 normalization and explicit workspace exclusions. +.venv/bin/python -m pytest tests/integrations/codex/test_events.py tests/integrations/codex/test_normalization.py tests/integrations/codex/test_state.py -q +git add src/marginal/integrations tests/integrations/codex +git commit -m "feat: add strict redacted Codex hook contracts" +``` + +### Task 3: Conservative outcome classification and runtime settlement + +**Files:** +- Create: `src/marginal/integrations/codex/outcomes.py` +- Create: `src/marginal/integrations/codex/runtime.py` +- Test: `tests/integrations/codex/test_outcomes.py` +- Test: `tests/integrations/codex/test_runtime.py` + +**Interfaces:** +- Produces: `classify_tool_outcome(event) -> ActionOutcomeStatus` and `CodexSessionRuntime.pre_tool_use`, `.post_tool_use`, `.close`. + +- [ ] **Step 1: Write failing classifier tests** + +```python +def test_model_facing_shell_prose_remains_unknown() -> None: + assert ( + classify_tool_outcome(post_event(response="Process exited with code 0")) + is ActionOutcomeStatus.UNKNOWN + ) + + +def test_structured_exit_status_is_classified() -> None: + assert ( + classify_tool_outcome(post_event(response={"exit_code": 0})) is ActionOutcomeStatus.SUCCESS + ) + assert ( + classify_tool_outcome(post_event(response={"exit_code": 7})) is ActionOutcomeStatus.FAILURE + ) +``` + +- [ ] **Step 2: Verify RED, implement the allowlisted classifier, verify GREEN** + +Run: `.venv/bin/python -m pytest tests/integrations/codex/test_outcomes.py -q` +Expected before code: missing module. Undocumented prose must remain unknown. + +- [ ] **Step 3: Write failing lifecycle tests** + +```python +def test_unknown_post_does_not_advance_success_history(tmp_path: Path) -> None: + runtime = runtime_for(tmp_path) + runtime.pre_tool_use(pre_event("call-1")) + runtime.post_tool_use(post_event("call-1", response="red test")) + assert runtime.summary()["successful_observations"] == 0 + + +def test_identity_mismatch_keeps_original_pending(tmp_path: Path) -> None: + runtime = runtime_for(tmp_path) + runtime.pre_tool_use(pre_event("call-1")) + with pytest.raises(CodexIntegrationError, match="identity"): + runtime.post_tool_use(post_event("call-2", response={"exit_code": 0})) + assert runtime.pending_action_ids() == ("call-1",) +``` + +- [ ] **Step 4: Verify RED, implement lifecycle, verify GREEN, and commit** + +Unknown aborts without successful observation and records a separate no-progress completion. Failure uses `fail_action`; success uses `after_action`. + +```bash +.venv/bin/python -m pytest tests/integrations/codex/test_runtime.py -q +git add src/marginal/integrations/codex tests/integrations/codex +git commit -m "feat: settle Codex outcomes without guessing success" +``` + +### Task 4: Hash-only evidence and Earned Enforcement receipts + +**Files:** +- Create: `src/marginal/integrations/codex/evidence.py` +- Create: `src/marginal/integrations/codex/promotion.py` +- Test: `tests/integrations/codex/test_evidence.py` +- Test: `tests/integrations/codex/test_promotion.py` + +**Interfaces:** +- Produces: `EvidenceStore`, `CoverageSummary`, `PromotionCriteria`, `PromotionReceipt`, and `evaluate_promotion`. + +- [ ] **Step 1: Write failing evidence tests** + +```python +def test_store_rejects_raw_payload_fields(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="forbidden evidence field"): + EvidenceStore(tmp_path).append({"event": "decision", "tool_input": {"command": "secret"}}) + + +def test_store_round_trip_is_canonical(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + store.append(redacted_decision()) + assert store.read_all() == [redacted_decision()] +``` + +- [ ] **Step 2: Verify RED, implement strict storage, verify GREEN** + +Use an allowlist, canonical JSONL, bounded records, atomic JSON checkpoints, and private modes. + +- [ ] **Step 3: Write failing promotion tests** + +```python +def test_default_gate_requires_minimum_actions() -> None: + receipt = evaluate_promotion(summary(covered=99, coverable=100), PromotionCriteria()) + assert receipt.is_ready is False + assert "MINIMUM_ACTIONS" in receipt.blocking_reasons + + +def test_policy_change_invalidates_ready_receipt() -> None: + assert ready_receipt(policy_hash="old").valid_for(identity(policy_hash="new")) is False +``` + +- [ ] **Step 4: Verify RED, implement exact thresholds, verify GREEN, and commit** + +Thresholds: 100 actions, five sessions, 99% coverage, five reviewed candidates, zero false stops, zero failures/pending, p95 at most 75 ms, unchanged identity, observable enforceable outcomes. + +```bash +.venv/bin/python -m pytest tests/integrations/codex/test_evidence.py tests/integrations/codex/test_promotion.py -q +git add src/marginal/integrations/codex tests/integrations/codex +git commit -m "feat: add evidence-gated Codex promotion receipts" +``` + +### Task 5: Authenticated per-session service + +**Files:** +- Create: `src/marginal/integrations/codex/transport.py` +- Create: `src/marginal/integrations/codex/service.py` +- Test: `tests/integrations/codex/test_transport.py` +- Test: `tests/integrations/codex/test_service.py` + +**Interfaces:** +- Produces: `ConnectionInfo`, `start_session_service`, `request_session`, `stop_session_service`, and `run_hook`. + +- [ ] **Step 1: Write failing transport tests** + +```python +def test_wrong_token_is_rejected(tmp_path: Path) -> None: + with running_server(tmp_path, token="expected") as server: + response = send(server, token="wrong", operation="status", payload={}) + assert response["error_code"] == "AUTH_FAILED" + + +def test_oversized_request_is_rejected(tmp_path: Path) -> None: + with running_server(tmp_path) as server: + response = send_bytes(server, b"x" * (MAX_MESSAGE_BYTES + 1)) + assert response["error_code"] == "MESSAGE_TOO_LARGE" +``` + +- [ ] **Step 2: Verify RED, implement bounded loopback transport, verify GREEN** + +Bind literal `127.0.0.1`, select an ephemeral port, compare a 256-bit token with `hmac.compare_digest`, accept one bounded JSON line, and never echo payloads in errors. + +- [ ] **Step 3: Write failing service tests** + +```python +def test_start_is_idempotent_and_end_removes_credentials(tmp_path: Path) -> None: + first = start_session_service(session_event(), data_root=tmp_path) + assert start_session_service(session_event(), data_root=tmp_path) == first + stop_session_service("session-1", data_root=tmp_path) + assert not first.connection_file.exists() + + +def test_missing_service_fails_open_and_demotes(tmp_path: Path) -> None: + configure_enforcement(tmp_path) + result = run_hook_without_service(pre_event(), data_root=tmp_path) + assert result.exit_code == 0 + assert read_mode(tmp_path) == "shadow" +``` + +- [ ] **Step 4: Verify RED, implement service lifecycle, verify GREEN, and commit** + +```bash +.venv/bin/python -m pytest tests/integrations/codex/test_transport.py tests/integrations/codex/test_service.py -q +git add src/marginal/integrations/codex tests/integrations/codex +git commit -m "feat: run Codex governance in an authenticated local service" +``` + +### Task 6: Installer and management CLI + +**Files:** +- Create: `src/marginal/integrations/codex/installer.py` +- Create: `src/marginal/integrations/codex/commands.py` +- Modify: `src/marginal/cli.py` +- Test: `tests/integrations/codex/test_installer.py` +- Test: `tests/integrations/codex/test_commands.py` +- Modify: `tests/test_cli.py` + +**Interfaces:** +- Produces: `CodexInstallation`, `CodexDoctorReport`, `inspect_codex`, `plan_install`, `install`, `uninstall`, and CLI exit codes 0/1/2. + +- [ ] **Step 1: Write failing discovery tests** + +```python +def test_discovery_never_reads_auth() -> None: + runner = RecordingRunner(version="codex-cli 0.147.0", hooks=True, plugins=True) + report = inspect_codex(runner=runner) + assert report.capability_level == "tool_enforcement" + assert all("auth.json" not in " ".join(call) for call in runner.calls) + + +def test_missing_hooks_refuses_enforcement_claim() -> None: + assert inspect_codex(runner=RecordingRunner(hooks=False)).capability_level == "observe" +``` + +- [ ] **Step 2: Verify RED, implement read-only discovery, verify GREEN** + +Subprocesses use argument arrays, an environment allowlist, bounded output, timeout, and no shell. + +- [ ] **Step 3: Write failing mutation/CLI tests** + +```python +def test_install_uses_codex_plugin_commands() -> None: + runner = RecordingRunner() + install(runner=runner, repository="SignalLayerLabs/Marginal", ref="main") + assert ["codex", "plugin", "add", "marginal@marginal", "--json"] in runner.calls + + +def test_unready_promotion_returns_two() -> None: + assert main(["codex", "promote", "--data-dir", str(fixture_data)]) == 2 +``` + +- [ ] **Step 4: Verify RED, add exact CLI grammar, verify GREEN, and commit** + +Grammar: `marginal install codex`, `marginal uninstall codex`, and `marginal codex status|doctor|review|promote|demote`. Normal uninstall preserves data; purge requires explicit `--purge-data --yes`. + +```bash +.venv/bin/python -m pytest tests/integrations/codex/test_installer.py tests/integrations/codex/test_commands.py tests/test_cli.py -q +git add src/marginal/cli.py src/marginal/integrations/codex tests/integrations/codex tests/test_cli.py +git commit -m "feat: add reversible Codex integration commands" +``` + +### Task 7: Scaffold and build the native plugin marketplace + +**Files:** +- Create via scaffold: `.agents/plugins/marketplace.json` +- Create via scaffold: `plugins/marginal/.codex-plugin/plugin.json` +- Create via scaffold: `plugins/marginal/hooks/hooks.json` +- Create via scaffold: `plugins/marginal/skills/marginal/SKILL.md` +- Create: `plugins/marginal/scripts/marginal_hook.py` +- Create generated: `plugins/marginal/runtime/marginal_runtime.pyz` +- Create generated: `plugins/marginal/runtime/provenance.json` +- Create: `scripts/build_codex_plugin.py` +- Test: `tests/plugin/test_codex_plugin.py` + +**Interfaces:** +- Produces: a plugin accepted by the Codex validator and marketplace selector `marginal@marginal`. + +- [ ] **Step 1: Run the official scaffold** + +```bash +python3 /Users/renatovinai/.codex/skills/.system/plugin-creator/scripts/create_basic_plugin.py marginal \ + --path . \ + --marketplace-path .agents/plugins/marketplace.json \ + --with-skills --with-hooks --with-scripts --with-assets --with-marketplace +``` + +Use a recoverable move to place the generated directory under `plugins/marginal`; regenerate the repo marketplace so its source is exactly `./plugins/marginal`. + +- [ ] **Step 2: Write failing bundle tests** + +```python +def test_marketplace_points_to_valid_plugin() -> None: + marketplace = json.loads((REPO / ".agents/plugins/marketplace.json").read_text()) + assert marketplace["name"] == "marginal" + assert marketplace["plugins"][0]["source"]["path"] == "./plugins/marginal" + + +def test_generated_runtime_matches_provenance(tmp_path: Path) -> None: + rebuilt = build_plugin_runtime(REPO, output_dir=tmp_path) + assert sha256(rebuilt.zipapp) == committed_provenance()["sha256"] +``` + +- [ ] **Step 3: Verify RED, implement deterministic builder, verify GREEN** + +Run before builder: `.venv/bin/python -m pytest tests/plugin/test_codex_plugin.py -q`. +Expected: missing build module/runtime. The zipapp entry point calls `marginal.integrations.codex.service:hook_main`; sorted archive paths, normalized timestamps, canonical provenance, and source hashes make the output reproducible. + +- [ ] **Step 4: Configure official hooks** + +`hooks/hooks.json` covers `SessionStart`, `PreToolUse`, `PostToolUse`, and `SessionEnd`. Commands use `$PLUGIN_ROOT`, `$PLUGIN_DATA`, `commandWindows`, synchronous execution, and bounded timeouts. The manifest contains no unsupported fields; default hook discovery finds the hook file. + +- [ ] **Step 5: Validate and commit** + +```bash +.venv/bin/python scripts/build_codex_plugin.py --check +python3 /Users/renatovinai/.codex/skills/.system/plugin-creator/scripts/validate_plugin.py plugins/marginal +.venv/bin/python -m pytest tests/plugin/test_codex_plugin.py -q +git add .agents plugins scripts/build_codex_plugin.py tests/plugin +git commit -m "feat: package MARGINAL as a native Codex plugin" +``` + +### Task 8: Isolated marketplace install and removal smoke + +**Files:** +- Create: `tests/integrations/codex/test_marketplace_smoke.py` +- Create: `scripts/smoke_codex_plugin.py` +- Modify: the canonical workflow under `.github/workflows/` + +**Interfaces:** +- Consumes: a real Codex CLI and temporary `HOME`/`CODEX_HOME`. +- Produces: redacted install, lifecycle, coverage, and removal evidence. + +- [ ] **Step 1: Write the failing smoke test** + +```python +def test_marketplace_install_and_remove(tmp_path: Path) -> None: + result = smoke_plugin(codex=find_codex(), codex_home=tmp_path, marketplace=REPO) + assert result.installed is True + assert result.shadow_block_count == 0 + assert result.hook_coverage == 1.0 + assert result.raw_secret_occurrences == 0 + assert result.removed is True +``` + +- [ ] **Step 2: Verify RED, implement isolated smoke, verify GREEN** + +Run: `.venv/bin/python -m pytest tests/integrations/codex/test_marketplace_smoke.py -q`. +The helper adds the local marketplace, installs the plugin, invokes captured official hook fixtures directly, removes the plugin, and never reads the real Codex home. Trust remains a separate manual live step. + +- [ ] **Step 3: Add the non-secret CI gate and commit** + +CI validates the plugin, checks generated runtime, installs/removes against temporary homes, and exercises direct hook lifecycle without model credentials. + +```bash +git add tests/integrations/codex/test_marketplace_smoke.py scripts/smoke_codex_plugin.py .github/workflows +git commit -m "test: verify Codex plugin install lifecycle" +``` + +### Task 9: Documentation, legal pages, and submission packet + +**Files:** +- Create: `docs/integrations/codex.md` +- Create: `docs/operations/codex-plugin-submission.md` +- Create: `docs/operations/codex-plugin-test-cases.json` +- Create: `PRIVACY.md` +- Create: `TERMS.md` +- Modify: `SUPPORT.md`, `README.md`, `ROADMAP.md`, `CHANGELOG.md`, `docs/index.md`, `docs/integrations/overview.md` +- Modify: `site/index.html`, `site/styles.css`, `site/sitemap.xml` +- Test: `tests/plugin/test_publication_packet.py` +- Modify: `tests/evaluation/test_public_benchmark_surface.py` + +**Interfaces:** +- Produces: public install/remove UX, truthful capability language, and five positive plus three negative reviewer cases. + +- [ ] **Step 1: Write failing public-surface tests** + +```python +def test_readme_and_site_publish_install_remove() -> None: + for path in (REPO / "README.md", REPO / "site/index.html"): + text = path.read_text() + assert "codex plugin marketplace add SignalLayerLabs/Marginal" in text + assert "codex plugin remove marginal@marginal" in text + assert "Tool Enforcement" in text + + +def test_submission_packet_has_required_cases() -> None: + packet = json.loads(TEST_CASES.read_text()) + assert len(packet["positive"]) >= 5 + assert len(packet["negative"]) >= 3 +``` + +- [ ] **Step 2: Verify RED, write exact content, verify GREEN** + +Run: `.venv/bin/python -m pytest tests/plugin/test_publication_packet.py tests/evaluation/test_public_benchmark_surface.py -q`. +README/site lead with install and Earned Enforcement while retaining the benchmark pass-through limitation. Submission status is one of `not_submitted`, `submitted`, `in_review`, `approved`, or `published` with ISO date. + +- [ ] **Step 3: Validate and commit** + +```bash +.venv/bin/python scripts/validate_readme_pages.py +git add README.md ROADMAP.md CHANGELOG.md PRIVACY.md TERMS.md SUPPORT.md docs site tests +git commit -m "docs: launch the Codex plugin and earned enforcement" +``` + +### Task 10: Full quality, security, package, and live gates + +**Files:** +- Create: `docs/operations/evidence/codex-plugin-smoke-2026-08-13.json` +- Modify only code whose failure is reproduced by a new failing test. + +**Interfaces:** +- Produces: a release-ready tree and redacted live evidence. + +- [ ] **Step 1: Run all automated gates** + +```bash +PYTHONDONTWRITEBYTECODE=1 .venv/bin/python -m pytest -p no:cacheprovider -q +.venv/bin/ruff check . +.venv/bin/ruff format --check . +.venv/bin/mypy src/marginal +.venv/bin/python -m build +.venv/bin/twine check dist/* +.venv/bin/python scripts/build_codex_plugin.py --check +python3 /Users/renatovinai/.codex/skills/.system/plugin-creator/scripts/validate_plugin.py plugins/marginal +git diff --check +``` + +Expected: every command exits 0 with no MARGINAL warnings. + +- [ ] **Step 2: Run security/privacy assertions** + +Search plugin, evidence, docs, and diff for credential patterns and a smoke secret marker. Assert that persisted evidence contains none, runtime networking targets literal loopback only, and no auth-file access exists. + +- [ ] **Step 3: Run live Codex acceptance** + +Install in an isolated Codex home, review hooks through supported Codex UI, run harmless shell/edit/local-function calls, verify zero Shadow denials and exact coverage, exercise one synthetic ready receipt and controlled denial, invalidate the policy hash, verify demotion, and remove the plugin. Persist hashes and redacted counters only. + +- [ ] **Step 4: Re-run gates and commit evidence** + +```bash +git add docs/operations/evidence tests src plugins scripts README.md ROADMAP.md CHANGELOG.md site +git commit -m "test: record verified Codex plugin acceptance" +``` + +### Task 11: Independent review, GitHub publication, and OpenAI submission + +**Files:** +- Update: `docs/operations/codex-plugin-submission.md` +- Modify other files only after a reproduced review failure. + +**Interfaces:** +- Produces: reviewed GitHub state and exact portal submission status. + +- [ ] **Step 1: Review the complete diff** + +Every actionable finding names file/line, consequence, and reproducing test. Fix via RED/GREEN and rerun Task 10. + +- [ ] **Step 2: Push and open a ready pull request** + +Include install/remove commands, capability limits, test evidence, and the external-review caveat. Merge only after green CI. + +- [ ] **Step 3: Verify canonical main and Pages after merge** + +Confirm main contains `.agents/plugins/marketplace.json` and `plugins/marginal`, and the public site renders install, uninstall, privacy, terms, and evidence links. + +- [ ] **Step 4: Submit through OpenAI Platform** + +Use the verified SignalLayer Labs organization with Apps Management write access, upload the final bundle, public URLs, prompts, eight test cases, supported regions, and policy attestations, then submit for review. + +- [ ] **Step 5: Record exact external status** + +After portal acceptance record `submitted` or `in_review` with timestamp and non-secret identifier. Record `published` only after OpenAI approval and publisher release. + +--- + +## Plan self-review + +- Spec coverage: all design sections map to Tasks 1–11; external approval is separate from implementation completion. +- Placeholder scan: every task names exact files, interfaces, RED/GREEN commands, and commit boundaries. +- Type consistency: outcome, progress, event, runtime, evidence, receipt, transport, installer, and CLI names are introduced once and reused. +- Dependency direction: benchmark may consume stable production contracts; production never imports benchmark code. diff --git a/docs/superpowers/specs/2026-08-13-codex-plugin-earned-enforcement-design.md b/docs/superpowers/specs/2026-08-13-codex-plugin-earned-enforcement-design.md new file mode 100644 index 0000000..f497f46 --- /dev/null +++ b/docs/superpowers/specs/2026-08-13-codex-plugin-earned-enforcement-design.md @@ -0,0 +1,452 @@ +# MARGINAL Codex Plugin and Earned Enforcement Design + +**Status:** Approved product direction; implementation checkpoint +**Date:** 2026-08-13 +**Target milestone:** v0.3 — Codex Reference Integration + +## 1. Purpose + +Ship a production Codex integration that anyone can install and remove through native Codex plugin +workflows, while preserving MARGINAL's evidence-first standard. + +The differentiating product contract is **Earned Enforcement**: + +> MARGINAL starts as an observer. It earns the right to block tool actions for one repository only +> after it can prove that its own coverage, recommendations, false-stop review, and overhead meet a +> versioned local evidence gate. + +This is not a claim that MARGINAL is the first budget limiter, cost dashboard, loop detector, or +agent hook. Those categories already exist. The product distinction is the combination of: + +- marginal-value allocation rather than a hard cap alone; +- state/evidence-aware progress analysis; +- quality, false-stop, and governance-tax accounting; +- a capability label that refuses to overstate the native control surface; +- automatic demotion when the evidence contract no longer holds; +- native, reversible distribution with local-first telemetry. + +## 2. Goals + +1. Make the public Codex plugin the canonical installation surface. +2. Provide an immediate GitHub marketplace fallback before OpenAI review completes. +3. Keep global installation non-blocking in Shadow Mode. +4. Allow repository-scoped Tool Enforcement only through Earned Enforcement. +5. Make install, update, status, diagnostics, demotion, and uninstall idempotent and auditable. +6. Reuse the provider-neutral runtime and avoid a second policy implementation. +7. Persist hashes and structured decisions locally without storing prompts, source, raw commands, + or raw tool output by default. +8. Produce a plugin bundle, submission materials, positive/negative tests, and public legal/support + pages suitable for the universal Plugins Directory. +9. Preserve a thin engine boundary that can later support Claude Code without changing economic + policy. + +## 3. Non-goals + +- Claiming Full Compute Enforcement when Codex hooks do not cover every model or hosted-tool path. +- Parsing session transcripts as a stable API. +- Silently trusting hooks or using `--dangerously-bypass-hook-trust` for users. +- Reading or copying Codex authentication files. +- Uploading prompts, source, commands, outputs, or local telemetry. +- Automatically claiming token savings from tool-call suppression. +- Replacing the frozen benchmark adapter with unvalidated production assumptions. +- Solving multi-engine installation in v0.3; the architecture must permit it, but Codex is the + release target. + +## 4. Product capability label + +The v0.3 adapter reports **Tool Enforcement** when all required hooks are trusted and observed. + +It does not report Full Compute Enforcement because official Codex documentation says specialized +tool paths can opt out and hosted tools do not use the local hook path. The capability report lists +each supported event and tool family rather than collapsing support into one boolean. + +Required events: + +- `SessionStart` for runtime startup and capability attestation; +- `PreToolUse` for allow/recommend/deny decisions; +- `PostToolUse` for completion evidence and state correlation; +- `SessionEnd` for settlement, coverage summary, and clean shutdown. + +Optional later events include `Stop`, `SubagentStart`, and `SubagentStop`. They are not part of the +first enforcement gate. + +## 5. User experience + +### 5.1 Public directory + +Once OpenAI approves and the publisher releases it, users install MARGINAL from the universal +Plugins Directory shared by ChatGPT and Codex. The directory listing is the primary product route. + +Submission starts immediately when the implementation, legal pages, test cases, and final bundle +pass their gates. Public appearance cannot be represented as immediate because OpenAI approval is +an external review step. + +### 5.2 Immediate GitHub marketplace fallback + +Before directory approval, the supported one-line installation target is: + +```bash +codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal +``` + +The marketplace name in `.agents/plugins/marketplace.json` is `marginal`. Repeating the command is +safe: existing marketplace/plugin state is detected and upgraded rather than duplicated. + +Removal target: + +```bash +codex plugin remove marginal@marginal +``` + +Removing the marketplace itself is optional and separate because it may later distribute more +SignalLayer Labs plugins. + +### 5.3 Python management CLI + +The package also provides: + +```text +marginal install codex +marginal codex status +marginal codex doctor +marginal codex review +marginal codex promote +marginal codex demote +marginal uninstall codex +``` + +`marginal install codex` delegates plugin registration to the Codex CLI instead of editing Codex +configuration directly. It is a secondary automation path, not a competing installation system. + +The plugin remains functional without a separately installed wheel because its release artifact +contains a generated, dependency-free Python runtime. The CLI and plugin artifact are built from +the same source modules; generated plugin runtime files are never edited manually. + +### 5.4 Hook trust + +Codex requires review of non-managed command hooks. MARGINAL surfaces the exact `/hooks` review +step after installation and remains visibly inactive until trust is granted. It never bypasses this +security boundary. + +## 6. Runtime architecture + +```mermaid +flowchart TD + Directory["Universal directory or Git marketplace"] --> Plugin["MARGINAL plugin"] + Plugin --> HookConfig["Official lifecycle hooks"] + HookConfig --> Client["Small hook client"] + Client --> SessionRuntime["Per-session local runtime"] + SessionRuntime --> Adapter["Codex anti-corruption layer"] + Adapter --> UniversalRuntime + UniversalRuntime --> Treasury + Treasury --> Policy + Treasury --> Ledger["Hash-only local evidence"] + Ledger --> Promotion["Earned Enforcement evaluator"] + Promotion --> Shadow["Global Shadow Mode"] + Promotion --> Enforce["Repository Tool Enforcement"] +``` + +### 6.1 Source boundaries + +Production code lives under `src/marginal/integrations/codex/`: + +- `capabilities.py`: Codex version, feature, hook, and tool-family capability reporting; +- `events.py`: strict official hook input/output values; +- `normalization.py`: native event to Universal Agent Protocol translation; +- `outcomes.py`: success/failure/unknown classification without undocumented guesses; +- `runtime.py`: one session's adapter and Treasury lifecycle; +- `transport.py`: authenticated local client/runtime messages; +- `service.py`: per-session process startup, health, shutdown, and crash evidence; +- `evidence.py`: coverage counters, redacted ledger, checkpoints, and receipts; +- `promotion.py`: Earned Enforcement evaluation and automatic demotion; +- `installer.py`: Codex CLI detection and idempotent plugin operations; +- `commands.py`: Codex-specific CLI handlers. + +The generic CLI only dispatches to this package. The top-level `marginal` facade does not import +Codex integration modules. + +`benchmark/codex_adapter` keeps experiment orchestration, pinned task containers, and frozen run +records. It imports stable production event/normalization contracts when doing so does not change +the frozen scientific definition. Production never imports from `benchmark`. + +### 6.2 Plugin package + +The repository contains: + +```text +.agents/plugins/marketplace.json +plugins/marginal/ + .codex-plugin/plugin.json + hooks/hooks.json + skills/marginal/SKILL.md + scripts/marginal_hook.py + runtime/marginal_runtime.pyz + assets/ +``` + +The plugin runtime zipapp is reproducibly generated from selected `src/marginal` modules. CI fails +if the committed bundle and source tree differ. Release provenance records source commit, Python +version, manifest hash, and bundle SHA-256. + +### 6.3 Per-session service + +`SessionStart` launches one local service per Codex session. It binds only to loopback, selects an +ephemeral port, and authenticates every hook call with a random 256-bit token stored in a +user-private connection file under `PLUGIN_DATA`. + +The service: + +- owns the in-memory Treasury and pending-action map; +- serializes lifecycle mutations; +- writes an atomic checkpoint after every settled decision; +- can restore completed observation history after a safe restart; +- marks interrupted pending actions as unknown rather than successful; +- reports health and coverage independently from policy decisions; +- shuts down on `SessionEnd` and removes connection credentials. + +Loopback transport is chosen over Unix-only sockets so the same design works on macOS, Linux, +WSL, and native Windows. File permissions are restrictive where the platform supports POSIX modes; +Windows uses the current-user data directory and never places connection material in a repository. + +## 7. Event and outcome semantics + +### 7.1 Identity + +Each action uses: + +- Codex `session_id`, `turn_id`, and `tool_use_id` for lifecycle identity; +- normalized tool name and canonicalized input hash for semantic identity; +- repository state hash excluding `.git`, `.codex`, `.marginal`, caches, virtual environments, + plugin data, and generated runtime evidence; +- post-action evidence hash derived in memory from `tool_response`. + +Raw tool input and response are not persisted by default. + +### 7.2 Success is not completion + +Official `PostToolUse` is a completion signal and also runs after non-zero shell exits. Therefore +the adapter uses an explicit outcome enum: + +- `success`: supported structured evidence proves success; +- `failure`: supported structured evidence proves failure; +- `unknown`: Codex completed the handler but the supported contract cannot prove the result. + +Only `success` advances the existing `DiminishingReturnDetector`. `failure` uses failure +settlement; `unknown` releases or settles conservatively without advancing successful repetition +history. The adapter never treats all shell calls as successful and never classifies all shell calls +as failed. + +Version-pinned parsers may be added only with captured real fixtures for success, non-zero exit, +background completion, and transport failure. Undocumented prose parsing cannot enable +enforcement. + +### 7.3 No-progress recommendations + +A separate provider-neutral **No Progress** signal may recommend against a repeated completed +attempt when semantic input, repository state, and evidence all remain unchanged. It is distinct +from successful-action diminishing returns. + +No-progress signals begin as Shadow recommendations. They may become enforcement-eligible only +after their own reviewed evidence meets the promotion gate. This prevents a repeated failing test +or flaky check from silently being treated as waste. + +## 8. Earned Enforcement + +### 8.1 Scope + +Promotion is repository-scoped and stored outside the repository under `PLUGIN_DATA`, keyed by an +HMAC of the canonical repository identity. Installing the plugin never adds project files. + +### 8.2 Promotion receipt + +A versioned receipt contains: + +- repository pseudonym; +- Codex, plugin, adapter, policy, and estimator versions; +- hook and tool-family capability matrix; +- observation window and successful session count; +- hook-coverable calls, covered calls, gaps, and integration failures; +- recommendations by reason code; +- reviewed stop candidates and false stops; +- governance decision latency distribution; +- local governance tokens and USD when non-zero; +- unresolved or unknown outcomes; +- policy and evidence hashes; +- readiness status and machine-readable blocking reasons. + +### 8.3 Default gate + +The initial conservative gate requires all of the following: + +- at least 100 covered tool actions across at least five completed sessions; +- at least 99% coverage of hook-coverable local tool calls; +- no integration failures or unresolved reservations in the evaluation window; +- at least five intervention candidates, all manually reviewed; +- zero reviewed false stops in the window; +- p95 local decision latency no greater than 75 ms; +- no Codex/plugin/policy version change since the evidence window began; +- only action families with an observable outcome contract are enforcement-eligible. + +These thresholds are transparent safety defaults, not universal statistical proof. The receipt +states sample size and limitations. Configuration changes invalidate the receipt. + +### 8.4 Promotion and demotion + +`marginal codex promote` succeeds only with a ready receipt and records explicit user intent. +There is no silent auto-promotion. + +The runtime automatically demotes the repository to Shadow Mode when: + +- Codex, adapter, policy, or hook hashes change; +- coverage drops below the supported threshold; +- the local service crashes or a lifecycle mismatch occurs; +- a new false stop is recorded; +- the outcome contract becomes unknown for an enforced action family. + +Demotion never prevents Codex from continuing. It writes a visible reason and a new receipt. + +## 9. Installation transactions + +The installer performs read-only discovery before mutation: + +1. locate `codex` and record its exact version; +2. query stable feature flags and plugin commands; +3. validate the plugin/marketplace manifest and runtime hash; +4. inspect installed marketplace/plugin state through Codex JSON output; +5. execute the minimum required Codex command; +6. verify the installed plugin identity and enabled state; +7. run a local hook-client/service self-test without reading authentication data; +8. write an installation receipt. + +No direct edit to `~/.codex/config.toml`, `hooks.json`, or authentication files occurs in the normal +plugin path. If a future compatibility fallback must edit configuration, it requires an atomic +backup, exact ownership markers, rollback on failure, and explicit capability labeling. + +Uninstall delegates to `codex plugin remove`, verifies absence, and preserves local evidence by +default. `--purge-data` is a separate explicit destructive option. Reinstall and upgrade preserve +receipts but invalidate promotion when code or policy hashes change. + +## 10. Failure behavior + +- Shadow Mode fails open and records the coverage gap. +- Tool Enforcement also fails open on integration failure, immediately demotes to Shadow, and + emits a visible warning. MARGINAL is an efficiency governor, not a security boundary. +- Invalid or oversized hook input is rejected by the adapter, recorded without raw payload, and + causes demotion rather than a permanent Codex outage. +- A denied `PreToolUse` action never enters successful execution history. +- `PostToolUse` identity mismatch never settles a different reservation. +- Session shutdown marks remaining pending work unknown and reports it. + +## 11. Privacy and security + +- Runtime behavior is local and performs no network requests. +- No prompt, source, raw command, raw tool output, transcript, or credential is persisted by + default. +- Commands and responses are canonicalized and hashed in memory before redacted evidence is + written. +- `transcript_path` is never parsed because OpenAI does not define it as stable. +- Codex auth files and credential environment values are neither opened nor copied. +- Hook subprocess environments explicitly exclude credential variables where supported. +- Plugin data and connection tokens use user-private permissions. +- The service accepts authenticated loopback messages only, applies size/time limits, and uses + constant-time token comparison. +- Plugin hooks require normal Codex trust review; bypass flags are prohibited in user guidance. +- Release artifacts include provenance and checksums, and CI scans plugin and publication bundles + for secrets. + +## 12. Claude compatibility direction + +The domain policy, outcome enum, no-progress signal, promotion receipt, and evidence gate are +provider-neutral. The Codex plugin may use compatibility environment variables supplied by Codex, +but Codex-specific names remain inside the adapter. + +A later Claude Code package supplies a separate native manifest and hook translator pointing to the +same generated runtime. No Claude behavior is claimed or shipped as complete in v0.3. + +## 13. Verification strategy + +### Unit and contract tests + +- strict hook event validation and output shapes; +- canonical semantic/state/evidence hashing; +- success/failure/unknown settlement; +- no-progress separation from successful diminishing returns; +- pending identity and replay invariants; +- receipt thresholds, invalidation, promotion, and demotion; +- redaction and no-secret persistence; +- installer command planning and idempotency. + +### Integration tests + +- real Codex fixture capture for supported event types; +- trusted plugin install, list, update, remove, and reinstall against a temporary Codex home; +- SessionStart/service/PreToolUse/PostToolUse/SessionEnd lifecycle; +- concurrent hook calls and service crash recovery; +- macOS, Linux, Windows, and WSL command generation; +- plugin validation and marketplace ingestion; +- wheel/sdist install plus generated zipapp smoke. + +### Live acceptance + +On the pinned supported Codex version: + +1. install from the Git marketplace command; +2. review/trust the hook definition through the supported UI; +3. run a harmless session that exercises shell, edit, and local function tools; +4. prove Shadow Mode does not block; +5. inspect status, coverage, redaction, and latency; +6. exercise a synthetic ready receipt and one controlled denial; +7. verify automatic demotion after a version/hash mismatch; +8. uninstall and verify ordinary Codex operation remains intact. + +The repository test suite, Ruff, strict mypy, build, Twine, plugin validator, marketplace validator, +secret scan, and documentation/site checks must all pass before publication. + +## 14. Public catalog submission + +The repository ships a submission packet containing: + +- final plugin bundle and manifest metadata; +- public website, support, privacy policy, and terms URLs; +- concise capability and limitation language; +- at least five positive and three negative reviewer test cases; +- release notes and policy attestations checklist; +- local test evidence and artifact hashes. + +Submission is attempted through the OpenAI Platform organization with Apps Management write access +and a verified SignalLayer Labs identity. The repository may truthfully say **submitted** after the +portal accepts it, but may say **available in the universal directory** only after OpenAI approval +and publisher release. + +## 15. Documentation and claims + +README and site lead with the install/remove experience and Earned Enforcement contract only after +live verification passes. They must state: + +- Shadow Mode is global by default; +- promotion is repository-scoped and evidence-gated; +- capability is Tool Enforcement, not Full Compute Enforcement; +- raw content stays local and is not persisted by default; +- benchmark savings remain separate from installer validation; +- directory submission status is external and timestamped. + +No token-saving headline is added unless matched, verified, net evidence supports it. + +## 16. Acceptance criteria + +The v0.3 installation slice is complete when: + +- a validated plugin and marketplace are committed; +- the GitHub one-line install and one-command removal work on a clean Codex home; +- `marginal install codex`, status, doctor, promotion/demotion, and uninstall are tested; +- install is global Shadow Mode and makes no project-code change; +- a repository cannot enter verified Tool Enforcement without a ready receipt; +- capability/version changes automatically demote enforcement; +- official hooks execute through the production adapter with exact lifecycle coverage; +- raw commands/output/auth material do not appear in the evidence store; +- package and plugin share one generated core implementation; +- full verification and live smoke pass; +- README, site, roadmap, changelog, integration docs, privacy, terms, and support are current; +- the public-directory submission packet is complete and portal submission is attempted; +- external review status is reported exactly, without implying approval. + diff --git a/plugins/marginal/.codex-plugin/plugin.json b/plugins/marginal/.codex-plugin/plugin.json new file mode 100644 index 0000000..b5b6a66 --- /dev/null +++ b/plugins/marginal/.codex-plugin/plugin.json @@ -0,0 +1,19 @@ +{ + "name": "marginal", + "version": "0.3.0", + "description": "Evidence-gated compute governance for Codex, local-first and Shadow Mode by default.", + "author": { + "name": "SignalLayer Labs", + "url": "https://github.com/SignalLayerLabs/Marginal" + }, + "skills": "./skills/", + "interface": { + "displayName": "Marginal", + "shortDescription": "Earned enforcement for agent compute.", + "longDescription": "MARGINAL measures repeated Codex tool work locally, starts in Shadow Mode, and permits repository Tool Enforcement only after a versioned evidence gate proves coverage, reviewed false stops, and governance overhead.", + "developerName": "SignalLayer Labs", + "category": "Productivity", + "capabilities": [], + "defaultPrompt": "Use $marginal to inspect this repository's Shadow Mode evidence and explain whether Tool Enforcement is ready." + } +} diff --git a/plugins/marginal/hooks/hooks.json b/plugins/marginal/hooks/hooks.json new file mode 100644 index 0000000..5cee9db --- /dev/null +++ b/plugins/marginal/hooks/hooks.json @@ -0,0 +1,60 @@ +{ + "description": "Local-only MARGINAL observation and Earned Enforcement hooks.", + "hooks": { + "SessionStart": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 \"$PLUGIN_ROOT/scripts/marginal_hook.py\"", + "commandWindows": "py -3 \"%PLUGIN_ROOT%\\scripts\\marginal_hook.py\"", + "timeout": 10, + "statusMessage": "Starting MARGINAL Shadow Mode" + } + ] + } + ], + "PreToolUse": [ + { + "matcher": "Bash|apply_patch|MCP|Read|Write|Edit", + "hooks": [ + { + "type": "command", + "command": "python3 \"$PLUGIN_ROOT/scripts/marginal_hook.py\"", + "commandWindows": "py -3 \"%PLUGIN_ROOT%\\scripts\\marginal_hook.py\"", + "timeout": 5, + "statusMessage": "Measuring marginal value" + } + ] + } + ], + "PostToolUse": [ + { + "matcher": "Bash|apply_patch|MCP|Read|Write|Edit", + "hooks": [ + { + "type": "command", + "command": "python3 \"$PLUGIN_ROOT/scripts/marginal_hook.py\"", + "commandWindows": "py -3 \"%PLUGIN_ROOT%\\scripts\\marginal_hook.py\"", + "timeout": 5, + "statusMessage": "Recording redacted completion evidence" + } + ] + } + ], + "SessionEnd": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 \"$PLUGIN_ROOT/scripts/marginal_hook.py\"", + "commandWindows": "py -3 \"%PLUGIN_ROOT%\\scripts\\marginal_hook.py\"", + "timeout": 5, + "statusMessage": "Closing MARGINAL session" + } + ] + } + ] + } +} + diff --git a/plugins/marginal/runtime/marginal_runtime.pyz b/plugins/marginal/runtime/marginal_runtime.pyz new file mode 100644 index 0000000..acff7a2 Binary files /dev/null and b/plugins/marginal/runtime/marginal_runtime.pyz differ diff --git a/plugins/marginal/runtime/provenance.json b/plugins/marginal/runtime/provenance.json new file mode 100644 index 0000000..662644f --- /dev/null +++ b/plugins/marginal/runtime/provenance.json @@ -0,0 +1 @@ +{"builder":"scripts/build_codex_plugin.py","python_requires":">=3.10","schema_version":1,"sha256":"f148d51c8e00323637225479a2b26b75ea79255ae850ba3e3df0c65e41ecc630","source_hash":"908d1090899c672c8a19bf55268e9583ca3c3486b1f4b9dbc58815bf0027fa63"} diff --git a/plugins/marginal/scripts/marginal_hook.py b/plugins/marginal/scripts/marginal_hook.py new file mode 100644 index 0000000..a08a623 --- /dev/null +++ b/plugins/marginal/scripts/marginal_hook.py @@ -0,0 +1,31 @@ +#!/usr/bin/env python3 +"""Tiny dependency-free launcher for the generated MARGINAL runtime.""" + +from __future__ import annotations + +import os +import sys +from pathlib import Path + + +def main() -> int: + plugin_root = os.environ.get("PLUGIN_ROOT") + plugin_data = os.environ.get("PLUGIN_DATA") + if not plugin_root or not plugin_data: + return 0 + runtime = Path(plugin_root).resolve() / "runtime" / "marginal_runtime.pyz" + if not runtime.is_file(): + return 0 + environment = { + name: value + for name in ("PATH", "LANG", "LC_ALL", "SYSTEMROOT") + if (value := os.environ.get(name)) is not None + } + environment["PLUGIN_DATA"] = str(Path(plugin_data).resolve()) + environment["PLUGIN_ROOT"] = str(Path(plugin_root).resolve()) + os.execve(sys.executable, [sys.executable, str(runtime)], environment) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/marginal/skills/marginal/SKILL.md b/plugins/marginal/skills/marginal/SKILL.md new file mode 100644 index 0000000..eb9d121 --- /dev/null +++ b/plugins/marginal/skills/marginal/SKILL.md @@ -0,0 +1,40 @@ +--- +name: marginal +description: Use when inspecting, reviewing, promoting, demoting, or explaining MARGINAL governance in Codex, especially for repeated tool work, token-saving claims, and Earned Enforcement readiness. +--- + +# MARGINAL + +Treat compute as scarce and claims as evidence-bound. The plugin starts globally in Shadow Mode; +it may exercise repository-scoped Tool Enforcement only after a valid Earned Enforcement receipt +and explicit promotion. + +## Workflow + +1. Run `marginal codex status` before describing the active mode. +2. Run `marginal codex doctor` when hooks, coverage, or compatibility are uncertain. +3. Use `/hooks` to inspect and grant trust to the exact lifecycle commands. +4. Run `marginal codex review`, then label each local redacted candidate with + `--candidate HASH --verdict waste|helpful` before promotion. +5. Run `marginal codex promote` only when the evidence receipt is ready. +6. Run `marginal codex demote` whenever identity, coverage, outcome observability, or policy drifts. + +## Claims contract + +- Say **Tool Enforcement**, never Full Compute Enforcement. +- Describe recommendations as counterfactual until an enforced run measures them. +- Never claim token savings without a matched benchmark that reports quality and governance tax. +- Treat `PostToolUse` as completion, not success; prose-only outcomes remain unknown. +- Never read Codex auth files, prompts, source, raw commands, raw outputs, or transcripts for evidence. + +## Quick reference + +| Need | Command | +| --- | --- | +| Current mode | `marginal codex status` | +| Capability diagnosis | `marginal codex doctor` | +| Unreviewed evidence | `marginal codex review` | +| Evidence-gated enforcement | `marginal codex promote` | +| Immediate fail-open reset | `marginal codex demote` | + +If evidence is incomplete or contradictory, keep Shadow Mode and report the exact blocking reason. diff --git a/plugins/marginal/skills/marginal/agents/openai.yaml b/plugins/marginal/skills/marginal/agents/openai.yaml new file mode 100644 index 0000000..17d1dfe --- /dev/null +++ b/plugins/marginal/skills/marginal/agents/openai.yaml @@ -0,0 +1,4 @@ +interface: + display_name: "MARGINAL" + short_description: "Inspect evidence-gated Codex governance" + default_prompt: "Use $marginal to inspect this repository’s Shadow Mode evidence and explain whether Tool Enforcement is ready." diff --git a/pyproject.toml b/pyproject.toml index f337113..b97bcbe 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "marginal-ai" -version = "0.2.0" +version = "0.3.0" description = "Universal learning-loop and compute-governance foundation for economically disciplined AI agents." readme = "README.md" requires-python = ">=3.10" @@ -59,6 +59,7 @@ dev = [ "build>=1.2.2", "mypy>=1.17", "pytest>=8.3", + "PyYAML>=6.0", "jsonschema>=4.23", "ruff==0.16.2", "twine>=6.1", diff --git a/scripts/build_codex_plugin.py b/scripts/build_codex_plugin.py new file mode 100644 index 0000000..1844b8e --- /dev/null +++ b/scripts/build_codex_plugin.py @@ -0,0 +1,114 @@ +#!/usr/bin/env python3 +"""Reproducibly build the dependency-free MARGINAL Codex runtime zipapp.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import tempfile +import zipfile +from dataclasses import dataclass +from pathlib import Path + +_ZIP_TIMESTAMP = (2020, 1, 1, 0, 0, 0) +_MAIN = ( + b"from marginal.integrations.codex.service import hook_main\nraise SystemExit(hook_main())\n" +) + + +@dataclass(frozen=True, slots=True) +class PluginBuild: + zipapp: Path + source_hash: str + sha256: str + + +def _source_files(repo: Path) -> list[Path]: + package = repo / "src" / "marginal" + return sorted( + path + for path in package.rglob("*") + if (path.is_file() and "__pycache__" not in path.parts and path.suffix in {".py", ".json"}) + or path == package / "py.typed" + ) + + +def _source_hash(repo: Path, files: list[Path]) -> str: + digest = hashlib.sha256() + for path in files: + relative = path.relative_to(repo / "src").as_posix().encode("utf-8") + digest.update(relative + b"\0" + path.read_bytes() + b"\0") + digest.update(b"__main__.py\0" + _MAIN) + return digest.hexdigest() + + +def _write_archive(target: Path, repo: Path, files: list[Path]) -> None: + target.parent.mkdir(parents=True, exist_ok=True) + with zipfile.ZipFile( + target, + "w", + compression=zipfile.ZIP_STORED, + ) as archive: + entries = [("__main__.py", _MAIN)] + [ + (path.relative_to(repo / "src").as_posix(), path.read_bytes()) for path in files + ] + for name, content in sorted(entries): + info = zipfile.ZipInfo(name, date_time=_ZIP_TIMESTAMP) + info.compress_type = zipfile.ZIP_STORED + info.external_attr = 0o100644 << 16 + info.create_system = 3 + archive.writestr(info, content, compress_type=zipfile.ZIP_STORED) + + +def build_plugin_runtime(repo: str | Path, *, output_dir: str | Path) -> PluginBuild: + root = Path(repo).resolve() + destination = Path(output_dir).resolve() + files = _source_files(root) + source_hash = _source_hash(root, files) + zipapp = destination / "marginal_runtime.pyz" + _write_archive(zipapp, root, files) + archive_hash = hashlib.sha256(zipapp.read_bytes()).hexdigest() + return PluginBuild(zipapp=zipapp, source_hash=source_hash, sha256=archive_hash) + + +def _write_provenance(runtime_dir: Path, build: PluginBuild) -> None: + payload = { + "schema_version": 1, + "builder": "scripts/build_codex_plugin.py", + "python_requires": ">=3.10", + "source_hash": build.source_hash, + "sha256": build.sha256, + } + (runtime_dir / "provenance.json").write_text( + json.dumps(payload, sort_keys=True, separators=(",", ":")) + "\n", + encoding="utf-8", + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--check", action="store_true") + args = parser.parse_args() + repo = Path(__file__).resolve().parents[1] + committed_dir = repo / "plugins" / "marginal" / "runtime" + if args.check: + with tempfile.TemporaryDirectory(prefix="marginal-plugin-check-") as temporary: + build = build_plugin_runtime(repo, output_dir=temporary) + committed = committed_dir / "marginal_runtime.pyz" + provenance = json.loads((committed_dir / "provenance.json").read_text(encoding="utf-8")) + if not committed.exists() or committed.read_bytes() != build.zipapp.read_bytes(): + raise SystemExit("committed Codex runtime is stale") + if provenance.get("sha256") != build.sha256: + raise SystemExit("Codex runtime provenance is stale") + if provenance.get("source_hash") != build.source_hash: + raise SystemExit("Codex runtime source hash is stale") + return 0 + build = build_plugin_runtime(repo, output_dir=committed_dir) + _write_provenance(committed_dir, build) + print(f"built {build.zipapp} ({build.sha256})") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/smoke_codex_plugin.py b/scripts/smoke_codex_plugin.py new file mode 100644 index 0000000..94ce9e6 --- /dev/null +++ b/scripts/smoke_codex_plugin.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Credential-free install, hook lifecycle, privacy, and removal acceptance test.""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import sys +import time +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + + +@dataclass(frozen=True, slots=True) +class CodexPluginSmokeResult: + installed: bool + shadow_block_count: int + hook_coverage: float + evidence_records: int + completed_sessions: int + raw_secret_occurrences: int + removed: bool + codex_version: str + + +def _run( + args: list[str], + *, + environment: dict[str, str], + cwd: Path, + input_text: str | None = None, +) -> subprocess.CompletedProcess[str]: + completed = subprocess.run( + args, + cwd=cwd, + env=environment, + input=input_text, + check=False, + capture_output=True, + text=True, + timeout=30, + ) + if completed.returncode != 0: + raise RuntimeError(f"command failed ({completed.returncode}): {' '.join(args[:4])}") + return completed + + +def _initialize_repository(path: Path, environment: dict[str, str]) -> None: + path.mkdir(parents=True) + for args in ( + ["git", "init", "-q"], + ["git", "config", "user.email", "smoke@example.com"], + ["git", "config", "user.name", "MARGINAL Smoke"], + ): + _run(args, environment=environment, cwd=path) + (path / "tracked.txt").write_text("initial\n", encoding="utf-8") + _run(["git", "add", "tracked.txt"], environment=environment, cwd=path) + _run(["git", "commit", "-qm", "initial"], environment=environment, cwd=path) + + +def _hook_payloads(workspace: Path, secret: str) -> list[dict[str, Any]]: + common: dict[str, Any] = { + "session_id": "smoke-session", + "transcript_path": None, + "cwd": str(workspace), + "model": "smoke-model", + "permission_mode": "default", + } + tool = { + **common, + "turn_id": "smoke-turn", + "tool_name": "Bash", + "tool_use_id": "smoke-call", + "tool_input": {"command": f"echo {secret}", "description": secret}, + } + return [ + {**common, "hook_event_name": "SessionStart", "source": "startup"}, + {**tool, "hook_event_name": "PreToolUse"}, + {**tool, "hook_event_name": "PostToolUse", "tool_response": {"exit_code": 0}}, + {**common, "hook_event_name": "SessionEnd", "reason": "other"}, + ] + + +def _count_secret(root: Path, secret: str) -> int: + occurrences = 0 + if not root.exists(): + return 0 + marker = secret.encode("utf-8") + for path in root.rglob("*"): + if path.is_file(): + occurrences += path.read_bytes().count(marker) + return occurrences + + +def _read_evidence(plugin_data: Path) -> list[dict[str, Any]]: + records: list[dict[str, Any]] = [] + for path in (plugin_data / "evidence").glob("*/evidence.jsonl"): + for line in path.read_text(encoding="utf-8").splitlines(): + payload = json.loads(line) + if isinstance(payload, dict): + records.append(payload) + return records + + +def smoke_plugin( + *, + codex: Path, + isolation_root: Path, + marketplace: Path, +) -> CodexPluginSmokeResult: + """Run the public install/remove path without using the caller's Codex home.""" + + root = isolation_root.resolve() + home = root / "home" + codex_home = root / "codex" + plugin_data = root / "plugin-data" + workspace = root / "workspace" + for directory in (home, codex_home, plugin_data): + directory.mkdir(parents=True, exist_ok=True) + environment = { + "PATH": os.environ.get("PATH", ""), + "HOME": str(home), + "CODEX_HOME": str(codex_home), + "LANG": "C", + "LC_ALL": "C", + } + _initialize_repository(workspace, environment) + version = _run([str(codex), "--version"], environment=environment, cwd=root).stdout.strip() + _run( + [str(codex), "plugin", "marketplace", "add", str(marketplace), "--json"], + environment=environment, + cwd=root, + ) + add = _run( + [str(codex), "plugin", "add", "marginal@marginal", "--json"], + environment=environment, + cwd=root, + ) + installed_payload = json.loads(add.stdout) + plugin_root = Path(installed_payload["installedPath"]).resolve() + hook_script = plugin_root / "scripts" / "marginal_hook.py" + hook_environment = { + "PATH": environment["PATH"], + "LANG": "C", + "LC_ALL": "C", + "PLUGIN_ROOT": str(plugin_root), + "PLUGIN_DATA": str(plugin_data), + } + + secret = "MARGINAL_SMOKE_SECRET_7fcd98" + completed_hooks = 0 + shadow_blocks = 0 + removed = False + try: + for payload in _hook_payloads(workspace, secret): + result = _run( + [sys.executable, str(hook_script)], + environment=hook_environment, + cwd=workspace, + input_text=json.dumps(payload), + ) + completed_hooks += 1 + if payload["hook_event_name"] == "PreToolUse" and result.stdout.strip(): + hook_output = json.loads(result.stdout) + decision = hook_output.get("hookSpecificOutput", {}).get("permissionDecision") + shadow_blocks += int(decision == "deny") + deadline = time.monotonic() + 2.0 + sessions_root = plugin_data / "sessions" + while list(sessions_root.glob("*.json")) and time.monotonic() < deadline: + time.sleep(0.02) + finally: + remove = _run( + [str(codex), "plugin", "remove", "marginal@marginal", "--json"], + environment=environment, + cwd=root, + ) + removed = json.loads(remove.stdout).get("pluginId") == "marginal@marginal" + + evidence = _read_evidence(plugin_data) + decisions = [record for record in evidence if record.get("event") == "decision"] + coverable = sum(record.get("coverable") is True for record in decisions) + covered = sum(record.get("covered") is True for record in decisions) + if completed_hooks != 4: + raise RuntimeError("direct hook lifecycle did not complete") + + return CodexPluginSmokeResult( + installed=True, + shadow_block_count=shadow_blocks, + hook_coverage=covered / coverable if coverable else 0.0, + evidence_records=len(evidence), + completed_sessions=len( + { + str(record.get("session_hash")) + for record in evidence + if record.get("event") == "session_end" and record.get("session_hash") + } + ), + raw_secret_occurrences=_count_secret(plugin_data, secret), + removed=removed, + codex_version=version, + ) + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--codex", type=Path, default=Path("codex")) + parser.add_argument("--isolation-root", type=Path, required=True) + parser.add_argument("--marketplace", type=Path, default=Path.cwd()) + parser.add_argument("--json", action="store_true") + args = parser.parse_args() + result = smoke_plugin( + codex=args.codex.resolve(), + isolation_root=args.isolation_root.resolve(), + marketplace=args.marketplace.resolve(), + ) + if args.json: + print(json.dumps(asdict(result), sort_keys=True)) + else: + print(result) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/site/index.html b/site/index.html index 9bd8d98..26eaac7 100644 --- a/site/index.html +++ b/site/index.html @@ -46,6 +46,12 @@

Measured OFF vs ON. Correctness first.

Exploratory 3-task smoke, one paired run per task. Codex CLI 0.147.0 with GPT-5.6-sol, identical prompts and limits, verified in official SWE-bench Lite task environments on Modal.

+
+
Native Codex plugin · Shadow Mode by defaultEarned Enforcement, one-command install.
+ codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin add marginal@marginal + codex plugin remove marginal@marginal +

Tool Enforcement, never an overstated Full Compute Enforcement claim. Repository blocking must earn a local evidence receipt and demotes automatically on drift. Universal directory review is pending; the Git marketplace works now.

+
Verified quality0/3 → 0/3OFF → ON resolved
Effective tokens24.93% fewer1,098,747 → 824,839
@@ -180,7 +186,7 @@

What if GPT-5.7 — or any future model — is already efficient?

Build the evidence first

Observe. Measure. Let intervention earn enforcement.

- Install v0.2 + Install for Codex Challenge the project
@@ -188,8 +194,8 @@

Observe. Measure. Let intervention earn enforcement.

diff --git a/site/privacy.html b/site/privacy.html new file mode 100644 index 0000000..f09fa5d --- /dev/null +++ b/site/privacy.html @@ -0,0 +1,4 @@ + +MARGINAL Privacy +

Effective 2026-08-13

MARGINAL Privacy

MARGINAL is local-first. The Codex plugin makes no network request and does not operate a SignalLayer Labs telemetry service.

Raw prompts, source, commands, tool output, transcripts, authentication files and credential environment values are not persisted by default. Local evidence is limited to hashes, structured decisions, coverage, outcomes, latency, review labels and receipts in user-private plugin data.

Uninstall preserves evidence; explicit marginal uninstall codex --purge-data --yes removes it. Third-party platforms retain their own privacy terms.

Complete privacy notice →

+ diff --git a/site/sitemap.xml b/site/sitemap.xml index e9beff8..2b021f5 100644 --- a/site/sitemap.xml +++ b/site/sitemap.xml @@ -1,2 +1,2 @@ -https://signallayerlabs.github.io/Marginal/weekly1.0https://signallayerlabs.github.io/Marginal/demo/monthly0.7 +https://signallayerlabs.github.io/Marginal/weekly1.0https://signallayerlabs.github.io/Marginal/privacy.htmlmonthly0.6https://signallayerlabs.github.io/Marginal/terms.htmlmonthly0.6https://signallayerlabs.github.io/Marginal/support.htmlmonthly0.6https://signallayerlabs.github.io/Marginal/demo/monthly0.7 diff --git a/site/styles.css b/site/styles.css index d20d793..7359a62 100644 --- a/site/styles.css +++ b/site/styles.css @@ -49,6 +49,13 @@ code { font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace; co .benchmark-heading { max-width: 900px; } .benchmark-heading h1 { font-size: clamp(2.5rem, 5vw, 4.8rem); } .benchmark-heading > p:last-child { color: var(--muted); max-width: 820px; font-size: 1.05rem; } +.install-card { display: grid; gap: .75rem; margin-top: 1.5rem; padding: 1.2rem; border: 1px solid var(--accent); border-radius: 14px; background: var(--accent-soft); } +.install-card > div { display: grid; } +.install-card span { color: var(--accent); font-size: .76rem; font-weight: 820; letter-spacing: .08em; text-transform: uppercase; } +.install-card strong { font-size: 1.2rem; } +.install-card code { display: block; overflow-x: auto; padding: .75rem; border: 1px solid var(--line); border-radius: 8px; background: #05080c; white-space: nowrap; } +.install-card p { margin: 0; color: var(--muted); font-size: .9rem; } +.install-card b { color: var(--text); } .benchmark-metrics { display: grid; grid-template-columns: repeat(5, 1fr); gap: .75rem; margin-top: 2.4rem; } .benchmark-metrics article { display: grid; gap: .3rem; min-height: 145px; padding: 1.15rem; border: 1px solid var(--line); border-radius: 14px; background: rgba(16,21,29,.82); } .benchmark-metrics article > span { color: var(--muted); font-size: .73rem; font-weight: 800; letter-spacing: .07em; text-transform: uppercase; } diff --git a/site/support.html b/site/support.html new file mode 100644 index 0000000..4872446 --- /dev/null +++ b/site/support.html @@ -0,0 +1,4 @@ + +MARGINAL Support +

Support

Get help without sharing secrets.

Use GitHub Issues for reproducible bugs and GitHub Discussions for design questions. Security and private privacy reports use GitHub private vulnerability reporting.

Include redacted marginal codex doctor output, versions and operating system. Never attach auth files, prompts, source, commands, tool output, transcripts, tokens or connection files.

Complete support policy →

+ diff --git a/site/terms.html b/site/terms.html new file mode 100644 index 0000000..568763f --- /dev/null +++ b/site/terms.html @@ -0,0 +1,4 @@ + +MARGINAL Terms +

Effective 2026-08-13

MARGINAL Terms

MARGINAL is Apache-2.0 experimental developer infrastructure with no guarantee of token savings, task success, uninterrupted operation or fitness for a particular purpose.

Shadow Mode is the default. Tool Enforcement is not a security boundary and fails open. Users must review hooks, policies, stop candidates, evidence and generated work.

The n=3 Codex smoke returned pass_through and is not a general performance claim.

Complete terms →

+ diff --git a/src/marginal/__init__.py b/src/marginal/__init__.py index ef38b5c..f9fe4f1 100644 --- a/src/marginal/__init__.py +++ b/src/marginal/__init__.py @@ -137,4 +137,4 @@ "validate_safe_telemetry_record", ] -__version__ = "0.2.0" +__version__ = "0.3.0" diff --git a/src/marginal/cli.py b/src/marginal/cli.py index 456ca81..5fa1f66 100644 --- a/src/marginal/cli.py +++ b/src/marginal/cli.py @@ -141,6 +141,27 @@ def _build_parser() -> argparse.ArgumentParser: help="maximum reviewed false-stop rate allowed for a supported intervention", ) public_eval.add_argument("--seed", type=int, default=42) + + install_parser = subparsers.add_parser("install", help="install a native integration") + install_parser.add_argument("target", choices=["codex"]) + install_parser.add_argument("--repository", default="SignalLayerLabs/Marginal") + install_parser.add_argument("--ref", default="main") + install_parser.add_argument("--json", action="store_true", dest="as_json") + + uninstall_parser = subparsers.add_parser("uninstall", help="remove a native integration") + uninstall_parser.add_argument("target", choices=["codex"]) + uninstall_parser.add_argument("--purge-data", action="store_true") + uninstall_parser.add_argument("--yes", action="store_true") + uninstall_parser.add_argument("--data-dir", type=Path) + uninstall_parser.add_argument("--json", action="store_true", dest="as_json") + + codex = subparsers.add_parser("codex", help="manage the Codex integration") + codex.add_argument("codex_command", choices=["status", "doctor", "review", "promote", "demote"]) + codex.add_argument("--data-dir", type=Path) + codex.add_argument("--workspace", type=Path) + codex.add_argument("--candidate") + codex.add_argument("--verdict", choices=["helpful", "waste"]) + codex.add_argument("--json", action="store_true", dest="as_json") return parser @@ -148,6 +169,45 @@ def main(argv: Sequence[str] | None = None) -> int: parser = _build_parser() args = parser.parse_args(argv) + if args.command == "install": + from .integrations.codex.installer import install + + result = install(repository=args.repository, ref=args.ref) + payload = result.to_dict() + if args.as_json: + print(json.dumps(payload, sort_keys=True)) + else: + print(result.message or result.error_code) + return 0 if result.installed else 1 + + if args.command == "uninstall": + from .integrations.codex.commands import default_data_dir, purge_data + from .integrations.codex.installer import uninstall + + if args.purge_data and not args.yes: + print("--purge-data requires --yes", file=sys.stderr) + return 2 + result = uninstall() + if args.purge_data and not result.installed: + purge_data(args.data_dir or default_data_dir(), confirmed=True) + if args.as_json: + print(json.dumps(result.to_dict(), sort_keys=True)) + else: + print(result.message or result.error_code) + return 0 if not result.installed else 1 + + if args.command == "codex": + from .integrations.codex.commands import codex_command + + return codex_command( + args.codex_command, + data_dir=args.data_dir, + workspace=args.workspace, + candidate=args.candidate, + verdict=args.verdict, + as_json=args.as_json, + ) + if args.command == "ledger-export": from .ledger import export_decision_ledger diff --git a/src/marginal/controls/__init__.py b/src/marginal/controls/__init__.py index b341b57..6a8222c 100644 --- a/src/marginal/controls/__init__.py +++ b/src/marginal/controls/__init__.py @@ -6,10 +6,20 @@ DiminishingReturnSignal, ) from .governance import GovernanceTracker +from .progress import ( + ActionOutcomeStatus, + NoProgressConfig, + NoProgressDetector, + NoProgressSignal, +) __all__ = [ + "ActionOutcomeStatus", "DiminishingReturnConfig", "DiminishingReturnDetector", "DiminishingReturnSignal", "GovernanceTracker", + "NoProgressConfig", + "NoProgressDetector", + "NoProgressSignal", ] diff --git a/src/marginal/controls/progress.py b/src/marginal/controls/progress.py new file mode 100644 index 0000000..8a5e306 --- /dev/null +++ b/src/marginal/controls/progress.py @@ -0,0 +1,161 @@ +"""Provider-neutral evidence-invariant progress detection.""" + +from __future__ import annotations + +from dataclasses import dataclass +from enum import Enum + + +class ActionOutcomeStatus(str, Enum): + """What an adapter can prove about a completed action.""" + + SUCCESS = "success" + FAILURE = "failure" + UNKNOWN = "unknown" + + @classmethod + def parse(cls, value: ActionOutcomeStatus | str) -> ActionOutcomeStatus: + if isinstance(value, cls): + return value + try: + return cls(str(value).strip().lower()) + except ValueError as exc: + raise ValueError(f"unknown action outcome status: {value!r}") from exc + + +@dataclass(frozen=True, slots=True) +class NoProgressConfig: + """Threshold for repeated completions with identical state and evidence.""" + + max_same_evidence_completions: int = 2 + + def __post_init__(self) -> None: + value = self.max_same_evidence_completions + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError("max_same_evidence_completions must be an integer") + if value < 1: + raise ValueError("max_same_evidence_completions must be at least 1") + + +@dataclass(frozen=True, slots=True) +class NoProgressSignal: + """Explain whether unchanged completion evidence warrants a stop recommendation.""" + + semantic_key: str + same_evidence_completions: int + should_recommend_stop: bool + enforcement_eligible: bool + reason_code: str + reason: str + + +@dataclass(slots=True) +class _ProgressObservation: + state_hash: str + evidence_hash: str + completions: int + all_successful: bool + + +class NoProgressDetector: + """Track evidence-invariant completions without equating completion with success.""" + + def __init__(self, config: NoProgressConfig | None = None) -> None: + self.config = config or NoProgressConfig() + self._observations: dict[str, _ProgressObservation] = {} + + def evaluate( + self, + semantic_key: str, + state_hash: str, + evidence_hash: str, + ) -> NoProgressSignal: + semantic_key = _required_or_empty(semantic_key, "semantic_key") + state_hash = _required_or_empty(state_hash, "state_hash") + evidence_hash = _required_or_empty(evidence_hash, "evidence_hash") + if not semantic_key or not state_hash or not evidence_hash: + return NoProgressSignal( + semantic_key=semantic_key, + same_evidence_completions=0, + should_recommend_stop=False, + enforcement_eligible=False, + reason_code="NO_PROGRESS_UNOBSERVABLE", + reason="semantic identity, state, or completion evidence is unavailable", + ) + + previous = self._observations.get(semantic_key) + same_observation = ( + previous is not None + and previous.state_hash == state_hash + and previous.evidence_hash == evidence_hash + ) + if not same_observation or previous is None: + return NoProgressSignal( + semantic_key=semantic_key, + same_evidence_completions=0, + should_recommend_stop=False, + enforcement_eligible=False, + reason_code="NO_PROGRESS_CLEAR", + reason="state or completion evidence changed", + ) + + completions = previous.completions + should_stop = completions >= self.config.max_same_evidence_completions + enforcement_eligible = should_stop and previous.all_successful + if enforcement_eligible: + reason_code = "NO_PROGRESS_ENFORCEMENT_ELIGIBLE" + reason = "successful completions repeated without state or evidence change" + elif should_stop: + reason_code = "NO_PROGRESS_RECOMMENDED_UNKNOWN" + reason = "completions repeated without progress, but success is not proven" + else: + reason_code = "NO_PROGRESS_OBSERVED" + reason = "unchanged completion evidence remains below the stop threshold" + return NoProgressSignal( + semantic_key=semantic_key, + same_evidence_completions=completions, + should_recommend_stop=should_stop, + enforcement_eligible=enforcement_eligible, + reason_code=reason_code, + reason=reason, + ) + + def observe( + self, + semantic_key: str, + state_hash: str, + evidence_hash: str, + outcome: ActionOutcomeStatus | str, + ) -> None: + semantic_key = _required_or_empty(semantic_key, "semantic_key") + state_hash = _required_or_empty(state_hash, "state_hash") + evidence_hash = _required_or_empty(evidence_hash, "evidence_hash") + normalized_outcome = ActionOutcomeStatus.parse(outcome) + if not semantic_key or not state_hash or not evidence_hash: + return + + previous = self._observations.get(semantic_key) + same_observation = ( + previous is not None + and previous.state_hash == state_hash + and previous.evidence_hash == evidence_hash + ) + completions = previous.completions + 1 if same_observation and previous else 1 + all_successful = normalized_outcome is ActionOutcomeStatus.SUCCESS + if same_observation and previous is not None: + all_successful = previous.all_successful and all_successful + self._observations[semantic_key] = _ProgressObservation( + state_hash=state_hash, + evidence_hash=evidence_hash, + completions=completions, + all_successful=all_successful, + ) + + def reset(self) -> None: + self._observations.clear() + + +def _required_or_empty(value: str, name: str) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a string") + return value.strip() diff --git a/src/marginal/integrations/__init__.py b/src/marginal/integrations/__init__.py new file mode 100644 index 0000000..1a3169f --- /dev/null +++ b/src/marginal/integrations/__init__.py @@ -0,0 +1 @@ +"""Provider-specific adapters kept outside MARGINAL's policy core.""" diff --git a/src/marginal/integrations/codex/__init__.py b/src/marginal/integrations/codex/__init__.py new file mode 100644 index 0000000..b7d7204 --- /dev/null +++ b/src/marginal/integrations/codex/__init__.py @@ -0,0 +1,19 @@ +"""Codex hook integration for privacy-safe tool governance.""" + +from .events import PostToolUseEvent, PreToolUseEvent, SessionEvent, parse_hook_event +from .normalization import normalize_pre_tool_use +from .outcomes import classify_tool_outcome +from .runtime import CodexIntegrationError, CodexSessionRuntime +from .state import workspace_state_hash + +__all__ = [ + "CodexIntegrationError", + "CodexSessionRuntime", + "PostToolUseEvent", + "PreToolUseEvent", + "SessionEvent", + "classify_tool_outcome", + "normalize_pre_tool_use", + "parse_hook_event", + "workspace_state_hash", +] diff --git a/src/marginal/integrations/codex/commands.py b/src/marginal/integrations/codex/commands.py new file mode 100644 index 0000000..cb51685 --- /dev/null +++ b/src/marginal/integrations/codex/commands.py @@ -0,0 +1,209 @@ +"""User-facing Codex integration management commands.""" + +from __future__ import annotations + +import json +import os +import shutil +from pathlib import Path +from typing import Any + +from .evidence import EvidenceStore, summarize_evidence +from .identity import current_promotion_identity +from .installer import inspect_codex +from .promotion import ( + PromotionCriteria, + activate_enforcement, + demote_enforcement, + evaluate_promotion, + write_promotion_receipt, +) +from .service import read_mode + + +def default_data_dir() -> Path: + plugin_data = os.environ.get("PLUGIN_DATA") + if plugin_data: + return Path(plugin_data).resolve() + return Path.home() / ".local" / "share" / "marginal" / "codex" + + +def _state_path(data_dir: Path) -> Path: + return data_dir / "state.json" + + +def _read_state(data_dir: Path) -> dict[str, Any]: + path = _state_path(data_dir) + if not path.exists(): + return { + "schema_version": 1, + "mode": "shadow", + "capability": "Tool Enforcement", + "reason": "Earned Enforcement evidence not yet promoted", + } + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError("Codex state must be a JSON object") + return payload + + +def _write_state(data_dir: Path, payload: dict[str, Any]) -> None: + data_dir.mkdir(parents=True, exist_ok=True, mode=0o700) + path = _state_path(data_dir) + temporary = path.with_suffix(".tmp") + temporary.write_text( + json.dumps(payload, sort_keys=True, separators=(",", ":")) + "\n", + encoding="utf-8", + ) + if os.name == "posix": + temporary.chmod(0o600) + os.replace(temporary, path) + + +def _emit(payload: dict[str, Any], *, as_json: bool) -> None: + if as_json: + print(json.dumps(payload, sort_keys=True)) + else: + for key, value in payload.items(): + print(f"{key}: {value}") + + +def codex_command( + command: str, + *, + data_dir: str | Path | None = None, + workspace: str | Path | None = None, + candidate: str | None = None, + verdict: str | None = None, + as_json: bool = False, +) -> int: + root = Path(data_dir).resolve() if data_dir is not None else default_data_dir() + selected_workspace = Path(workspace).resolve() if workspace is not None else Path.cwd() + identity = current_promotion_identity(selected_workspace) + evidence_store = EvidenceStore(root / "evidence" / identity.repository_hash) + payload: dict[str, Any] + if command == "status": + state = read_mode(root, repository_hash=identity.repository_hash) + _emit( + { + **state, + "capability": "Tool Enforcement", + "repository_hash": identity.repository_hash, + }, + as_json=as_json, + ) + return 0 + if command == "doctor": + _emit(inspect_codex().to_dict(), as_json=as_json) + return 0 + if command == "review": + records = evidence_store.read_all() + candidates = { + str(record["action_hash"]) + for record in records + if record.get("event") == "decision" + and record.get("recommended_stop") is True + and isinstance(record.get("action_hash"), str) + } + already_reviewed = { + str(record["action_hash"]) + for record in records + if record.get("reviewed") is True and isinstance(record.get("action_hash"), str) + } + if candidate is None and verdict is None: + payload = { + "review_command": "/hooks", + "message": "Review hook trust, then label each redacted candidate explicitly.", + "unreviewed_candidates": sorted(candidates - already_reviewed), + } + _emit(payload, as_json=as_json) + return 0 + if candidate not in candidates or verdict not in {"helpful", "waste"}: + _emit( + { + "error_code": "INVALID_REVIEW", + "message": ( + "Candidate must be identified only by its hash; " + "verdict is helpful or waste." + ), + }, + as_json=as_json, + ) + return 2 + if candidate in already_reviewed: + _emit( + {"error_code": "ALREADY_REVIEWED", "candidate": candidate}, + as_json=as_json, + ) + return 2 + evidence_store.append( + { + "schema_version": 1, + "event": "review", + "action_hash": candidate, + "reviewed": True, + "false_stop": verdict == "helpful", + } + ) + if verdict == "helpful": + demote_enforcement( + root, + repository_hash=identity.repository_hash, + reason="FALSE_STOP_REVIEWED", + ) + evidence_store.start_new_window(reason_code="FALSE_STOP_REVIEWED") + payload = { + "candidate": candidate, + "reviewed": True, + "false_stop": verdict == "helpful", + } + _emit(payload, as_json=as_json) + return 0 + if command == "demote": + demote_enforcement( + root, + repository_hash=identity.repository_hash, + reason="EXPLICIT_USER_DEMOTION", + ) + payload = { + "schema_version": 1, + "mode": "shadow", + "capability": "Tool Enforcement", + "reason": "Explicit user demotion", + "repository_hash": identity.repository_hash, + } + _emit(payload, as_json=as_json) + return 0 + if command == "promote": + summary = summarize_evidence(evidence_store.read_all()) + receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + write_promotion_receipt(root, receipt) + if not receipt.is_ready: + payload = { + "mode": "shadow", + "error_code": "EVIDENCE_NOT_READY", + "blocking_reasons": list(receipt.blocking_reasons), + "receipt_hash": receipt.receipt_hash, + } + _emit(payload, as_json=as_json) + return 2 + activate_enforcement(root, receipt) + payload = { + "schema_version": 1, + "mode": "enforce", + "capability": "Tool Enforcement", + "receipt_hash": receipt.receipt_hash, + "reason": "Explicit promotion with a ready evidence receipt", + } + _emit(payload, as_json=as_json) + return 0 + raise ValueError(f"unsupported Codex command: {command}") + + +def purge_data(data_dir: str | Path, *, confirmed: bool) -> bool: + if not confirmed: + return False + root = Path(data_dir).resolve() + if root.exists(): + shutil.rmtree(root) + return True diff --git a/src/marginal/integrations/codex/events.py b/src/marginal/integrations/codex/events.py new file mode 100644 index 0000000..509a2f9 --- /dev/null +++ b/src/marginal/integrations/codex/events.py @@ -0,0 +1,148 @@ +"""Strict, minimal value objects for the supported Codex hook lifecycle.""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + + +@dataclass(frozen=True, slots=True) +class SessionEvent: + session_id: str + cwd: str + hook_event_name: str + model: str + permission_mode: str + transcript_path: str | None = None + source: str | None = None + reason: str | None = None + + +@dataclass(frozen=True, slots=True) +class PreToolUseEvent: + session_id: str + cwd: str + hook_event_name: str + model: str + permission_mode: str + turn_id: str + tool_name: str + tool_use_id: str + tool_input: Mapping[str, Any] + transcript_path: str | None = None + + +@dataclass(frozen=True, slots=True) +class PostToolUseEvent: + session_id: str + cwd: str + hook_event_name: str + model: str + permission_mode: str + turn_id: str + tool_name: str + tool_use_id: str + tool_input: Mapping[str, Any] + tool_response: Any + transcript_path: str | None = None + + +CodexHookEvent = SessionEvent | PreToolUseEvent | PostToolUseEvent + + +def _required_text(payload: Mapping[str, Any], name: str) -> str: + value = payload.get(name) + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{name} must be a non-empty string") + return value + + +def _optional_text(payload: Mapping[str, Any], name: str) -> str | None: + value = payload.get(name) + if value is None: + return None + if not isinstance(value, str): + raise ValueError(f"{name} must be a string or null") + return value + + +def _common(payload: Mapping[str, Any]) -> dict[str, Any]: + return { + "session_id": _required_text(payload, "session_id"), + "cwd": _required_text(payload, "cwd"), + "hook_event_name": _required_text(payload, "hook_event_name"), + "model": _required_text(payload, "model"), + "permission_mode": _required_text(payload, "permission_mode"), + "transcript_path": _optional_text(payload, "transcript_path"), + } + + +def _tool_fields(payload: Mapping[str, Any]) -> dict[str, Any]: + tool_input = payload.get("tool_input") + if not isinstance(tool_input, Mapping): + raise ValueError("tool_input must be a mapping") + return { + "turn_id": _required_text(payload, "turn_id"), + "tool_name": _required_text(payload, "tool_name"), + "tool_use_id": _required_text(payload, "tool_use_id"), + "tool_input": dict(tool_input), + } + + +def parse_hook_event(payload: Mapping[str, Any]) -> CodexHookEvent: + """Parse only hook events for which MARGINAL has an explicit contract.""" + + if not isinstance(payload, Mapping): + raise TypeError("Codex hook payload must be a mapping") + name = _required_text(payload, "hook_event_name") + common = _common(payload) + if name in {"SessionStart", "SessionEnd"}: + return SessionEvent( + **common, + source=_optional_text(payload, "source"), + reason=_optional_text(payload, "reason"), + ) + if name == "PreToolUse": + return PreToolUseEvent(**common, **_tool_fields(payload)) + if name == "PostToolUse": + if "tool_response" not in payload: + raise ValueError("tool_response is required") + return PostToolUseEvent( + **common, + **_tool_fields(payload), + tool_response=payload["tool_response"], + ) + raise ValueError(f"unsupported Codex hook event: {name}") + + +def _reason_with_code(reason: str, reason_code: str) -> str: + if not isinstance(reason, str) or not reason.strip(): + raise ValueError("reason must be a non-empty string") + if not isinstance(reason_code, str) or not reason_code.strip(): + raise ValueError("reason_code must be a non-empty string") + return f"{reason.strip()} [{reason_code.strip()}]" + + +def build_pre_tool_output(*, allowed: bool, reason: str, reason_code: str) -> dict[str, Any] | None: + """Build the documented Codex PreToolUse denial shape.""" + + if allowed: + return None + return { + "hookSpecificOutput": { + "hookEventName": "PreToolUse", + "permissionDecision": "deny", + "permissionDecisionReason": _reason_with_code(reason, reason_code), + } + } + + +def build_post_tool_output( + *, blocked: bool, reason: str, reason_code: str +) -> dict[str, str] | None: + """Build the documented Codex PostToolUse result-blocking shape.""" + + if not blocked: + return None + return {"decision": "block", "reason": _reason_with_code(reason, reason_code)} diff --git a/src/marginal/integrations/codex/evidence.py b/src/marginal/integrations/codex/evidence.py new file mode 100644 index 0000000..5cd28ff --- /dev/null +++ b/src/marginal/integrations/codex/evidence.py @@ -0,0 +1,222 @@ +"""Bounded, local-only evidence storage for Codex governance receipts.""" + +from __future__ import annotations + +import json +import os +import tempfile +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from .promotion import CoverageSummary + +_ALLOWED_EVIDENCE_FIELDS = { + "schema_version", + "event", + "session_hash", + "action_hash", + "semantic_key", + "state_hash", + "evidence_hash", + "outcome", + "reason_code", + "latency_ms", + "covered", + "coverable", + "recommended_stop", + "reviewed", + "false_stop", + "integration_failure", + "pending", + "timestamp", +} +_FORBIDDEN_FIELDS = { + "auth", + "command", + "credential", + "prompt", + "source", + "tool_input", + "tool_response", + "transcript", +} + + +def _canonical_bytes(payload: Mapping[str, Any]) -> bytes: + try: + return json.dumps( + dict(payload), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ).encode("utf-8") + except (TypeError, ValueError) as exc: + raise ValueError("evidence must be canonical JSON") from exc + + +class EvidenceStore: + """Append redacted JSONL records and atomically persist small checkpoints.""" + + def __init__(self, root: str | Path, *, max_record_bytes: int = 16_384) -> None: + if isinstance(max_record_bytes, bool) or not isinstance(max_record_bytes, int): + raise TypeError("max_record_bytes must be an integer") + if max_record_bytes < 128: + raise ValueError("max_record_bytes must be at least 128") + self.root = Path(root).resolve() + self.root.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + self.root.chmod(0o700) + self.path = self.root / "evidence.jsonl" + self.checkpoint_path = self.root / "checkpoint.json" + self.max_record_bytes = max_record_bytes + + def append(self, record: Mapping[str, Any]) -> None: + if not isinstance(record, Mapping): + raise TypeError("evidence record must be a mapping") + fields = set(record) + forbidden = fields & _FORBIDDEN_FIELDS + if forbidden: + raise ValueError(f"forbidden evidence field: {sorted(forbidden)[0]}") + unsupported = fields - _ALLOWED_EVIDENCE_FIELDS + if unsupported: + raise ValueError(f"unsupported evidence field: {sorted(unsupported)[0]}") + serialized = _canonical_bytes(record) + if len(serialized) > self.max_record_bytes: + raise ValueError("evidence record is too large") + descriptor = os.open(self.path, os.O_APPEND | os.O_CREAT | os.O_WRONLY, 0o600) + try: + os.write(descriptor, serialized + b"\n") + os.fsync(descriptor) + finally: + os.close(descriptor) + if os.name == "posix": + self.path.chmod(0o600) + + def read_all(self) -> list[dict[str, Any]]: + if not self.path.exists(): + return [] + records: list[dict[str, Any]] = [] + for line in self.path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + value = json.loads(line) + if not isinstance(value, dict): + raise ValueError("evidence record must decode to an object") + records.append(value) + return records + + def start_new_window(self, *, reason_code: str) -> None: + if not isinstance(reason_code, str) or not reason_code.strip(): + raise ValueError("reason_code must be a non-empty string") + self.append( + { + "schema_version": 1, + "event": "window_start", + "reason_code": reason_code.strip(), + } + ) + + def write_checkpoint(self, checkpoint: Mapping[str, Any]) -> None: + if not isinstance(checkpoint, Mapping): + raise TypeError("checkpoint must be a mapping") + serialized = _canonical_bytes(checkpoint) + if len(serialized) > self.max_record_bytes: + raise ValueError("checkpoint is too large") + descriptor, temporary_name = tempfile.mkstemp( + prefix=".checkpoint-", suffix=".tmp", dir=self.root + ) + temporary = Path(temporary_name) + try: + os.fchmod(descriptor, 0o600) + os.write(descriptor, serialized + b"\n") + os.fsync(descriptor) + os.close(descriptor) + descriptor = -1 + os.replace(temporary, self.checkpoint_path) + if os.name == "posix": + self.checkpoint_path.chmod(0o600) + finally: + if descriptor >= 0: + os.close(descriptor) + temporary.unlink(missing_ok=True) + + def read_checkpoint(self) -> dict[str, Any] | None: + if not self.checkpoint_path.exists(): + return None + value = json.loads(self.checkpoint_path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError("checkpoint must decode to an object") + return value + + +def summarize_evidence(records: list[dict[str, Any]]) -> CoverageSummary: + """Reduce redacted evidence into the exact Earned Enforcement gate surface.""" + + for index in range(len(records) - 1, -1, -1): + if records[index].get("event") == "window_start": + records = records[index + 1 :] + break + + decisions = [record for record in records if record.get("event") == "decision"] + completed_sessions = { + str(record.get("session_hash")) + for record in records + if record.get("event") == "session_end" and record.get("session_hash") + } + outcomes_by_action = { + str(record.get("action_hash")): str(record.get("outcome")) + for record in decisions + if record.get("outcome") in {"success", "failure", "unknown"} and record.get("action_hash") + } + outcomes_by_action.update( + { + str(record.get("action_hash")): str(record.get("outcome")) + for record in records + if record.get("event") == "outcome" and record.get("action_hash") + } + ) + candidates = { + str(record.get("action_hash")) + for record in decisions + if record.get("recommended_stop") is True and record.get("action_hash") + } + reviewed = { + str(record.get("action_hash")) + for record in records + if record.get("reviewed") is True and record.get("action_hash") in candidates + } + false_stops = { + str(record.get("action_hash")) + for record in records + if record.get("false_stop") is True and record.get("action_hash") in reviewed + } + outcomes = [ + outcomes_by_action[str(record.get("action_hash"))] + for record in decisions + if str(record.get("action_hash")) in outcomes_by_action + ] + unknown_outcomes = sum(outcome == "unknown" for outcome in outcomes) + latencies = tuple( + float(record["latency_ms"]) + for record in decisions + if isinstance(record.get("latency_ms"), (int, float)) + and not isinstance(record.get("latency_ms"), bool) + ) + return CoverageSummary( + covered_actions=sum(record.get("covered") is True for record in decisions), + coverable_actions=sum(record.get("coverable") is True for record in decisions), + completed_sessions=len(completed_sessions), + reviewed_candidates=len(reviewed), + false_stops=len(false_stops), + integration_failures=sum(record.get("integration_failure") is True for record in records), + pending_actions=sum( + record.get("pending") is True + and str(record.get("action_hash")) not in outcomes_by_action + for record in decisions + ), + unknown_enforceable_outcomes=unknown_outcomes, + decision_latencies_ms=latencies, + enforceable_outcomes_observable=bool(outcomes) and unknown_outcomes == 0, + intervention_candidates=len(candidates), + ) diff --git a/src/marginal/integrations/codex/identity.py b/src/marginal/integrations/codex/identity.py new file mode 100644 index 0000000..9fd27be --- /dev/null +++ b/src/marginal/integrations/codex/identity.py @@ -0,0 +1,94 @@ +"""Stable local identity for repository-scoped Codex promotion receipts.""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path + +from .installer import CommandRunner, SubprocessRunner, inspect_codex +from .promotion import PromotionIdentity + +PLUGIN_VERSION = "0.3.0" +ADAPTER_VERSION = "1" +POLICY_HASH = hashlib.sha256(b"marginal:no-progress:v1:max-same-evidence=2").hexdigest() +DEFAULT_HOOK_HASH = "46b7a85a3a542957d055c615ab501f9fee284bb3193ac9ecfbe8951cce5a9942" + + +def repository_identity_hash(workspace: str | Path) -> str: + return hashlib.sha256(str(Path(workspace).resolve()).encode("utf-8")).hexdigest() + + +def _installed_plugin_root(runner: CommandRunner) -> Path | None: + result = runner.run(["codex", "plugin", "list", "--json"]) + if result.returncode != 0: + return None + try: + payload = json.loads(result.stdout) + except json.JSONDecodeError: + return None + installed = payload.get("installed") if isinstance(payload, dict) else None + if not isinstance(installed, list): + return None + for plugin in installed: + if not isinstance(plugin, dict) or plugin.get("pluginId") != "marginal@marginal": + continue + source = plugin.get("source") + if isinstance(source, dict) and isinstance(source.get("path"), str): + candidate = Path(source["path"]).resolve() + if candidate.is_dir(): + return candidate + return None + + +def _plugin_identity( + plugin_root: str | Path | None, + runner: CommandRunner, +) -> tuple[str, str]: + selected_root = Path(plugin_root).resolve() if plugin_root is not None else None + if selected_root is None: + environment_root = os.environ.get("PLUGIN_ROOT") + selected_root = Path(environment_root).resolve() if environment_root else None + if selected_root is None: + selected_root = _installed_plugin_root(runner) + if selected_root is None: + return PLUGIN_VERSION, DEFAULT_HOOK_HASH + manifest_path = selected_root / ".codex-plugin" / "plugin.json" + hook_path = selected_root / "hooks" / "hooks.json" + version = PLUGIN_VERSION + if manifest_path.is_file(): + try: + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + if isinstance(manifest, dict) and isinstance(manifest.get("version"), str): + version = manifest["version"] + except (OSError, json.JSONDecodeError): + pass + hook_hash = ( + hashlib.sha256(hook_path.read_bytes()).hexdigest() + if hook_path.is_file() + else DEFAULT_HOOK_HASH + ) + return version, hook_hash + + +def current_promotion_identity( + workspace: str | Path, + *, + codex_version: str | None = None, + plugin_root: str | Path | None = None, + runner: CommandRunner | None = None, +) -> PromotionIdentity: + selected = runner or SubprocessRunner() + resolved_codex_version = codex_version + if resolved_codex_version is None: + resolved_codex_version = inspect_codex(runner=selected).version or "unobservable" + plugin_version, hook_hash = _plugin_identity(plugin_root, selected) + return PromotionIdentity( + repository_hash=repository_identity_hash(workspace), + codex_version=resolved_codex_version, + plugin_version=plugin_version, + adapter_version=ADAPTER_VERSION, + policy_hash=POLICY_HASH, + hook_hash=hook_hash, + ) diff --git a/src/marginal/integrations/codex/installer.py b/src/marginal/integrations/codex/installer.py new file mode 100644 index 0000000..7cd8cc7 --- /dev/null +++ b/src/marginal/integrations/codex/installer.py @@ -0,0 +1,180 @@ +"""Read-only Codex discovery and reversible native plugin operations.""" + +from __future__ import annotations + +import os +import re +import shutil +import subprocess +from dataclasses import dataclass +from typing import Protocol + + +@dataclass(frozen=True, slots=True) +class CommandResult: + returncode: int + stdout: str + stderr: str + + +class CommandRunner(Protocol): + def run(self, args: list[str]) -> CommandResult: ... + + +class SubprocessRunner: + """Run Codex without a shell, credential inspection, or inherited secret variables.""" + + def __init__(self, *, timeout_seconds: float = 15.0, max_output_bytes: int = 1_048_576): + self.timeout_seconds = timeout_seconds + self.max_output_bytes = max_output_bytes + + def run(self, args: list[str]) -> CommandResult: + environment = { + name: value + for name in ("PATH", "HOME", "CODEX_HOME", "LANG", "LC_ALL", "SYSTEMROOT") + if (value := os.environ.get(name)) is not None + } + try: + completed = subprocess.run( + args, + check=False, + capture_output=True, + text=False, + timeout=self.timeout_seconds, + env=environment, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + return CommandResult(127, "", type(exc).__name__) + stdout = completed.stdout[: self.max_output_bytes].decode("utf-8", errors="replace") + stderr = completed.stderr[: self.max_output_bytes].decode("utf-8", errors="replace") + return CommandResult(completed.returncode, stdout, stderr) + + +@dataclass(frozen=True, slots=True) +class CodexDoctorReport: + available: bool + version: str + hooks_enabled: bool + plugins_enabled: bool + capability_level: str + blocking_reasons: tuple[str, ...] + + def to_dict(self) -> dict[str, object]: + return { + "available": self.available, + "version": self.version, + "hooks_enabled": self.hooks_enabled, + "plugins_enabled": self.plugins_enabled, + "capability_level": self.capability_level, + "capability_label": ( + "Tool Enforcement" if self.capability_level == "tool_enforcement" else "Observe" + ), + "blocking_reasons": list(self.blocking_reasons), + } + + +@dataclass(frozen=True, slots=True) +class CodexInstallation: + installed: bool + changed: bool + selector: str = "marginal@marginal" + error_code: str = "" + message: str = "" + + def to_dict(self) -> dict[str, object]: + return { + "installed": self.installed, + "changed": self.changed, + "selector": self.selector, + "error_code": self.error_code, + "message": self.message, + } + + +def _feature_enabled(output: str, name: str) -> bool: + for line in output.splitlines(): + fields = line.split() + if len(fields) >= 3 and fields[0] == name: + return fields[-1].casefold() == "true" + return False + + +def inspect_codex(*, runner: CommandRunner | None = None) -> CodexDoctorReport: + """Discover stable features exclusively through public Codex CLI commands.""" + + selected = runner or SubprocessRunner() + if runner is None and shutil.which("codex") is None: + return CodexDoctorReport(False, "", False, False, "observe", ("CODEX_NOT_FOUND",)) + version_result = selected.run(["codex", "--version"]) + if version_result.returncode != 0: + return CodexDoctorReport(False, "", False, False, "observe", ("CODEX_NOT_FOUND",)) + match = re.search(r"(\d+\.\d+\.\d+)", version_result.stdout) + version = match.group(1) if match else version_result.stdout.strip() + features = selected.run(["codex", "features", "list"]) + hooks = features.returncode == 0 and _feature_enabled(features.stdout, "hooks") + plugins = features.returncode == 0 and _feature_enabled(features.stdout, "plugins") + reasons: list[str] = [] + if not hooks: + reasons.append("HOOKS_UNAVAILABLE") + if not plugins: + reasons.append("PLUGINS_UNAVAILABLE") + level = "tool_enforcement" if hooks and plugins else "observe" + return CodexDoctorReport(True, version, hooks, plugins, level, tuple(reasons)) + + +def install( + *, + runner: CommandRunner | None = None, + repository: str = "SignalLayerLabs/Marginal", + ref: str = "main", +) -> CodexInstallation: + selected = runner or SubprocessRunner() + report = inspect_codex(runner=selected) + if report.capability_level != "tool_enforcement": + return CodexInstallation( + False, + False, + error_code="CODEX_CAPABILITIES_UNAVAILABLE", + message=", ".join(report.blocking_reasons), + ) + marketplace = selected.run( + [ + "codex", + "plugin", + "marketplace", + "add", + repository, + "--ref", + ref, + "--json", + ] + ) + if marketplace.returncode != 0 and "already" not in marketplace.stderr.casefold(): + return CodexInstallation( + False, + False, + error_code="MARKETPLACE_ADD_FAILED", + message=marketplace.stderr.strip(), + ) + plugin = selected.run(["codex", "plugin", "add", "marginal@marginal", "--json"]) + if plugin.returncode != 0 and "already" not in plugin.stderr.casefold(): + return CodexInstallation( + False, + False, + error_code="PLUGIN_ADD_FAILED", + message=plugin.stderr.strip(), + ) + return CodexInstallation(True, True, message="installed in Shadow Mode") + + +def uninstall(*, runner: CommandRunner | None = None) -> CodexInstallation: + selected = runner or SubprocessRunner() + result = selected.run(["codex", "plugin", "remove", "marginal@marginal", "--json"]) + if result.returncode != 0 and "not installed" not in result.stderr.casefold(): + return CodexInstallation( + True, + False, + error_code="PLUGIN_REMOVE_FAILED", + message=result.stderr.strip(), + ) + return CodexInstallation(False, True, message="plugin removed; local evidence preserved") diff --git a/src/marginal/integrations/codex/normalization.py b/src/marginal/integrations/codex/normalization.py new file mode 100644 index 0000000..e793e54 --- /dev/null +++ b/src/marginal/integrations/codex/normalization.py @@ -0,0 +1,97 @@ +"""Convert raw Codex hook inputs into privacy-safe protocol actions.""" + +from __future__ import annotations + +import hashlib +import json +import re +from typing import Any + +from marginal.models import Cost +from marginal.protocol import AgentAction, DeduplicationScope + +from .events import PreToolUseEvent + +_VERIFICATION_PATTERN = re.compile( + r"(?:^|[\s/])(?:pytest|tox|nox|unittest|jest|vitest|mocha|rspec|" + r"go\s+test|cargo\s+test|npm\s+(?:run\s+)?test|pnpm\s+(?:run\s+)?test|" + r"yarn\s+test|ruff|mypy|pyright|eslint|tsc|git\s+diff\s+--check)(?:\s|$)", + re.IGNORECASE, +) + + +def _canonical_json(value: Any) -> str: + try: + return json.dumps( + value, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + except (TypeError, ValueError) as exc: + raise ValueError("tool_input must have a canonical JSON representation") from exc + + +def _semantic_key(event: PreToolUseEvent) -> str: + canonical = _canonical_json( + {"tool_name": event.tool_name.casefold(), "tool_input": event.tool_input} + ) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _command(event: PreToolUseEvent) -> str: + command = event.tool_input.get("command") + return command if isinstance(command, str) else "" + + +def _action_kind(event: PreToolUseEvent) -> tuple[str, bool]: + tool = event.tool_name.casefold() + command = _command(event) + if command and _VERIFICATION_PATTERN.search(command): + return "verification", True + if tool in {"bash", "shell", "exec", "exec_command", "terminal"}: + return "shell", False + if tool in {"apply_patch", "edit", "write", "write_file"}: + return "edit", False + if tool.startswith("mcp"): + return "mcp", False + return "tool", False + + +def normalize_pre_tool_use( + event: PreToolUseEvent, + *, + state_hash: str, + previous_evidence_hash: str = "", +) -> AgentAction: + """Return a normalized action without retaining raw arguments or descriptions.""" + + if not isinstance(event, PreToolUseEvent): + raise TypeError("event must be a PreToolUseEvent") + if not isinstance(state_hash, str) or not state_hash.strip(): + raise ValueError("state_hash must be a non-empty string") + if not isinstance(previous_evidence_hash, str): + raise TypeError("previous_evidence_hash must be a string") + + semantic_key = _semantic_key(event) + kind, is_verification = _action_kind(event) + metadata = { + "session_id": event.session_id, + "turn_id": event.turn_id, + "tool_name": event.tool_name, + "state_hash": state_hash, + "evidence_hash": previous_evidence_hash, + "semantic_key": semantic_key, + } + return AgentAction( + action_id=event.tool_use_id, + name=f"Codex {event.tool_name} action", + kind=kind, + estimated_cost=Cost(), + is_verification=is_verification, + state_hash=state_hash, + phase="codex-tool-use", + deduplication_scope=DeduplicationScope.ONCE_PER_STATE, + metadata=metadata, + ) diff --git a/src/marginal/integrations/codex/outcomes.py b/src/marginal/integrations/codex/outcomes.py new file mode 100644 index 0000000..1d28972 --- /dev/null +++ b/src/marginal/integrations/codex/outcomes.py @@ -0,0 +1,71 @@ +"""Conservative outcome classification for documented structured tool results.""" + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Mapping +from typing import Any + +from marginal.controls import ActionOutcomeStatus + +from .events import PostToolUseEvent + +_SUCCESS_TEXT = {"success", "succeeded", "passed"} +_FAILURE_TEXT = {"failure", "failed", "error"} + + +def _structured_signals(response: Mapping[str, Any]) -> set[ActionOutcomeStatus]: + signals: set[ActionOutcomeStatus] = set() + exit_code = response.get("exit_code") + if isinstance(exit_code, int) and not isinstance(exit_code, bool): + signals.add(ActionOutcomeStatus.SUCCESS if exit_code == 0 else ActionOutcomeStatus.FAILURE) + success = response.get("success") + if isinstance(success, bool): + signals.add(ActionOutcomeStatus.SUCCESS if success else ActionOutcomeStatus.FAILURE) + is_error = response.get("is_error") + if isinstance(is_error, bool): + signals.add(ActionOutcomeStatus.FAILURE if is_error else ActionOutcomeStatus.SUCCESS) + for key in ("status", "outcome"): + value = response.get(key) + if not isinstance(value, str): + continue + normalized = value.strip().casefold() + if normalized in _SUCCESS_TEXT: + signals.add(ActionOutcomeStatus.SUCCESS) + elif normalized in _FAILURE_TEXT: + signals.add(ActionOutcomeStatus.FAILURE) + return signals + + +def classify_tool_outcome(event: PostToolUseEvent) -> ActionOutcomeStatus: + """Classify only explicit, mutually consistent structured outcome signals. + + Codex runs PostToolUse after non-zero shell exits. Human-readable response text is + therefore evidence of completion, not evidence of success. + """ + + if not isinstance(event, PostToolUseEvent): + raise TypeError("event must be a PostToolUseEvent") + if not isinstance(event.tool_response, Mapping): + return ActionOutcomeStatus.UNKNOWN + signals = _structured_signals(event.tool_response) + if len(signals) != 1: + return ActionOutcomeStatus.UNKNOWN + return next(iter(signals)) + + +def completion_evidence_hash(response: Any) -> str: + """Hash JSON-compatible completion evidence without retaining its contents.""" + + try: + canonical = json.dumps( + response, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + except (TypeError, ValueError): + return "" + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() diff --git a/src/marginal/integrations/codex/promotion.py b/src/marginal/integrations/codex/promotion.py new file mode 100644 index 0000000..99cb3b5 --- /dev/null +++ b/src/marginal/integrations/codex/promotion.py @@ -0,0 +1,334 @@ +"""Evidence gate that earns and continuously validates Codex enforcement.""" + +from __future__ import annotations + +import hashlib +import json +import math +import os +from dataclasses import asdict, dataclass, replace +from pathlib import Path +from typing import Any + + +@dataclass(frozen=True, slots=True) +class PromotionIdentity: + repository_hash: str + codex_version: str + plugin_version: str + adapter_version: str + policy_hash: str + hook_hash: str + + +@dataclass(frozen=True, slots=True) +class CoverageSummary: + covered_actions: int + coverable_actions: int + completed_sessions: int + reviewed_candidates: int + false_stops: int + integration_failures: int + pending_actions: int + unknown_enforceable_outcomes: int + decision_latencies_ms: tuple[float, ...] + enforceable_outcomes_observable: bool + intervention_candidates: int = 0 + + def __post_init__(self) -> None: + for name in ( + "covered_actions", + "coverable_actions", + "completed_sessions", + "reviewed_candidates", + "false_stops", + "integration_failures", + "pending_actions", + "unknown_enforceable_outcomes", + "intervention_candidates", + ): + value = getattr(self, name) + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise ValueError(f"{name} must be a non-negative integer") + normalized: list[float] = [] + for value in self.decision_latencies_ms: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError("decision latency must be numeric") + measured = float(value) + if not math.isfinite(measured) or measured < 0: + raise ValueError("decision latency must be finite and non-negative") + normalized.append(measured) + object.__setattr__(self, "decision_latencies_ms", tuple(normalized)) + + +@dataclass(frozen=True, slots=True) +class PromotionCriteria: + minimum_actions: int = 100 + minimum_sessions: int = 5 + minimum_coverage_ratio: float = 0.99 + minimum_reviewed_candidates: int = 5 + maximum_false_stops: int = 0 + maximum_p95_latency_ms: float = 75.0 + + +@dataclass(frozen=True, slots=True) +class PromotionReceipt: + schema_version: int + identity: PromotionIdentity + criteria: PromotionCriteria + summary: CoverageSummary + coverage_ratio: float + p95_latency_ms: float + blocking_reasons: tuple[str, ...] + is_ready: bool + receipt_hash: str + + def _hash_payload(self) -> dict[str, Any]: + payload = self.to_dict() + payload.pop("receipt_hash", None) + return payload + + def verify_hash(self) -> bool: + return self.receipt_hash == _hash(self._hash_payload()) + + def valid_for(self, identity: PromotionIdentity) -> bool: + return self.is_ready and self.verify_hash() and identity == self.identity + + def to_dict(self) -> dict[str, Any]: + return { + "schema_version": self.schema_version, + "identity": asdict(self.identity), + "criteria": asdict(self.criteria), + "summary": asdict(self.summary), + "coverage_ratio": self.coverage_ratio, + "p95_latency_ms": self.p95_latency_ms, + "blocking_reasons": list(self.blocking_reasons), + "is_ready": self.is_ready, + "receipt_hash": self.receipt_hash, + } + + @classmethod + def from_dict(cls, payload: dict[str, Any]) -> PromotionReceipt: + summary_data = dict(payload["summary"]) + summary_data["decision_latencies_ms"] = tuple(summary_data["decision_latencies_ms"]) + return cls( + schema_version=int(payload["schema_version"]), + identity=PromotionIdentity(**payload["identity"]), + criteria=PromotionCriteria(**payload["criteria"]), + summary=CoverageSummary(**summary_data), + coverage_ratio=float(payload["coverage_ratio"]), + p95_latency_ms=float(payload["p95_latency_ms"]), + blocking_reasons=tuple(payload["blocking_reasons"]), + is_ready=bool(payload["is_ready"]), + receipt_hash=str(payload["receipt_hash"]), + ) + + +def _hash(payload: dict[str, Any]) -> str: + canonical = json.dumps( + payload, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _p95(values: tuple[float, ...]) -> float: + if not values: + return 0.0 + ordered = sorted(values) + return ordered[max(0, math.ceil(len(ordered) * 0.95) - 1)] + + +def evaluate_promotion( + summary: CoverageSummary, + criteria: PromotionCriteria, + *, + identity: PromotionIdentity, +) -> PromotionReceipt: + """Create a self-verifying receipt for the conservative default evidence gate.""" + + ratio = ( + summary.covered_actions / summary.coverable_actions if summary.coverable_actions else 0.0 + ) + latency = _p95(summary.decision_latencies_ms) + reasons: list[str] = [] + if summary.covered_actions < criteria.minimum_actions: + reasons.append("MINIMUM_ACTIONS") + if summary.completed_sessions < criteria.minimum_sessions: + reasons.append("MINIMUM_SESSIONS") + if ratio < criteria.minimum_coverage_ratio: + reasons.append("COVERAGE") + if summary.reviewed_candidates < criteria.minimum_reviewed_candidates: + reasons.append("MINIMUM_REVIEWS") + if summary.reviewed_candidates < summary.intervention_candidates: + reasons.append("UNREVIEWED_CANDIDATES") + if summary.false_stops > criteria.maximum_false_stops: + reasons.append("FALSE_STOPS") + if summary.integration_failures: + reasons.append("INTEGRATION_FAILURES") + if summary.pending_actions: + reasons.append("PENDING_ACTIONS") + if latency > criteria.maximum_p95_latency_ms: + reasons.append("LATENCY") + if not summary.enforceable_outcomes_observable: + reasons.append("OUTCOME_UNOBSERVABLE") + if summary.unknown_enforceable_outcomes: + reasons.append("UNKNOWN_ENFORCEABLE_OUTCOMES") + + provisional = PromotionReceipt( + schema_version=1, + identity=identity, + criteria=criteria, + summary=summary, + coverage_ratio=ratio, + p95_latency_ms=latency, + blocking_reasons=tuple(reasons), + is_ready=not reasons, + receipt_hash="", + ) + return replace(provisional, receipt_hash=_hash(provisional._hash_payload())) + + +def _repositories_root(data_root: str | Path) -> Path: + root = Path(data_root).resolve() / "repositories" + root.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + root.chmod(0o700) + return root + + +def _receipt_path(data_root: str | Path, repository_hash: str) -> Path: + return _repositories_root(data_root) / f"{repository_hash}.receipt.json" + + +def _state_path(data_root: str | Path, repository_hash: str) -> Path: + return _repositories_root(data_root) / f"{repository_hash}.json" + + +def _atomic_json(path: Path, payload: dict[str, Any]) -> None: + serialized = json.dumps(payload, sort_keys=True, separators=(",", ":"), allow_nan=False) + "\n" + temporary = path.with_suffix(path.suffix + ".tmp") + descriptor = os.open(temporary, os.O_CREAT | os.O_TRUNC | os.O_WRONLY, 0o600) + try: + os.write(descriptor, serialized.encode("utf-8")) + os.fsync(descriptor) + finally: + os.close(descriptor) + os.replace(temporary, path) + if os.name == "posix": + path.chmod(0o600) + + +def write_promotion_receipt(data_root: str | Path, receipt: PromotionReceipt) -> Path: + if not receipt.verify_hash(): + raise ValueError("promotion receipt hash is invalid") + path = _receipt_path(data_root, receipt.identity.repository_hash) + _atomic_json(path, receipt.to_dict()) + return path + + +def read_promotion_receipt( + data_root: str | Path, + repository_hash: str, +) -> PromotionReceipt | None: + path = _receipt_path(data_root, repository_hash) + if not path.exists(): + return None + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError("promotion receipt must be a JSON object") + return PromotionReceipt.from_dict(payload) + + +def activate_enforcement(data_root: str | Path, receipt: PromotionReceipt) -> Path: + if not receipt.is_ready or not receipt.verify_hash(): + raise ValueError("a ready, hash-valid receipt is required for enforcement") + stored = read_promotion_receipt(data_root, receipt.identity.repository_hash) + if stored != receipt: + raise ValueError("promotion receipt must be persisted before enforcement") + path = _state_path(data_root, receipt.identity.repository_hash) + _atomic_json( + path, + { + "schema_version": 1, + "mode": "enforce", + "reason": "EARNED_ENFORCEMENT_PROMOTED", + "receipt_hash": receipt.receipt_hash, + "identity": asdict(receipt.identity), + }, + ) + return path + + +def demote_enforcement( + data_root: str | Path, + *, + repository_hash: str, + reason: str, +) -> Path: + path = _state_path(data_root, repository_hash) + _atomic_json( + path, + {"schema_version": 1, "mode": "shadow", "reason": reason}, + ) + return path + + +def enforcement_is_active( + data_root: str | Path, + *, + identity: PromotionIdentity, + summary: CoverageSummary | None = None, +) -> bool: + state_path = _state_path(data_root, identity.repository_hash) + if not state_path.exists(): + return False + try: + state = json.loads(state_path.read_text(encoding="utf-8")) + if not isinstance(state, dict) or state.get("mode") != "enforce": + return False + receipt = read_promotion_receipt(data_root, identity.repository_hash) + if ( + receipt is None + or state.get("receipt_hash") != receipt.receipt_hash + or not receipt.valid_for(identity) + ): + demote_enforcement( + data_root, + repository_hash=identity.repository_hash, + reason="IDENTITY_DRIFT", + ) + return False + if summary is not None: + coverage_ratio = ( + summary.covered_actions / summary.coverable_actions + if summary.coverable_actions + else 0.0 + ) + evidence_drift = ( + coverage_ratio < receipt.criteria.minimum_coverage_ratio + or summary.false_stops > receipt.summary.false_stops + or summary.integration_failures > 0 + or summary.pending_actions > 0 + or summary.unknown_enforceable_outcomes > 0 + or summary.reviewed_candidates < summary.intervention_candidates + or _p95(summary.decision_latencies_ms) > receipt.criteria.maximum_p95_latency_ms + ) + if evidence_drift: + demote_enforcement( + data_root, + repository_hash=identity.repository_hash, + reason="EVIDENCE_DRIFT", + ) + return False + return True + except (OSError, ValueError, KeyError, TypeError, json.JSONDecodeError): + demote_enforcement( + data_root, + repository_hash=identity.repository_hash, + reason="RECEIPT_INVALID", + ) + return False diff --git a/src/marginal/integrations/codex/runtime.py b/src/marginal/integrations/codex/runtime.py new file mode 100644 index 0000000..a1770e5 --- /dev/null +++ b/src/marginal/integrations/codex/runtime.py @@ -0,0 +1,222 @@ +"""Transactional Codex session lifecycle over the provider-neutral runtime.""" + +from __future__ import annotations + +import hashlib +from collections.abc import Callable +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from marginal.controls import ActionOutcomeStatus, NoProgressDetector, NoProgressSignal +from marginal.models import Cost, Decision +from marginal.protocol import AgentAction, AgentDecision +from marginal.runtime import UniversalRuntime + +from .events import PostToolUseEvent, PreToolUseEvent +from .normalization import normalize_pre_tool_use +from .outcomes import classify_tool_outcome, completion_evidence_hash +from .state import workspace_state_hash + + +class CodexIntegrationError(RuntimeError): + """Raised when hook lifecycle identity is missing or inconsistent.""" + + +@dataclass(frozen=True, slots=True) +class _PendingAction: + event: PreToolUseEvent + action: AgentAction + + +class CodexSessionRuntime: + """Correlate Codex hook events and settle each accepted action exactly once.""" + + def __init__( + self, + runtime: UniversalRuntime, + *, + workspace: str | Path, + detector: NoProgressDetector | None = None, + enforcement_enabled: Callable[[], bool] | None = None, + ) -> None: + if not isinstance(runtime, UniversalRuntime): + raise TypeError("runtime must be a UniversalRuntime") + self.runtime = runtime + self.workspace = Path(workspace).resolve() + workspace_state_hash(self.workspace) + self.detector = detector or NoProgressDetector() + self._enforcement_enabled = enforcement_enabled or (lambda: False) + self._pending: dict[str, _PendingAction] = {} + self._evidence_by_semantic_key: dict[str, str] = {} + self._last_signal: NoProgressSignal | None = None + self._last_action_evidence: dict[str, str] | None = None + self._successful = 0 + self._failed = 0 + self._unknown = 0 + self._enforced_denials = 0 + self._closed = False + + @property + def last_no_progress_signal(self) -> NoProgressSignal | None: + return self._last_signal + + @property + def last_action_evidence(self) -> dict[str, str] | None: + return dict(self._last_action_evidence) if self._last_action_evidence else None + + def pre_tool_use(self, event: PreToolUseEvent) -> AgentDecision: + self._ensure_open() + self._validate_session(event.session_id) + if event.tool_use_id in self._pending: + raise CodexIntegrationError(f"tool identity is already pending: {event.tool_use_id}") + state_hash = workspace_state_hash(self.workspace) + action = normalize_pre_tool_use(event, state_hash=state_hash) + semantic_key = str(action.metadata["semantic_key"]) + evidence_hash = self._evidence_by_semantic_key.get(semantic_key, "") + if evidence_hash: + action = normalize_pre_tool_use( + event, + state_hash=state_hash, + previous_evidence_hash=evidence_hash, + ) + self._last_action_evidence = self._safe_action_evidence(action) + self._last_signal = self.detector.evaluate( + semantic_key, + state_hash, + evidence_hash, + ) + if self._last_signal.enforcement_eligible and self._is_enforcement_enabled(): + self._enforced_denials += 1 + return AgentDecision.from_core( + event.tool_use_id, + Decision( + allowed=False, + reason="Repeated proven-success action produced no new state or evidence", + recommended=False, + recommendation_reason=( + "Repeated proven-success action produced no new state or evidence" + ), + reason_code="NO_PROGRESS_ENFORCED", + recommendation_reason_code="NO_PROGRESS_ENFORCED", + mode="enforce", + confidence=1.0, + ), + ) + decision = self.runtime.before_action(action) + if decision.allowed: + self._pending[event.tool_use_id] = _PendingAction(event=event, action=action) + return decision + + def post_tool_use(self, event: PostToolUseEvent) -> ActionOutcomeStatus: + self._ensure_open() + self._validate_session(event.session_id) + pending = self._pending.get(event.tool_use_id) + if pending is None: + raise CodexIntegrationError( + f"PostToolUse identity does not match a pending action: {event.tool_use_id}" + ) + self._validate_post_identity(pending.event, event) + + outcome = classify_tool_outcome(event) + evidence_hash = completion_evidence_hash(event.tool_response) + semantic_key = str(pending.action.metadata["semantic_key"]) + post_state_hash = workspace_state_hash(self.workspace) + + if outcome is ActionOutcomeStatus.SUCCESS: + self.runtime.after_action(event.tool_use_id, actual_cost=Cost()) + self._successful += 1 + elif outcome is ActionOutcomeStatus.FAILURE: + self.runtime.fail_action( + event.tool_use_id, + reason="Codex returned an explicit structured failure", + actual_cost=Cost(), + ) + self._failed += 1 + else: + self.runtime.fail_action( + event.tool_use_id, + reason="Codex completion outcome was not observable", + ) + self._unknown += 1 + + self._pending.pop(event.tool_use_id) + self.detector.observe(semantic_key, post_state_hash, evidence_hash, outcome) + if evidence_hash: + self._evidence_by_semantic_key[semantic_key] = evidence_hash + return outcome + + def pending_action_ids(self) -> tuple[str, ...]: + return tuple(sorted(self._pending)) + + def action_evidence(self, action_id: str) -> dict[str, str] | None: + pending = self._pending.get(action_id) + return self._safe_action_evidence(pending.action) if pending is not None else None + + def summary(self) -> dict[str, int]: + return { + "successful_observations": self._successful, + "failed_observations": self._failed, + "unknown_observations": self._unknown, + "completed_observations": self._successful + self._failed + self._unknown, + "pending_actions": len(self._pending), + "enforced_denials": self._enforced_denials, + } + + def close(self) -> None: + if self._closed: + return + for action_id in tuple(self._pending): + self.runtime.fail_action( + action_id, + reason="Codex session ended before the action outcome was observable", + ) + self._unknown += 1 + self._pending.pop(action_id) + self._closed = True + + def _ensure_open(self) -> None: + if self._closed: + raise CodexIntegrationError("Codex session runtime is closed") + + def _is_enforcement_enabled(self) -> bool: + try: + enabled = self._enforcement_enabled() + except Exception: + return False + return enabled if isinstance(enabled, bool) else False + + @staticmethod + def _safe_action_evidence(action: AgentAction) -> dict[str, str]: + return { + "action_hash": hashlib.sha256(action.action_id.encode("utf-8")).hexdigest(), + "semantic_key": str(action.metadata.get("semantic_key", "")), + "state_hash": action.state_hash, + "evidence_hash": str(action.metadata.get("evidence_hash", "")), + } + + def _validate_session(self, session_id: str) -> None: + if session_id != self.runtime.session_id: + raise CodexIntegrationError("hook session identity does not match runtime identity") + + @staticmethod + def _validate_post_identity( + before: PreToolUseEvent, + after: PostToolUseEvent, + ) -> None: + expected: tuple[Any, ...] = ( + before.session_id, + before.turn_id, + before.tool_name, + before.tool_use_id, + dict(before.tool_input), + ) + observed: tuple[Any, ...] = ( + after.session_id, + after.turn_id, + after.tool_name, + after.tool_use_id, + dict(after.tool_input), + ) + if expected != observed: + raise CodexIntegrationError("PreToolUse and PostToolUse identity does not match") diff --git a/src/marginal/integrations/codex/service.py b/src/marginal/integrations/codex/service.py new file mode 100644 index 0000000..5125a3b --- /dev/null +++ b/src/marginal/integrations/codex/service.py @@ -0,0 +1,475 @@ +"""Per-session Codex governance service and fail-open hook entry point.""" + +from __future__ import annotations + +import hashlib +import json +import os +import secrets +import subprocess +import sys +import threading +import time +from contextlib import suppress +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +from marginal import BudgetLimits, Treasury +from marginal.protocol import AgentCapabilities +from marginal.runtime import UniversalRuntime + +from .events import ( + PostToolUseEvent, + PreToolUseEvent, + SessionEvent, + build_pre_tool_output, + parse_hook_event, +) +from .evidence import EvidenceStore, summarize_evidence +from .identity import current_promotion_identity, repository_identity_hash +from .promotion import PromotionIdentity, demote_enforcement, enforcement_is_active +from .runtime import CodexSessionRuntime +from .transport import ConnectionInfo, SessionServer, connection_filename, request_session + +_SERVERS: dict[tuple[Path, str], tuple[SessionServer, CodexSessionRuntime]] = {} + + +@dataclass(frozen=True, slots=True) +class HookResult: + exit_code: int + output: dict[str, Any] | None = None + warning_code: str = "" + + +def _connection_path(data_root: Path, session_id: str) -> Path: + return data_root / "sessions" / connection_filename(session_id) + + +def _handler( + runtime: CodexSessionRuntime, + *, + evidence_store: EvidenceStore, + session_hash: str, + data_root: Path, + identity: PromotionIdentity, + shutdown_event: threading.Event | None = None, +) -> Any: + def handle(operation: str, payload: dict[str, Any]) -> dict[str, Any] | None: + if operation == "status": + return runtime.summary() + if operation == "close": + runtime.close() + evidence_store.append( + { + "schema_version": 1, + "event": "session_end", + "session_hash": session_hash, + } + ) + if shutdown_event is not None: + threading.Timer(0.05, shutdown_event.set).start() + return runtime.summary() + event = parse_hook_event(payload) + if operation == "pre" and isinstance(event, PreToolUseEvent): + started = time.perf_counter_ns() + decision = runtime.pre_tool_use(event) + latency_ms = (time.perf_counter_ns() - started) / 1_000_000 + action_evidence = runtime.last_action_evidence or {} + signal = runtime.last_no_progress_signal + evidence_store.append( + { + "schema_version": 1, + "event": "decision", + "session_hash": session_hash, + **action_evidence, + "reason_code": decision.reason_code, + "latency_ms": latency_ms, + "covered": True, + "coverable": True, + "recommended_stop": bool(signal and signal.should_recommend_stop), + "reviewed": False, + "false_stop": False, + "pending": decision.allowed, + } + ) + return build_pre_tool_output( + allowed=decision.allowed, + reason=decision.reason, + reason_code=decision.reason_code, + ) + if operation == "post" and isinstance(event, PostToolUseEvent): + action_evidence = runtime.action_evidence(event.tool_use_id) or {} + outcome = runtime.post_tool_use(event) + evidence_store.append( + { + "schema_version": 1, + "event": "outcome", + "session_hash": session_hash, + **action_evidence, + "outcome": outcome.value, + "pending": False, + } + ) + if ( + outcome.value == "unknown" + and read_mode(data_root, repository_hash=identity.repository_hash).get("mode") + == "enforce" + ): + demote_enforcement( + data_root, + repository_hash=identity.repository_hash, + reason="OUTCOME_UNOBSERVABLE", + ) + evidence_store.start_new_window(reason_code="OUTCOME_UNOBSERVABLE") + return None + raise ValueError("unsupported service operation") + + return handle + + +def _session_hash(session_id: str) -> str: + return hashlib.sha256(session_id.encode("utf-8")).hexdigest() + + +def _evidence_store(data_root: Path, repository_hash: str) -> EvidenceStore: + return EvidenceStore(data_root / "evidence" / repository_hash) + + +def _bootstrap_path(data_root: Path, session_id: str) -> Path: + bootstrap_root = data_root / "bootstrap" + bootstrap_root.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + bootstrap_root.chmod(0o700) + session_key = hashlib.sha256(session_id.encode("utf-8")).hexdigest() + return bootstrap_root / f"{session_key}-{secrets.token_hex(8)}.json" + + +def _bootstrap_event_payload(event: SessionEvent) -> dict[str, Any]: + """Keep the ephemeral service bootstrap free of transcript and unrelated hook fields.""" + + return { + "session_id": event.session_id, + "cwd": event.cwd, + "hook_event_name": event.hook_event_name, + "model": event.model, + "permission_mode": event.permission_mode, + "source": event.source, + } + + +def _spawn_session_service( + event: SessionEvent, + *, + data_root: Path, +) -> ConnectionInfo: + existing_path = _connection_path(data_root, event.session_id) + if existing_path.exists(): + try: + existing = ConnectionInfo.from_file(existing_path) + if request_session(existing, operation="status", payload={}).get("ok") is True: + return existing + except (OSError, ValueError, KeyError): + pass + existing_path.unlink(missing_ok=True) + + bootstrap = _bootstrap_path(data_root, event.session_id) + payload = { + "event": _bootstrap_event_payload(event), + "data_root": str(data_root), + "token": secrets.token_hex(32), + "identity": asdict(current_promotion_identity(event.cwd)), + } + descriptor = os.open(bootstrap, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + try: + os.write( + descriptor, + json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8"), + ) + os.fsync(descriptor) + finally: + os.close(descriptor) + + executable = Path(sys.argv[0]).resolve() + if executable.suffix == ".pyz": + command = [sys.executable, str(executable), "--serve", str(bootstrap)] + else: + command = [ + sys.executable, + "-m", + "marginal.integrations.codex.service", + "--serve", + str(bootstrap), + ] + environment = { + name: value + for name in ("PATH", "LANG", "LC_ALL", "SYSTEMROOT", "PYTHONPATH") + if (value := os.environ.get(name)) is not None + } + subprocess.Popen( + command, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=environment, + close_fds=True, + start_new_session=True, + ) + deadline = time.monotonic() + 5.0 + while time.monotonic() < deadline: + if existing_path.exists(): + try: + connection = ConnectionInfo.from_file(existing_path) + if request_session(connection, operation="status", payload={}).get("ok") is True: + return connection + except (OSError, ValueError, KeyError): + pass + time.sleep(0.05) + bootstrap.unlink(missing_ok=True) + raise RuntimeError("Codex session service did not become ready") + + +def _serve_bootstrap(path: Path) -> int: + try: + payload = json.loads(path.read_text(encoding="utf-8")) + finally: + path.unlink(missing_ok=True) + event = parse_hook_event(payload["event"]) + if not isinstance(event, SessionEvent) or event.hook_event_name != "SessionStart": + return 2 + data_root = Path(payload["data_root"]).resolve() + token = str(payload["token"]) + identity = PromotionIdentity(**payload["identity"]) + evidence_store = _evidence_store(data_root, identity.repository_hash) + session_hash = _session_hash(event.session_id) + evidence_store.append( + {"schema_version": 1, "event": "session_start", "session_hash": session_hash} + ) + treasury = Treasury(BudgetLimits(), mode="shadow") + universal = UniversalRuntime( + treasury, + engine="codex", + session_id=event.session_id, + task_id=identity.repository_hash, + capabilities=AgentCapabilities(block_actions=True), + ) + runtime = CodexSessionRuntime( + universal, + workspace=event.cwd, + enforcement_enabled=lambda: enforcement_is_active( + data_root, + identity=identity, + summary=summarize_evidence(evidence_store.read_all()), + ), + ) + shutdown_event = threading.Event() + server = SessionServer( + data_root=data_root, + session_id=event.session_id, + token=token, + handler=_handler( + runtime, + evidence_store=evidence_store, + session_hash=session_hash, + data_root=data_root, + identity=identity, + shutdown_event=shutdown_event, + ), + ) + server.start() + shutdown_event.wait() + server.stop() + return 0 + + +def start_session_service( + event: SessionEvent, + *, + data_root: str | Path, +) -> ConnectionInfo: + if event.hook_event_name != "SessionStart": + raise ValueError("start_session_service requires SessionStart") + root = Path(data_root).resolve() + key = (root, event.session_id) + active = _SERVERS.get(key) + if active is not None: + response = request_session(active[0].connection, operation="status", payload={}) + if response.get("ok") is True: + return active[0].connection + active[0].stop() + _SERVERS.pop(key, None) + + identity = current_promotion_identity(event.cwd) + evidence_store = _evidence_store(root, identity.repository_hash) + session_hash = _session_hash(event.session_id) + evidence_store.append( + {"schema_version": 1, "event": "session_start", "session_hash": session_hash} + ) + treasury = Treasury(BudgetLimits(), mode="shadow") + universal = UniversalRuntime( + treasury, + engine="codex", + session_id=event.session_id, + task_id=identity.repository_hash, + capabilities=AgentCapabilities(block_actions=True), + ) + runtime = CodexSessionRuntime( + universal, + workspace=event.cwd, + enforcement_enabled=lambda: enforcement_is_active( + root, + identity=identity, + summary=summarize_evidence(evidence_store.read_all()), + ), + ) + server = SessionServer( + data_root=root, + session_id=event.session_id, + token=secrets.token_hex(32), + handler=_handler( + runtime, + evidence_store=evidence_store, + session_hash=session_hash, + data_root=root, + identity=identity, + ), + ) + connection = server.start() + _SERVERS[key] = (server, runtime) + return connection + + +def stop_session_service(session_id: str, *, data_root: str | Path) -> None: + root = Path(data_root).resolve() + key = (root, session_id) + active = _SERVERS.pop(key, None) + if active is not None: + server, _runtime = active + request_session(server.connection, operation="close", payload={}) + server.stop() + return + path = _connection_path(root, session_id) + if path.exists(): + try: + connection = ConnectionInfo.from_file(path) + request_session(connection, operation="close", payload={}) + finally: + path.unlink(missing_ok=True) + + +def _mode_path(data_root: Path, repository_hash: str) -> Path: + return data_root / "repositories" / f"{repository_hash}.json" + + +def read_mode(data_root: str | Path, *, repository_hash: str) -> dict[str, Any]: + target = _mode_path(Path(data_root).resolve(), repository_hash) + if not target.exists(): + return {"schema_version": 1, "mode": "shadow", "reason": "default"} + payload = json.loads(target.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError("repository mode must be a JSON object") + return payload + + +def _demote_all_enforced(data_root: Path, reason: str) -> None: + repository_root = data_root / "repositories" + if not repository_root.exists(): + return + for path in repository_root.glob("*.json"): + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + continue + if isinstance(payload, dict) and payload.get("mode") == "enforce": + try: + demote_enforcement( + data_root, + repository_hash=path.stem, + reason=reason, + ) + except OSError: + continue + + +def _fail_open_for_workspace(data_root: Path, cwd: str, reason: str) -> None: + repository_hash = repository_identity_hash(cwd) + try: + store = _evidence_store(data_root, repository_hash) + store.append( + { + "schema_version": 1, + "event": "integration_failure", + "reason_code": reason, + "integration_failure": True, + } + ) + store.start_new_window(reason_code=reason) + except OSError: + pass + with suppress(OSError): + demote_enforcement( + data_root, + repository_hash=repository_hash, + reason=reason, + ) + + +def run_hook(payload: dict[str, Any], *, data_root: str | Path) -> HookResult: + """Execute one hook. Integration faults always fail open and demote enforcement.""" + + root = Path(data_root).resolve() + event: SessionEvent | PreToolUseEvent | PostToolUseEvent | None = None + try: + event = parse_hook_event(payload) + if isinstance(event, SessionEvent): + if event.hook_event_name == "SessionStart": + _spawn_session_service(event, data_root=root) + else: + stop_session_service(event.session_id, data_root=root) + return HookResult(exit_code=0) + + connection_path = _connection_path(root, event.session_id) + if not connection_path.exists(): + _fail_open_for_workspace(root, event.cwd, "SERVICE_UNAVAILABLE") + return HookResult(exit_code=0, warning_code="SERVICE_UNAVAILABLE") + connection = ConnectionInfo.from_file(connection_path) + operation = "pre" if isinstance(event, PreToolUseEvent) else "post" + response = request_session(connection, operation=operation, payload=payload) + if response.get("ok") is not True: + code = str(response.get("error_code", "SERVICE_ERROR")) + _fail_open_for_workspace(root, event.cwd, code) + return HookResult(exit_code=0, warning_code=code) + result = response.get("result") + output = result if isinstance(result, dict) else None + return HookResult(exit_code=0, output=output) + except Exception: + if event is not None: + _fail_open_for_workspace(root, event.cwd, "INTEGRATION_ERROR") + else: + _demote_all_enforced(root, "INTEGRATION_ERROR") + return HookResult(exit_code=0, warning_code="INTEGRATION_ERROR") + + +def hook_main(argv: list[str] | None = None) -> int: + """Zipapp entry point used by the native plugin hook shim.""" + + selected = list(sys.argv[1:] if argv is None else argv) + if len(selected) == 2 and selected[0] == "--serve": + return _serve_bootstrap(Path(selected[1]).resolve()) + data_root_value = os.environ.get("PLUGIN_DATA") + if not data_root_value: + return 0 + try: + payload = json.load(sys.stdin) + except (json.JSONDecodeError, UnicodeDecodeError): + return 0 + if not isinstance(payload, dict): + return 0 + result = run_hook(payload, data_root=data_root_value) + if result.output is not None: + print(json.dumps(result.output, sort_keys=True, separators=(",", ":"))) + return result.exit_code + + +if __name__ == "__main__": + raise SystemExit(hook_main()) diff --git a/src/marginal/integrations/codex/state.py b/src/marginal/integrations/codex/state.py new file mode 100644 index 0000000..d40b1d9 --- /dev/null +++ b/src/marginal/integrations/codex/state.py @@ -0,0 +1,90 @@ +"""Privacy-safe Git workspace evidence for the Codex integration.""" + +from __future__ import annotations + +import hashlib +import os +import subprocess +from pathlib import Path +from typing import Protocol + +_IGNORED_PARTS = { + ".codex", + ".git", + ".marginal", + ".mypy_cache", + ".pytest_cache", + ".ruff_cache", + ".tox", + ".venv", + "__pycache__", + "node_modules", +} + + +class _Hash(Protocol): + def update(self, data: bytes, /) -> object: ... + + def hexdigest(self) -> str: ... + + +def _git(repo: Path, *args: str) -> bytes: + environment = { + "PATH": os.environ.get("PATH", ""), + "LANG": "C", + "LC_ALL": "C", + "GIT_CONFIG_NOSYSTEM": "1", + } + completed = subprocess.run( + ["git", *args], + cwd=repo, + env=environment, + check=False, + capture_output=True, + ) + if completed.returncode != 0: + raise ValueError("path is not a usable Git repository") + return completed.stdout + + +def _ignored(path: str) -> bool: + return any(part in _IGNORED_PARTS for part in Path(path).parts) + + +def _update_untracked(digest: _Hash, repo: Path) -> None: + output = _git(repo, "ls-files", "--others", "--exclude-standard", "-z") + for raw_path in sorted(filter(None, output.split(b"\0"))): + path = raw_path.decode("utf-8", errors="surrogateescape") + if _ignored(path): + continue + target = repo / path + if not target.is_file(): + continue + digest.update(b"untracked\0") + digest.update(raw_path) + digest.update(b"\0") + digest.update(target.read_bytes()) + digest.update(b"\0") + + +def workspace_state_hash(workspace: str | Path) -> str: + """Hash material tracked changes and safe untracked content in a Git workspace.""" + + repo = Path(workspace).resolve() + if not repo.is_dir(): + raise ValueError("path is not a usable Git repository") + root = _git(repo, "rev-parse", "--show-toplevel").decode("utf-8").strip() + if Path(root).resolve() != repo: + raise ValueError("path must be the root of a usable Git repository") + + digest = hashlib.sha256() + digest.update(_git(repo, "rev-parse", "HEAD")) + exclusions = [ + ":(exclude).codex/**", + ":(exclude).marginal/**", + ":(exclude).venv/**", + ":(exclude)**/__pycache__/**", + ] + digest.update(_git(repo, "diff", "--binary", "HEAD", "--", ".", *exclusions)) + _update_untracked(digest, repo) + return digest.hexdigest() diff --git a/src/marginal/integrations/codex/transport.py b/src/marginal/integrations/codex/transport.py new file mode 100644 index 0000000..e03d6ec --- /dev/null +++ b/src/marginal/integrations/codex/transport.py @@ -0,0 +1,221 @@ +"""Authenticated, bounded loopback transport for one Codex session.""" + +from __future__ import annotations + +import hashlib +import hmac +import json +import os +import socket +import socketserver +import threading +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +MAX_MESSAGE_BYTES = 256 * 1024 + + +def connection_filename(session_id: str) -> str: + """Return a stable receipt name without exposing the raw Codex session identity.""" + + if not isinstance(session_id, str) or not session_id: + raise ValueError("session_id must be a non-empty string") + digest = hashlib.sha256(session_id.encode("utf-8")).hexdigest() + return f"{digest}.json" + + +@dataclass(frozen=True, slots=True) +class ConnectionInfo: + session_id: str + host: str + port: int + token: str + pid: int + connection_file: Path + + def to_dict(self) -> dict[str, Any]: + payload = asdict(self) + payload["connection_file"] = str(self.connection_file) + return payload + + @classmethod + def from_file(cls, path: str | Path) -> ConnectionInfo: + source = Path(path).resolve() + payload = json.loads(source.read_text(encoding="utf-8")) + return cls( + session_id=str(payload["session_id"]), + host=str(payload["host"]), + port=int(payload["port"]), + token=str(payload["token"]), + pid=int(payload["pid"]), + connection_file=source, + ) + + +SessionHandler = Callable[[str, Mapping[str, Any]], Mapping[str, Any] | None] + + +def _response_bytes(payload: Mapping[str, Any]) -> bytes: + return ( + json.dumps( + dict(payload), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ).encode("utf-8") + + b"\n" + ) + + +class _BoundedRequestHandler(socketserver.StreamRequestHandler): + def handle(self) -> None: + owner: _LoopbackServer = self.server # type: ignore[assignment] + self.connection.settimeout(5.0) + raw = self.rfile.readline(MAX_MESSAGE_BYTES + 2) + if len(raw) > MAX_MESSAGE_BYTES + 1: + self.wfile.write(_response_bytes(_error("MESSAGE_TOO_LARGE"))) + return + try: + request = json.loads(raw) + except (json.JSONDecodeError, UnicodeDecodeError): + self.wfile.write(_response_bytes(_error("INVALID_MESSAGE"))) + return + if not isinstance(request, dict): + self.wfile.write(_response_bytes(_error("INVALID_MESSAGE"))) + return + supplied_token = request.get("token") + if not isinstance(supplied_token, str) or not hmac.compare_digest( + supplied_token, owner.token + ): + self.wfile.write(_response_bytes(_error("AUTH_FAILED"))) + return + operation = request.get("operation") + payload = request.get("payload") + if not isinstance(operation, str) or not isinstance(payload, dict): + self.wfile.write(_response_bytes(_error("INVALID_MESSAGE"))) + return + try: + result = owner.callback(operation, payload) + response: Mapping[str, Any] = {"ok": True, "result": result} + except Exception: + response = _error("SERVICE_ERROR") + self.wfile.write(_response_bytes(response)) + + +class _LoopbackServer(socketserver.ThreadingTCPServer): + allow_reuse_address = False + daemon_threads = True + + def __init__(self, token: str, callback: SessionHandler) -> None: + self.token = token + self.callback = callback + super().__init__(("127.0.0.1", 0), _BoundedRequestHandler) + + +def _error(code: str) -> dict[str, Any]: + return {"ok": False, "error_code": code} + + +class SessionServer: + """Own one authenticated server and its user-private connection receipt.""" + + def __init__( + self, + *, + data_root: str | Path, + session_id: str, + token: str, + handler: SessionHandler, + ) -> None: + if len(token.encode("utf-8")) < 16: + raise ValueError("session token must contain at least 128 bits") + self.data_root = Path(data_root).resolve() + self.data_root.mkdir(parents=True, exist_ok=True, mode=0o700) + self.sessions_root = self.data_root / "sessions" + self.sessions_root.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + self.data_root.chmod(0o700) + self.sessions_root.chmod(0o700) + self._server = _LoopbackServer(token, handler) + self._thread: threading.Thread | None = None + connection_path = self.sessions_root / connection_filename(session_id) + self.connection = ConnectionInfo( + session_id=session_id, + host="127.0.0.1", + port=int(self._server.server_address[1]), + token=token, + pid=os.getpid(), + connection_file=connection_path, + ) + + def start(self) -> ConnectionInfo: + if self._thread is not None: + return self.connection + descriptor = os.open( + self.connection.connection_file, + os.O_CREAT | os.O_TRUNC | os.O_WRONLY, + 0o600, + ) + try: + os.write(descriptor, _response_bytes(self.connection.to_dict())) + os.fsync(descriptor) + finally: + os.close(descriptor) + if os.name == "posix": + self.connection.connection_file.chmod(0o600) + self._thread = threading.Thread( + target=self._server.serve_forever, + name=f"marginal-{self.connection.session_id}", + daemon=True, + ) + self._thread.start() + return self.connection + + def stop(self) -> None: + if self._thread is not None: + self._server.shutdown() + self._thread.join(timeout=5) + self._thread = None + self._server.server_close() + self.connection.connection_file.unlink(missing_ok=True) + + +def request_session( + connection: ConnectionInfo, + *, + operation: str, + payload: Mapping[str, Any], + token: str | None = None, + timeout: float = 5.0, +) -> dict[str, Any]: + """Send one bounded request; transport errors are returned as stable error codes.""" + + request = _response_bytes( + { + "token": connection.token if token is None else token, + "operation": operation, + "payload": dict(payload), + } + ) + if len(request) > MAX_MESSAGE_BYTES + 1: + return _error("MESSAGE_TOO_LARGE") + try: + with socket.create_connection((connection.host, connection.port), timeout=timeout) as sock: + sock.settimeout(timeout) + sock.sendall(request) + reader = sock.makefile("rb") + raw = reader.readline(MAX_MESSAGE_BYTES + 2) + except (OSError, TimeoutError): + return _error("SERVICE_UNAVAILABLE") + if len(raw) > MAX_MESSAGE_BYTES + 1: + return _error("MESSAGE_TOO_LARGE") + try: + response = json.loads(raw) + except (json.JSONDecodeError, UnicodeDecodeError): + return _error("INVALID_RESPONSE") + if not isinstance(response, dict): + return _error("INVALID_RESPONSE") + return response diff --git a/tests/controls/test_progress.py b/tests/controls/test_progress.py new file mode 100644 index 0000000..d0c1e44 --- /dev/null +++ b/tests/controls/test_progress.py @@ -0,0 +1,100 @@ +from __future__ import annotations + +import pytest + +from marginal.controls import ( + ActionOutcomeStatus, + NoProgressConfig, + NoProgressDetector, +) + + +def test_unknown_completions_can_recommend_but_never_enforce() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.UNKNOWN) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.UNKNOWN) + + signal = detector.evaluate("semantic", "state", "evidence") + + assert signal.same_evidence_completions == 2 + assert signal.should_recommend_stop is True + assert signal.enforcement_eligible is False + assert signal.reason_code == "NO_PROGRESS_RECOMMENDED_UNKNOWN" + + +def test_same_successful_evidence_can_be_enforcement_eligible() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + + signal = detector.evaluate("semantic", "state", "evidence") + + assert signal.should_recommend_stop is True + assert signal.enforcement_eligible is True + assert signal.reason_code == "NO_PROGRESS_ENFORCEMENT_ELIGIBLE" + + +def test_one_unknown_completion_keeps_later_sequence_out_of_enforcement() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.UNKNOWN) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + + signal = detector.evaluate("semantic", "state", "evidence") + + assert signal.should_recommend_stop is True + assert signal.enforcement_eligible is False + + +@pytest.mark.parametrize("missing", ["semantic", "state", "evidence"]) +def test_missing_identity_fails_open(missing: str) -> None: + values = {"semantic": "semantic", "state": "state", "evidence": "evidence"} + values[missing] = "" + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=1)) + detector.observe( + values["semantic"], + values["state"], + values["evidence"], + ActionOutcomeStatus.SUCCESS, + ) + + signal = detector.evaluate(values["semantic"], values["state"], values["evidence"]) + + assert signal.same_evidence_completions == 0 + assert signal.should_recommend_stop is False + assert signal.enforcement_eligible is False + assert signal.reason_code == "NO_PROGRESS_UNOBSERVABLE" + + +def test_state_or_evidence_change_resets_pressure() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=1)) + detector.observe("semantic", "state-1", "evidence-1", ActionOutcomeStatus.SUCCESS) + + new_state = detector.evaluate("semantic", "state-2", "evidence-1") + new_evidence = detector.evaluate("semantic", "state-1", "evidence-2") + + assert new_state.same_evidence_completions == 0 + assert new_evidence.same_evidence_completions == 0 + assert new_state.should_recommend_stop is False + assert new_evidence.should_recommend_stop is False + + +def test_evaluate_does_not_advance_history() -> None: + detector = NoProgressDetector(NoProgressConfig(max_same_evidence_completions=2)) + detector.observe("semantic", "state", "evidence", ActionOutcomeStatus.SUCCESS) + + first = detector.evaluate("semantic", "state", "evidence") + second = detector.evaluate("semantic", "state", "evidence") + + assert first == second + assert second.same_evidence_completions == 1 + + +@pytest.mark.parametrize("value", [0, -1, True, 1.5]) +def test_completion_threshold_must_be_a_positive_integer(value: object) -> None: + with pytest.raises((TypeError, ValueError)): + NoProgressConfig(max_same_evidence_completions=value) # type: ignore[arg-type] + + +def test_outcome_status_rejects_unknown_values() -> None: + with pytest.raises(ValueError, match="unknown action outcome status"): + ActionOutcomeStatus.parse("completed") diff --git a/tests/integrations/__init__.py b/tests/integrations/__init__.py new file mode 100644 index 0000000..1192c0f --- /dev/null +++ b/tests/integrations/__init__.py @@ -0,0 +1 @@ +"""Integration tests use package-qualified names to avoid module collisions.""" diff --git a/tests/integrations/codex/__init__.py b/tests/integrations/codex/__init__.py new file mode 100644 index 0000000..ea45e53 --- /dev/null +++ b/tests/integrations/codex/__init__.py @@ -0,0 +1 @@ +"""Codex integration test package.""" diff --git a/tests/integrations/codex/test_commands.py b/tests/integrations/codex/test_commands.py new file mode 100644 index 0000000..176031f --- /dev/null +++ b/tests/integrations/codex/test_commands.py @@ -0,0 +1,137 @@ +from __future__ import annotations + +import json +import subprocess +from pathlib import Path + +from marginal.integrations.codex.commands import codex_command +from marginal.integrations.codex.evidence import EvidenceStore, summarize_evidence +from marginal.integrations.codex.identity import current_promotion_identity +from marginal.integrations.codex.promotion import enforcement_is_active + + +def _git(repo: Path, *args: str) -> None: + subprocess.run(["git", *args], cwd=repo, check=True, capture_output=True) + + +def _repository(path: Path) -> Path: + path.mkdir() + _git(path, "init", "-q") + _git(path, "config", "user.email", "test@example.com") + _git(path, "config", "user.name", "Test User") + (path / "tracked.txt").write_text("initial\n", encoding="utf-8") + _git(path, "add", "tracked.txt") + _git(path, "commit", "-qm", "initial") + return path + + +def test_status_defaults_to_shadow_without_receipt(tmp_path: Path, capsys) -> None: + exit_code = codex_command("status", data_dir=tmp_path, as_json=True) + + payload = json.loads(capsys.readouterr().out) + assert exit_code == 0 + assert payload["mode"] == "shadow" + assert payload["capability"] == "Tool Enforcement" + + +def test_unready_promotion_returns_two(tmp_path: Path, capsys) -> None: + exit_code = codex_command("promote", data_dir=tmp_path, as_json=True) + + payload = json.loads(capsys.readouterr().out) + assert exit_code == 2 + assert payload["error_code"] == "EVIDENCE_NOT_READY" + + +def test_demote_is_idempotent(tmp_path: Path, capsys) -> None: + assert codex_command("demote", data_dir=tmp_path, as_json=True) == 0 + capsys.readouterr() + assert codex_command("demote", data_dir=tmp_path, as_json=True) == 0 + assert json.loads(capsys.readouterr().out)["mode"] == "shadow" + + +def test_review_requires_explicit_candidate_verdict(tmp_path: Path, capsys) -> None: + workspace = _repository(tmp_path / "repo") + identity = current_promotion_identity(workspace) + store = EvidenceStore(tmp_path / "data" / "evidence" / identity.repository_hash) + store.append( + { + "schema_version": 1, + "event": "decision", + "session_hash": "session", + "action_hash": "candidate", + "semantic_key": "semantic", + "state_hash": "state", + "evidence_hash": "evidence", + "reason_code": "NO_PROGRESS_RECOMMENDED_UNKNOWN", + "latency_ms": 1.0, + "covered": True, + "coverable": True, + "recommended_stop": True, + "reviewed": False, + "false_stop": False, + "pending": False, + } + ) + + exit_code = codex_command( + "review", + data_dir=tmp_path / "data", + workspace=workspace, + candidate="candidate", + verdict="waste", + as_json=True, + ) + + assert exit_code == 0 + assert json.loads(capsys.readouterr().out)["reviewed"] is True + summary = summarize_evidence(store.read_all()) + assert summary.reviewed_candidates == 1 + assert summary.false_stops == 0 + + +def test_ready_evidence_promotes_repository_with_live_receipt(tmp_path: Path, capsys) -> None: + workspace = _repository(tmp_path / "repo") + data = tmp_path / "data" + identity = current_promotion_identity(workspace) + store = EvidenceStore(data / "evidence" / identity.repository_hash) + for index in range(100): + store.append( + { + "schema_version": 1, + "event": "decision", + "session_hash": f"session-{index % 5}", + "action_hash": f"action-{index}", + "semantic_key": f"semantic-{index}", + "state_hash": "state", + "evidence_hash": "evidence", + "outcome": "success", + "reason_code": "NO_PROGRESS_OBSERVED" if index < 5 else "APPROVED", + "latency_ms": 1.0, + "covered": True, + "coverable": True, + "recommended_stop": index < 5, + "reviewed": index < 5, + "false_stop": False, + "pending": False, + } + ) + for index in range(5): + store.append( + { + "schema_version": 1, + "event": "session_end", + "session_hash": f"session-{index}", + } + ) + + exit_code = codex_command( + "promote", + data_dir=data, + workspace=workspace, + as_json=True, + ) + + payload = json.loads(capsys.readouterr().out) + assert exit_code == 0 + assert payload["mode"] == "enforce" + assert enforcement_is_active(data, identity=identity) diff --git a/tests/integrations/codex/test_events.py b/tests/integrations/codex/test_events.py new file mode 100644 index 0000000..957304b --- /dev/null +++ b/tests/integrations/codex/test_events.py @@ -0,0 +1,104 @@ +from __future__ import annotations + +import pytest + +from marginal.integrations.codex.events import ( + PostToolUseEvent, + PreToolUseEvent, + SessionEvent, + build_post_tool_output, + build_pre_tool_output, + parse_hook_event, +) + + +def _common(event: str) -> dict[str, object]: + return { + "session_id": "session-1", + "transcript_path": None, + "cwd": "/workspace", + "hook_event_name": event, + "model": "gpt-5.6-sol", + "permission_mode": "default", + } + + +def test_pre_tool_event_requires_complete_tool_identity() -> None: + payload = _common("PreToolUse") + payload.update( + { + "turn_id": "turn-1", + "tool_name": "Bash", + "tool_input": {"command": "pytest -q"}, + } + ) + + with pytest.raises(ValueError, match="tool_use_id"): + parse_hook_event(payload) + + +def test_pre_and_post_events_preserve_typed_lifecycle_fields() -> None: + pre_payload = _common("PreToolUse") + pre_payload.update( + { + "turn_id": "turn-1", + "tool_name": "Bash", + "tool_use_id": "call-1", + "tool_input": {"command": "pytest -q"}, + } + ) + post_payload = {**pre_payload, "hook_event_name": "PostToolUse", "tool_response": "ok"} + + pre = parse_hook_event(pre_payload) + post = parse_hook_event(post_payload) + + assert isinstance(pre, PreToolUseEvent) + assert isinstance(post, PostToolUseEvent) + assert pre.tool_use_id == post.tool_use_id == "call-1" + assert post.tool_response == "ok" + + +@pytest.mark.parametrize( + ("name", "extra"), + [("SessionStart", {"source": "startup"}), ("SessionEnd", {"reason": "other"})], +) +def test_session_events_are_typed(name: str, extra: dict[str, str]) -> None: + event = parse_hook_event({**_common(name), **extra}) + + assert isinstance(event, SessionEvent) + assert event.hook_event_name == name + + +def test_unknown_hook_event_is_rejected() -> None: + with pytest.raises(ValueError, match="unsupported Codex hook event"): + parse_hook_event(_common("UserPromptSubmit")) + + +def test_denial_uses_official_codex_shape() -> None: + assert build_pre_tool_output( + allowed=False, + reason="No progress", + reason_code="NO_PROGRESS", + ) == { + "hookSpecificOutput": { + "hookEventName": "PreToolUse", + "permissionDecision": "deny", + "permissionDecisionReason": "No progress [NO_PROGRESS]", + } + } + + +def test_allow_and_non_blocking_post_emit_no_output() -> None: + assert build_pre_tool_output(allowed=True, reason="allowed", reason_code="ALLOW") is None + assert build_post_tool_output(blocked=False, reason="", reason_code="") is None + + +def test_blocking_post_replaces_result_with_redacted_feedback() -> None: + assert build_post_tool_output( + blocked=True, + reason="Review this result", + reason_code="REVIEW_REQUIRED", + ) == { + "decision": "block", + "reason": "Review this result [REVIEW_REQUIRED]", + } diff --git a/tests/integrations/codex/test_evidence.py b/tests/integrations/codex/test_evidence.py new file mode 100644 index 0000000..f970145 --- /dev/null +++ b/tests/integrations/codex/test_evidence.py @@ -0,0 +1,116 @@ +from __future__ import annotations + +import json +import stat +from pathlib import Path + +import pytest + +from marginal.integrations.codex.evidence import EvidenceStore, summarize_evidence + + +def _record() -> dict[str, object]: + return { + "schema_version": 1, + "event": "decision", + "session_hash": "session-hash", + "action_hash": "action-hash", + "semantic_key": "semantic-key", + "state_hash": "state-hash", + "evidence_hash": "evidence-hash", + "outcome": "unknown", + "reason_code": "ALLOW", + "latency_ms": 1.25, + "covered": True, + "coverable": True, + "recommended_stop": False, + "reviewed": False, + "false_stop": False, + } + + +@pytest.mark.parametrize("field", ["tool_input", "tool_response", "prompt", "command", "source"]) +def test_store_rejects_raw_payload_fields(tmp_path: Path, field: str) -> None: + with pytest.raises(ValueError, match="forbidden evidence field"): + EvidenceStore(tmp_path).append({**_record(), field: "secret"}) + + +def test_store_round_trip_is_canonical_and_private(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + store.append(_record()) + + assert store.read_all() == [_record()] + assert json.loads(store.path.read_text(encoding="utf-8")) == _record() + assert stat.S_IMODE(store.path.stat().st_mode) == 0o600 + + +def test_store_rejects_unknown_fields_and_oversized_records(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path, max_record_bytes=256) + with pytest.raises(ValueError, match="unsupported evidence field"): + store.append({**_record(), "surprise": True}) + with pytest.raises(ValueError, match="too large"): + store.append({**_record(), "reason_code": "X" * 1_000}) + + +def test_checkpoint_is_atomic_canonical_and_private(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + checkpoint = {"schema_version": 1, "mode": "shadow", "receipt_hash": "abc"} + + store.write_checkpoint(checkpoint) + + assert store.read_checkpoint() == checkpoint + assert stat.S_IMODE(store.checkpoint_path.stat().st_mode) == 0o600 + + +def test_redacted_records_build_a_promotion_summary(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + for index in range(100): + store.append( + { + **_record(), + "session_hash": f"session-{index % 5}", + "action_hash": f"action-{index}", + "outcome": "success", + "latency_ms": 2.0, + "recommended_stop": index < 5, + "reviewed": index < 5, + } + ) + for index in range(5): + store.append( + { + "schema_version": 1, + "event": "session_end", + "session_hash": f"session-{index}", + } + ) + + summary = summarize_evidence(store.read_all()) + + assert summary.covered_actions == 100 + assert summary.coverable_actions == 100 + assert summary.completed_sessions == 5 + assert summary.reviewed_candidates == 5 + assert summary.false_stops == 0 + assert summary.enforceable_outcomes_observable is True + + +def test_new_window_preserves_audit_history_but_requires_fresh_evidence(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + store.append( + { + "schema_version": 1, + "event": "integration_failure", + "reason_code": "SERVICE_UNAVAILABLE", + "integration_failure": True, + } + ) + + store.start_new_window(reason_code="SERVICE_UNAVAILABLE") + + records = store.read_all() + summary = summarize_evidence(records) + assert any(record.get("integration_failure") is True for record in records) + assert records[-1]["event"] == "window_start" + assert summary.integration_failures == 0 + assert summary.covered_actions == 0 diff --git a/tests/integrations/codex/test_identity.py b/tests/integrations/codex/test_identity.py new file mode 100644 index 0000000..07b7162 --- /dev/null +++ b/tests/integrations/codex/test_identity.py @@ -0,0 +1,14 @@ +from __future__ import annotations + +import hashlib +from pathlib import Path + +from marginal.integrations.codex.identity import DEFAULT_HOOK_HASH + +REPOSITORY_ROOT = Path(__file__).resolve().parents[3] + + +def test_default_hook_identity_matches_shipped_plugin() -> None: + hook_bytes = (REPOSITORY_ROOT / "plugins" / "marginal" / "hooks" / "hooks.json").read_bytes() + + assert hashlib.sha256(hook_bytes).hexdigest() == DEFAULT_HOOK_HASH diff --git a/tests/integrations/codex/test_installer.py b/tests/integrations/codex/test_installer.py new file mode 100644 index 0000000..1cf06cd --- /dev/null +++ b/tests/integrations/codex/test_installer.py @@ -0,0 +1,94 @@ +from __future__ import annotations + +from dataclasses import dataclass, field + +from marginal.integrations.codex.installer import ( + CommandResult, + inspect_codex, + install, + uninstall, +) + + +@dataclass +class RecordingRunner: + version: str = "codex-cli 0.147.0\n" + hooks: bool = True + plugins: bool = True + calls: list[list[str]] = field(default_factory=list) + + def run(self, args: list[str]) -> CommandResult: + self.calls.append(args) + if args == ["codex", "--version"]: + return CommandResult(0, self.version, "") + if args == ["codex", "features", "list"]: + return CommandResult( + 0, + f"hooks stable {str(self.hooks).lower()}\n" + f"plugins stable {str(self.plugins).lower()}\n", + "", + ) + if args == ["codex", "plugin", "marketplace", "list", "--json"]: + return CommandResult(0, "[]", "") + if args == ["codex", "plugin", "list", "--available", "--json"]: + return CommandResult(0, "[]", "") + return CommandResult(0, "{}", "") + + +def test_discovery_never_reads_auth() -> None: + runner = RecordingRunner() + + report = inspect_codex(runner=runner) + + assert report.capability_level == "tool_enforcement" + assert report.version == "0.147.0" + assert all("auth.json" not in " ".join(call) for call in runner.calls) + + +def test_missing_hooks_refuses_enforcement_claim() -> None: + report = inspect_codex(runner=RecordingRunner(hooks=False)) + + assert report.capability_level == "observe" + assert "HOOKS_UNAVAILABLE" in report.blocking_reasons + + +def test_install_uses_native_codex_plugin_commands() -> None: + runner = RecordingRunner() + + result = install( + runner=runner, + repository="SignalLayerLabs/Marginal", + ref="main", + ) + + assert result.installed + assert [ + "codex", + "plugin", + "marketplace", + "add", + "SignalLayerLabs/Marginal", + "--ref", + "main", + "--json", + ] in runner.calls + assert ["codex", "plugin", "add", "marginal@marginal", "--json"] in runner.calls + + +def test_install_refuses_when_stable_capabilities_are_missing() -> None: + runner = RecordingRunner(plugins=False) + + result = install(runner=runner) + + assert not result.installed + assert result.error_code == "CODEX_CAPABILITIES_UNAVAILABLE" + assert not any(call[1:3] == ["plugin", "add"] for call in runner.calls) + + +def test_uninstall_uses_native_command() -> None: + runner = RecordingRunner() + + result = uninstall(runner=runner) + + assert result.installed is False + assert ["codex", "plugin", "remove", "marginal@marginal", "--json"] in runner.calls diff --git a/tests/integrations/codex/test_marketplace_smoke.py b/tests/integrations/codex/test_marketplace_smoke.py new file mode 100644 index 0000000..17a3d86 --- /dev/null +++ b/tests/integrations/codex/test_marketplace_smoke.py @@ -0,0 +1,26 @@ +from __future__ import annotations + +import shutil +from pathlib import Path + +import pytest +from scripts.smoke_codex_plugin import smoke_plugin + +REPO = Path(__file__).resolve().parents[3] + + +@pytest.mark.skipif(shutil.which("codex") is None, reason="Codex CLI is not installed") +def test_marketplace_install_and_remove(tmp_path: Path) -> None: + result = smoke_plugin( + codex=Path(shutil.which("codex") or "codex"), + isolation_root=tmp_path, + marketplace=REPO, + ) + + assert result.installed is True + assert result.shadow_block_count == 0 + assert result.hook_coverage == 1.0 + assert result.evidence_records >= 4 + assert result.completed_sessions == 1 + assert result.raw_secret_occurrences == 0 + assert result.removed is True diff --git a/tests/integrations/codex/test_normalization.py b/tests/integrations/codex/test_normalization.py new file mode 100644 index 0000000..2c7ea88 --- /dev/null +++ b/tests/integrations/codex/test_normalization.py @@ -0,0 +1,78 @@ +from __future__ import annotations + +import json + +import pytest + +from marginal.integrations.codex.events import PreToolUseEvent +from marginal.integrations.codex.normalization import normalize_pre_tool_use + + +def _event(command: str, *, tool_name: str = "Bash") -> PreToolUseEvent: + return PreToolUseEvent( + session_id="session-1", + cwd="/workspace", + hook_event_name="PreToolUse", + model="gpt-5.6-sol", + permission_mode="default", + turn_id="turn-1", + tool_name=tool_name, + tool_use_id="call-1", + tool_input={"command": command, "description": "private description"}, + ) + + +def test_normalization_never_persists_raw_tool_input() -> None: + action = normalize_pre_tool_use(_event("echo secret-marker"), state_hash="state") + + serialized = json.dumps(action.to_dict(), sort_keys=True) + assert "secret-marker" not in serialized + assert "private description" not in serialized + assert action.name == "Codex Bash action" + assert action.metadata["semantic_key"] + + +def test_canonical_input_order_has_one_semantic_identity() -> None: + first = _event("pytest -q") + second = PreToolUseEvent( + session_id=first.session_id, + cwd=first.cwd, + hook_event_name=first.hook_event_name, + model=first.model, + permission_mode=first.permission_mode, + turn_id=first.turn_id, + tool_name=first.tool_name, + tool_use_id="call-2", + tool_input={"description": "private description", "command": "pytest -q"}, + ) + + normalized_first = normalize_pre_tool_use(first, state_hash="state") + normalized_second = normalize_pre_tool_use(second, state_hash="state") + + assert normalized_first.metadata["semantic_key"] == normalized_second.metadata["semantic_key"] + + +def test_verification_commands_are_classified_without_storing_command() -> None: + action = normalize_pre_tool_use(_event("python -m pytest -q"), state_hash="state") + + assert action.kind == "verification" + assert action.is_verification is True + assert "pytest" not in json.dumps(action.to_dict()) + + +def test_previous_evidence_is_available_to_policy_as_a_hash() -> None: + action = normalize_pre_tool_use( + _event("git status"), + state_hash="state", + previous_evidence_hash="evidence-hash", + ) + + assert action.metadata["evidence_hash"] == "evidence-hash" + + +def test_non_json_tool_input_is_rejected() -> None: + event = _event("git status") + object.__setattr__(event, "tool_input", {"bad": {1, 2}}) + + with pytest.raises(ValueError, match="canonical JSON"): + normalize_pre_tool_use(event, state_hash="state") diff --git a/tests/integrations/codex/test_outcomes.py b/tests/integrations/codex/test_outcomes.py new file mode 100644 index 0000000..d66bab8 --- /dev/null +++ b/tests/integrations/codex/test_outcomes.py @@ -0,0 +1,61 @@ +from __future__ import annotations + +import pytest + +from marginal.controls import ActionOutcomeStatus +from marginal.integrations.codex.events import PostToolUseEvent +from marginal.integrations.codex.outcomes import classify_tool_outcome + + +def _post(response: object) -> PostToolUseEvent: + return PostToolUseEvent( + session_id="session-1", + cwd="/workspace", + hook_event_name="PostToolUse", + model="gpt-5.6-sol", + permission_mode="default", + turn_id="turn-1", + tool_name="Bash", + tool_use_id="call-1", + tool_input={"command": "pytest -q"}, + tool_response=response, + ) + + +@pytest.mark.parametrize( + "response", + [ + "Process exited with code 0", + "success", + {"message": "exit_code=0"}, + {"exit_code": True}, + {"status": "completed"}, + ], +) +def test_prose_and_non_allowlisted_values_remain_unknown(response: object) -> None: + assert classify_tool_outcome(_post(response)) is ActionOutcomeStatus.UNKNOWN + + +@pytest.mark.parametrize( + ("response", "expected"), + [ + ({"exit_code": 0}, ActionOutcomeStatus.SUCCESS), + ({"exit_code": 7}, ActionOutcomeStatus.FAILURE), + ({"success": True}, ActionOutcomeStatus.SUCCESS), + ({"success": False}, ActionOutcomeStatus.FAILURE), + ({"is_error": False}, ActionOutcomeStatus.SUCCESS), + ({"is_error": True}, ActionOutcomeStatus.FAILURE), + ({"status": "success"}, ActionOutcomeStatus.SUCCESS), + ({"outcome": "failure"}, ActionOutcomeStatus.FAILURE), + ], +) +def test_explicit_structured_outcome_is_classified( + response: object, expected: ActionOutcomeStatus +) -> None: + assert classify_tool_outcome(_post(response)) is expected + + +def test_conflicting_structured_signals_fail_open() -> None: + response = {"exit_code": 0, "is_error": True} + + assert classify_tool_outcome(_post(response)) is ActionOutcomeStatus.UNKNOWN diff --git a/tests/integrations/codex/test_promotion.py b/tests/integrations/codex/test_promotion.py new file mode 100644 index 0000000..563593c --- /dev/null +++ b/tests/integrations/codex/test_promotion.py @@ -0,0 +1,149 @@ +from __future__ import annotations + +import json + +from marginal.integrations.codex.promotion import ( + CoverageSummary, + PromotionCriteria, + PromotionIdentity, + PromotionReceipt, + activate_enforcement, + enforcement_is_active, + evaluate_promotion, + read_promotion_receipt, + write_promotion_receipt, +) + + +def _identity(*, policy_hash: str = "policy") -> PromotionIdentity: + return PromotionIdentity( + repository_hash="repository", + codex_version="0.147.0", + plugin_version="0.3.0", + adapter_version="1", + policy_hash=policy_hash, + hook_hash="hooks", + ) + + +def _summary(**overrides: object) -> CoverageSummary: + defaults: dict[str, object] = { + "covered_actions": 100, + "coverable_actions": 100, + "completed_sessions": 5, + "reviewed_candidates": 5, + "false_stops": 0, + "integration_failures": 0, + "pending_actions": 0, + "unknown_enforceable_outcomes": 0, + "decision_latencies_ms": (1.0, 2.0, 3.0), + "enforceable_outcomes_observable": True, + } + defaults.update(overrides) + return CoverageSummary(**defaults) # type: ignore[arg-type] + + +def test_default_gate_requires_minimum_actions() -> None: + receipt = evaluate_promotion( + _summary(covered_actions=99, coverable_actions=100), + PromotionCriteria(), + identity=_identity(), + ) + + assert receipt.is_ready is False + assert "MINIMUM_ACTIONS" in receipt.blocking_reasons + + +def test_all_default_thresholds_produce_ready_receipt() -> None: + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=_identity()) + + assert receipt.is_ready is True + assert receipt.blocking_reasons == () + assert receipt.coverage_ratio == 1.0 + assert receipt.p95_latency_ms == 3.0 + + +def test_each_safety_failure_blocks_promotion() -> None: + cases = { + "MINIMUM_SESSIONS": {"completed_sessions": 4}, + "COVERAGE": {"covered_actions": 98}, + "MINIMUM_REVIEWS": {"reviewed_candidates": 4}, + "FALSE_STOPS": {"false_stops": 1}, + "INTEGRATION_FAILURES": {"integration_failures": 1}, + "PENDING_ACTIONS": {"pending_actions": 1}, + "LATENCY": {"decision_latencies_ms": (76.0,)}, + "OUTCOME_UNOBSERVABLE": {"enforceable_outcomes_observable": False}, + "UNKNOWN_ENFORCEABLE_OUTCOMES": {"unknown_enforceable_outcomes": 1}, + "UNREVIEWED_CANDIDATES": { + "intervention_candidates": 6, + "reviewed_candidates": 5, + }, + } + for reason, overrides in cases.items(): + receipt = evaluate_promotion( + _summary(**overrides), PromotionCriteria(), identity=_identity() + ) + assert reason in receipt.blocking_reasons + + +def test_policy_change_invalidates_ready_receipt() -> None: + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=_identity()) + + assert receipt.valid_for(_identity(policy_hash="new")) is False + + +def test_receipt_round_trip_is_hash_verifiable() -> None: + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=_identity()) + + restored = PromotionReceipt.from_dict(receipt.to_dict()) + + assert restored == receipt + assert restored.verify_hash() + + +def test_active_enforcement_requires_ready_matching_receipt(tmp_path) -> None: + identity = _identity() + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + write_promotion_receipt(tmp_path, receipt) + + activate_enforcement(tmp_path, receipt) + + assert enforcement_is_active(tmp_path, identity=identity) is True + assert read_promotion_receipt(tmp_path, identity.repository_hash) == receipt + + +def test_identity_drift_automatically_demotes_receipt(tmp_path) -> None: + identity = _identity() + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + write_promotion_receipt(tmp_path, receipt) + activate_enforcement(tmp_path, receipt) + + assert ( + enforcement_is_active( + tmp_path, + identity=_identity(policy_hash="changed"), + ) + is False + ) + state = json.loads((tmp_path / "repositories" / f"{identity.repository_hash}.json").read_text()) + assert state["mode"] == "shadow" + assert state["reason"] == "IDENTITY_DRIFT" + + +def test_evidence_drift_automatically_demotes_receipt(tmp_path) -> None: + identity = _identity() + receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + write_promotion_receipt(tmp_path, receipt) + activate_enforcement(tmp_path, receipt) + + assert ( + enforcement_is_active( + tmp_path, + identity=identity, + summary=_summary(integration_failures=1), + ) + is False + ) + state = json.loads((tmp_path / "repositories" / f"{identity.repository_hash}.json").read_text()) + assert state["mode"] == "shadow" + assert state["reason"] == "EVIDENCE_DRIFT" diff --git a/tests/integrations/codex/test_runtime.py b/tests/integrations/codex/test_runtime.py new file mode 100644 index 0000000..64a2261 --- /dev/null +++ b/tests/integrations/codex/test_runtime.py @@ -0,0 +1,162 @@ +from __future__ import annotations + +import subprocess +from pathlib import Path + +import pytest + +from marginal import BudgetLimits, Treasury +from marginal.integrations.codex.events import PostToolUseEvent, PreToolUseEvent +from marginal.integrations.codex.runtime import CodexIntegrationError, CodexSessionRuntime +from marginal.protocol import AgentCapabilities +from marginal.runtime import UniversalRuntime + + +def _git(repo: Path, *args: str) -> None: + subprocess.run(["git", *args], cwd=repo, check=True, capture_output=True) + + +def _repository(path: Path) -> Path: + _git(path, "init", "-q") + _git(path, "config", "user.email", "test@example.com") + _git(path, "config", "user.name", "Test User") + (path / "tracked.txt").write_text("initial\n", encoding="utf-8") + _git(path, "add", "tracked.txt") + _git(path, "commit", "-qm", "initial") + return path + + +def _runtime(workspace: Path, *, enforcement_enabled: bool = False) -> CodexSessionRuntime: + universal = UniversalRuntime( + Treasury(BudgetLimits(max_tokens=100), mode="shadow"), + engine="codex", + session_id="session-1", + task_id="workspace", + capabilities=AgentCapabilities(block_actions=True), + ) + return CodexSessionRuntime( + universal, + workspace=workspace, + enforcement_enabled=lambda: enforcement_enabled, + ) + + +def _pre(action_id: str, *, command: str = "pytest -q") -> PreToolUseEvent: + return PreToolUseEvent( + session_id="session-1", + cwd="/workspace", + hook_event_name="PreToolUse", + model="gpt-5.6-sol", + permission_mode="default", + turn_id="turn-1", + tool_name="Bash", + tool_use_id=action_id, + tool_input={"command": command}, + ) + + +def _post(action_id: str, response: object) -> PostToolUseEvent: + before = _pre(action_id) + return PostToolUseEvent( + session_id=before.session_id, + cwd=before.cwd, + hook_event_name="PostToolUse", + model=before.model, + permission_mode=before.permission_mode, + turn_id=before.turn_id, + tool_name=before.tool_name, + tool_use_id=before.tool_use_id, + tool_input=before.tool_input, + tool_response=response, + ) + + +def test_unknown_post_does_not_advance_success_history(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + + runtime.post_tool_use(_post("call-1", "red test output")) + + assert runtime.summary()["successful_observations"] == 0 + assert runtime.summary()["unknown_observations"] == 1 + assert runtime.pending_action_ids() == () + + +def test_failure_settles_without_success_observation(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + + runtime.post_tool_use(_post("call-1", {"exit_code": 1})) + + assert runtime.summary()["failed_observations"] == 1 + assert runtime.summary()["successful_observations"] == 0 + + +def test_explicit_success_advances_success_history(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + + runtime.post_tool_use(_post("call-1", {"exit_code": 0})) + + assert runtime.summary()["successful_observations"] == 1 + + +def test_identity_mismatch_keeps_original_pending(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + + with pytest.raises(CodexIntegrationError, match="identity"): + runtime.post_tool_use(_post("call-2", {"exit_code": 0})) + + assert runtime.pending_action_ids() == ("call-1",) + + +def test_replayed_pre_identity_is_rejected(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + + with pytest.raises(CodexIntegrationError, match="pending"): + runtime.pre_tool_use(_pre("call-1")) + + +def test_close_aborts_pending_actions_as_unknown(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + runtime.pre_tool_use(_pre("call-1")) + runtime.pre_tool_use(_pre("call-2", command="git status")) + + runtime.close() + + assert runtime.pending_action_ids() == () + assert runtime.summary()["unknown_observations"] == 2 + + +def test_earned_enforcement_denies_third_proven_success_without_progress( + tmp_path: Path, +) -> None: + runtime = _runtime(_repository(tmp_path), enforcement_enabled=True) + for action_id in ("call-1", "call-2"): + decision = runtime.pre_tool_use(_pre(action_id)) + assert decision.allowed + runtime.post_tool_use(_post(action_id, {"exit_code": 0})) + + denied = runtime.pre_tool_use(_pre("call-3")) + + assert denied.allowed is False + assert denied.recommended is False + assert denied.reason_code == "NO_PROGRESS_ENFORCED" + assert runtime.pending_action_ids() == () + assert runtime.summary()["enforced_denials"] == 1 + + +def test_shadow_mode_never_applies_no_progress_denial(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path), enforcement_enabled=False) + for action_id in ("call-1", "call-2"): + runtime.pre_tool_use(_pre(action_id)) + runtime.post_tool_use(_post(action_id, {"exit_code": 0})) + + observed = runtime.pre_tool_use(_pre("call-3")) + + assert observed.allowed is True + assert runtime.last_no_progress_signal is not None + assert runtime.last_no_progress_signal.enforcement_eligible is True + assert runtime.summary()["enforced_denials"] == 0 diff --git a/tests/integrations/codex/test_service.py b/tests/integrations/codex/test_service.py new file mode 100644 index 0000000..49be327 --- /dev/null +++ b/tests/integrations/codex/test_service.py @@ -0,0 +1,275 @@ +from __future__ import annotations + +import json +import subprocess +from dataclasses import asdict, replace +from pathlib import Path + +import marginal.integrations.codex.service as service_module +from marginal.integrations.codex.events import SessionEvent +from marginal.integrations.codex.evidence import EvidenceStore, summarize_evidence +from marginal.integrations.codex.identity import current_promotion_identity +from marginal.integrations.codex.promotion import ( + CoverageSummary, + PromotionCriteria, + activate_enforcement, + evaluate_promotion, + write_promotion_receipt, +) +from marginal.integrations.codex.service import ( + _bootstrap_event_payload, + _bootstrap_path, + read_mode, + run_hook, + start_session_service, + stop_session_service, +) +from marginal.integrations.codex.transport import connection_filename + + +def _git(repo: Path, *args: str) -> None: + subprocess.run(["git", *args], cwd=repo, check=True, capture_output=True) + + +def _repository(path: Path) -> Path: + _git(path, "init", "-q") + _git(path, "config", "user.email", "test@example.com") + _git(path, "config", "user.name", "Test User") + (path / "tracked.txt").write_text("initial\n", encoding="utf-8") + _git(path, "add", "tracked.txt") + _git(path, "commit", "-qm", "initial") + return path + + +def _start(workspace: Path) -> SessionEvent: + return SessionEvent( + session_id="session-1", + cwd=str(workspace), + hook_event_name="SessionStart", + model="gpt-5.6-sol", + permission_mode="default", + source="startup", + ) + + +def _seed_ready_evidence(store: EvidenceStore) -> None: + for session_index in range(5): + session_hash = f"session-{session_index}" + store.append({"schema_version": 1, "event": "session_start", "session_hash": session_hash}) + for action_index in range(20): + ordinal = session_index * 20 + action_index + action_hash = f"action-{ordinal}" + store.append( + { + "schema_version": 1, + "event": "decision", + "session_hash": session_hash, + "action_hash": action_hash, + "latency_ms": 1.0, + "covered": True, + "coverable": True, + "recommended_stop": ordinal < 5, + "reviewed": False, + "false_stop": False, + "pending": True, + } + ) + store.append( + { + "schema_version": 1, + "event": "outcome", + "session_hash": session_hash, + "action_hash": action_hash, + "outcome": "success", + "pending": False, + } + ) + if ordinal < 5: + store.append( + { + "schema_version": 1, + "event": "review", + "session_hash": session_hash, + "action_hash": action_hash, + "reviewed": True, + "false_stop": False, + } + ) + store.append({"schema_version": 1, "event": "session_end", "session_hash": session_hash}) + + +def test_start_is_idempotent_and_end_removes_credentials(tmp_path: Path) -> None: + workspace = tmp_path / "repo" + workspace.mkdir() + _repository(workspace) + data = tmp_path / "data" + first = start_session_service(_start(workspace), data_root=data) + try: + assert start_session_service(_start(workspace), data_root=data) == first + finally: + stop_session_service("session-1", data_root=data) + + assert not first.connection_file.exists() + + +def test_bootstrap_redacts_transcript_and_hashes_session_filename(tmp_path: Path) -> None: + event = replace(_start(tmp_path), transcript_path="/private/raw-transcript.jsonl") + + bootstrap = _bootstrap_path(tmp_path, event.session_id) + payload = _bootstrap_event_payload(event) + + assert event.session_id not in bootstrap.name + assert "transcript_path" not in payload + assert "/private/raw-transcript.jsonl" not in json.dumps(payload) + + +def test_missing_service_fails_open_and_demotes(tmp_path: Path) -> None: + workspace = tmp_path / "repo" + workspace.mkdir() + _repository(workspace) + data = tmp_path / "data" + identity = current_promotion_identity(workspace) + summary = CoverageSummary( + covered_actions=100, + coverable_actions=100, + completed_sessions=5, + reviewed_candidates=5, + false_stops=0, + integration_failures=0, + pending_actions=0, + unknown_enforceable_outcomes=0, + decision_latencies_ms=(1.0,), + enforceable_outcomes_observable=True, + ) + receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + write_promotion_receipt(data, receipt) + activate_enforcement(data, receipt) + pre_payload = { + "session_id": "missing", + "cwd": str(workspace), + "hook_event_name": "PreToolUse", + "model": "gpt-5.6-sol", + "permission_mode": "default", + "turn_id": "turn-1", + "tool_name": "Bash", + "tool_use_id": "call-1", + "tool_input": {"command": "git status"}, + } + + result = run_hook(pre_payload, data_root=data) + + assert result.exit_code == 0 + assert result.output is None + assert read_mode(data, repository_hash=identity.repository_hash)["mode"] == "shadow" + assert result.warning_code == "SERVICE_UNAVAILABLE" + evidence = EvidenceStore(data / "evidence" / identity.repository_hash).read_all() + failure = next(record for record in evidence if record.get("integration_failure") is True) + assert failure["reason_code"] == "SERVICE_UNAVAILABLE" + assert evidence[-1]["event"] == "window_start" + + +def test_fail_open_survives_unavailable_evidence_storage(tmp_path: Path, monkeypatch) -> None: + def unavailable_store(*_args, **_kwargs): + raise OSError("read-only data root") + + monkeypatch.setattr(service_module, "_evidence_store", unavailable_store) + payload = { + "session_id": "missing", + "cwd": str(tmp_path), + "hook_event_name": "PreToolUse", + "model": "gpt-5.6-sol", + "permission_mode": "default", + "turn_id": "turn-1", + "tool_name": "Bash", + "tool_use_id": "call-1", + "tool_input": {"command": "git status"}, + } + + result = run_hook(payload, data_root=tmp_path / "unavailable") + + assert result.exit_code == 0 + assert result.output is None + assert result.warning_code == "SERVICE_UNAVAILABLE" + + +def test_session_start_and_end_are_complete_hook_lifecycle(tmp_path: Path) -> None: + workspace = tmp_path / "repo" + workspace.mkdir() + _repository(workspace) + data = tmp_path / "data" + start_payload = json.loads(json.dumps(asdict(_start(workspace)))) + end_payload = { + **start_payload, + "hook_event_name": "SessionEnd", + "source": None, + "reason": "other", + } + + start_result = run_hook(start_payload, data_root=data) + end_result = run_hook(end_payload, data_root=data) + + assert start_result.exit_code == 0 + assert end_result.exit_code == 0 + assert not (data / "sessions" / connection_filename("session-1")).exists() + + +def test_ready_repository_enforces_proven_no_progress_and_only_that(tmp_path: Path) -> None: + workspace = tmp_path / "repo" + workspace.mkdir() + _repository(workspace) + data = tmp_path / "data" + identity = current_promotion_identity(workspace) + store = EvidenceStore(data / "evidence" / identity.repository_hash) + _seed_ready_evidence(store) + summary = summarize_evidence(store.read_all()) + receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + write_promotion_receipt(data, receipt) + activate_enforcement(data, receipt) + start_payload = json.loads(json.dumps(asdict(_start(workspace)))) + assert run_hook(start_payload, data_root=data).exit_code == 0 + try: + common = { + "session_id": "session-1", + "cwd": str(workspace), + "model": "gpt-5.6-sol", + "permission_mode": "default", + "turn_id": "turn-1", + "tool_name": "Bash", + "tool_input": {"command": "python -m pytest -q"}, + } + for index in (1, 2): + pre = { + **common, + "hook_event_name": "PreToolUse", + "tool_use_id": f"call-{index}", + } + post = { + **pre, + "hook_event_name": "PostToolUse", + "tool_response": {"exit_code": 0}, + } + assert run_hook(pre, data_root=data).output is None + assert run_hook(post, data_root=data).output is None + third = { + **common, + "hook_event_name": "PreToolUse", + "tool_use_id": "call-3", + } + + denied = run_hook(third, data_root=data) + + assert denied.output is not None + hook_output = denied.output["hookSpecificOutput"] + assert hook_output["permissionDecision"] == "deny" + assert "NO_PROGRESS_ENFORCED" in hook_output["permissionDecisionReason"] + finally: + stop_session_service("session-1", data_root=data) + + records = store.read_all() + observed_summary = summarize_evidence(records) + serialized = json.dumps(records) + assert observed_summary.covered_actions == summary.covered_actions + 3 + assert observed_summary.coverable_actions == summary.coverable_actions + 3 + assert observed_summary.completed_sessions == summary.completed_sessions + 1 + assert "python -m pytest" not in serialized + assert "tool_input" not in serialized diff --git a/tests/integrations/codex/test_state.py b/tests/integrations/codex/test_state.py new file mode 100644 index 0000000..1d4a5ec --- /dev/null +++ b/tests/integrations/codex/test_state.py @@ -0,0 +1,50 @@ +from __future__ import annotations + +import subprocess +from pathlib import Path + +from marginal.integrations.codex.state import workspace_state_hash + + +def _git(repo: Path, *args: str) -> None: + subprocess.run(["git", *args], cwd=repo, check=True, capture_output=True) + + +def _repository(tmp_path: Path) -> Path: + _git(tmp_path, "init", "-q") + _git(tmp_path, "config", "user.email", "test@example.com") + _git(tmp_path, "config", "user.name", "Test User") + (tmp_path / "tracked.txt").write_text("initial\n", encoding="utf-8") + _git(tmp_path, "add", "tracked.txt") + _git(tmp_path, "commit", "-qm", "initial") + return tmp_path + + +def test_state_hash_changes_for_material_workspace_progress(tmp_path: Path) -> None: + repo = _repository(tmp_path) + before = workspace_state_hash(repo) + + (repo / "tracked.txt").write_text("changed\n", encoding="utf-8") + + assert workspace_state_hash(repo) != before + + +def test_state_hash_ignores_governor_and_codex_runtime_data(tmp_path: Path) -> None: + repo = _repository(tmp_path) + before = workspace_state_hash(repo) + + for directory in (".marginal", ".codex", ".venv", "__pycache__"): + target = repo / directory + target.mkdir() + (target / "runtime.json").write_text("changed\n", encoding="utf-8") + + assert workspace_state_hash(repo) == before + + +def test_state_hash_rejects_non_repository(tmp_path: Path) -> None: + try: + workspace_state_hash(tmp_path) + except ValueError as exc: + assert "usable Git repository" in str(exc) + else: + raise AssertionError("non-repository path was accepted") diff --git a/tests/integrations/codex/test_transport.py b/tests/integrations/codex/test_transport.py new file mode 100644 index 0000000..929095d --- /dev/null +++ b/tests/integrations/codex/test_transport.py @@ -0,0 +1,75 @@ +from __future__ import annotations + +from collections.abc import Iterator +from contextlib import contextmanager +from pathlib import Path + +from marginal.integrations.codex.transport import ( + MAX_MESSAGE_BYTES, + SessionServer, + connection_filename, + request_session, +) + + +@contextmanager +def _server( + tmp_path: Path, *, token: str = "expected-token-000000000000" +) -> Iterator[SessionServer]: + server = SessionServer( + data_root=tmp_path, + session_id="session-1", + token=token, + handler=lambda operation, payload: {"operation": operation, "payload": payload}, + ) + server.start() + try: + yield server + finally: + server.stop() + + +def test_wrong_token_is_rejected_without_echoing_it(tmp_path: Path) -> None: + with _server(tmp_path) as server: + response = request_session( + server.connection, + operation="status", + payload={}, + token="wrong-secret", + ) + + assert response["ok"] is False + assert response["error_code"] == "AUTH_FAILED" + assert "wrong-secret" not in str(response) + + +def test_authenticated_bounded_request_round_trips(tmp_path: Path) -> None: + with _server(tmp_path) as server: + response = request_session( + server.connection, + operation="status", + payload={"safe": True}, + ) + + assert response == { + "ok": True, + "result": {"operation": "status", "payload": {"safe": True}}, + } + + +def test_oversized_request_is_rejected_client_side(tmp_path: Path) -> None: + with _server(tmp_path) as server: + response = request_session( + server.connection, + operation="status", + payload={"value": "x" * MAX_MESSAGE_BYTES}, + ) + + assert response["error_code"] == "MESSAGE_TOO_LARGE" + + +def test_connection_file_is_user_private(tmp_path: Path) -> None: + with _server(tmp_path) as server: + assert server.connection.connection_file.stat().st_mode & 0o077 == 0 + assert server.connection.connection_file.name == connection_filename("session-1") + assert "session-1" not in server.connection.connection_file.name diff --git a/tests/plugin/test_codex_plugin.py b/tests/plugin/test_codex_plugin.py new file mode 100644 index 0000000..4997332 --- /dev/null +++ b/tests/plugin/test_codex_plugin.py @@ -0,0 +1,89 @@ +from __future__ import annotations + +import hashlib +import json +import subprocess +import sys +from pathlib import Path + +import pytest +from scripts.build_codex_plugin import build_plugin_runtime + +REPO = Path(__file__).resolve().parents[2] +PLUGIN = REPO / "plugins" / "marginal" +MARKETPLACE = REPO / ".agents" / "plugins" / "marketplace.json" +VALIDATOR = Path( + "/Users/renatovinai/.codex/skills/.system/plugin-creator/scripts/validate_plugin.py" +) + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def test_marketplace_points_to_native_plugin() -> None: + marketplace = json.loads(MARKETPLACE.read_text(encoding="utf-8")) + + assert marketplace["name"] == "marginal" + assert marketplace["plugins"][0]["source"] == { + "source": "local", + "path": "./plugins/marginal", + } + + +def test_manifest_is_validator_clean() -> None: + if not VALIDATOR.exists(): + pytest.skip("official Codex plugin validator is not installed") + completed = subprocess.run( + [sys.executable, str(VALIDATOR), str(PLUGIN)], + cwd=REPO, + check=False, + capture_output=True, + text=True, + ) + + assert completed.returncode == 0, completed.stdout + completed.stderr + + +def test_hooks_cover_exact_supported_lifecycle() -> None: + hooks = json.loads((PLUGIN / "hooks" / "hooks.json").read_text(encoding="utf-8")) + + assert set(hooks["hooks"]) == {"SessionStart", "PreToolUse", "PostToolUse", "SessionEnd"} + for groups in hooks["hooks"].values(): + command = groups[0]["hooks"][0] + assert command["type"] == "command" + assert "$PLUGIN_ROOT" in command["command"] + assert command["commandWindows"] + assert command["timeout"] <= 10 + + +def test_generated_runtime_matches_provenance(tmp_path: Path) -> None: + rebuilt = build_plugin_runtime(REPO, output_dir=tmp_path) + provenance = json.loads((PLUGIN / "runtime" / "provenance.json").read_text(encoding="utf-8")) + + assert _sha256(rebuilt.zipapp) == provenance["sha256"] + assert _sha256(PLUGIN / "runtime" / "marginal_runtime.pyz") == provenance["sha256"] + assert rebuilt.source_hash == provenance["source_hash"] + + +def test_plugin_runtime_contains_no_live_repository_paths() -> None: + runtime = (PLUGIN / "runtime" / "marginal_runtime.pyz").read_bytes() + + assert str(REPO).encode() not in runtime + + +def test_skill_teaches_truthful_earned_enforcement_workflow() -> None: + skill_path = PLUGIN / "skills" / "marginal" / "SKILL.md" + text = skill_path.read_text(encoding="utf-8") + frontmatter = text.split("---", 2)[1] + + assert "name: marginal" in frontmatter + assert "description: Use when" in frontmatter + for phrase in ( + "Shadow Mode", + "Tool Enforcement", + "marginal codex status", + "marginal codex promote", + "never claim token savings", + ): + assert phrase.casefold() in text.casefold() diff --git a/tests/plugin/test_publication_packet.py b/tests/plugin/test_publication_packet.py new file mode 100644 index 0000000..ba88837 --- /dev/null +++ b/tests/plugin/test_publication_packet.py @@ -0,0 +1,49 @@ +from __future__ import annotations + +import json +from pathlib import Path + +REPO = Path(__file__).resolve().parents[2] +TEST_CASES = REPO / "docs" / "operations" / "codex-plugin-test-cases.json" +SUBMISSION = REPO / "docs" / "operations" / "codex-plugin-submission.md" + + +def test_readme_and_site_lead_with_install_remove_and_truthful_capability() -> None: + required = ( + "codex plugin marketplace add SignalLayerLabs/Marginal --ref main", + "codex plugin add marginal@marginal", + "codex plugin remove marginal@marginal", + "Tool Enforcement", + "Earned Enforcement", + "Shadow Mode", + "24.93%", + "pass_through", + ) + for path in (REPO / "README.md", REPO / "site" / "index.html"): + text = path.read_text(encoding="utf-8") + for phrase in required: + assert phrase in text, f"{phrase!r} missing from {path}" + + +def test_submission_packet_has_required_positive_and_negative_cases() -> None: + packet = json.loads(TEST_CASES.read_text(encoding="utf-8")) + + assert packet["schema_version"] == 1 + assert len(packet["positive"]) >= 5 + assert len(packet["negative"]) >= 3 + assert all(case["expected"] for case in packet["positive"] + packet["negative"]) + + +def test_submission_status_is_exact_and_not_overclaimed() -> None: + text = SUBMISSION.read_text(encoding="utf-8") + + assert "status: not_submitted" in text + assert "status_date: 2026-08-13" in text + assert "published" not in text.casefold().replace("not published", "") + + +def test_public_legal_and_support_surfaces_exist() -> None: + site = (REPO / "site" / "index.html").read_text(encoding="utf-8") + for filename in ("PRIVACY.md", "TERMS.md", "SUPPORT.md"): + assert (REPO / filename).is_file() + assert filename in site diff --git a/tests/test_cli.py b/tests/test_cli.py index 3b30894..8e31dbb 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -64,3 +64,10 @@ def test_demo_matches_committed_benchmark(capsys) -> None: assert exit_code == 0 assert capsys.readouterr().out == render_markdown(run_benchmark()) + + +def test_codex_status_dispatches_without_importing_at_cli_module_load(tmp_path, capsys) -> None: + exit_code = main(["codex", "status", "--data-dir", str(tmp_path), "--json"]) + + assert exit_code == 0 + assert json.loads(capsys.readouterr().out)["mode"] == "shadow" diff --git a/tests/test_public_api_v2.py b/tests/test_public_api_v2.py index 277a957..27f7dd6 100644 --- a/tests/test_public_api_v2.py +++ b/tests/test_public_api_v2.py @@ -23,7 +23,7 @@ def test_v02_public_exports_and_version() -> None: "ValueEstimate", } assert expected.issubset(set(marginal.__all__)) - assert marginal.__version__ == "0.2.0" + assert marginal.__version__ == "0.3.0" def test_json_schemas_exist_and_are_valid() -> None: diff --git a/tests/test_repository_consistency_v2.py b/tests/test_repository_consistency_v2.py index fa5725d..b465423 100644 --- a/tests/test_repository_consistency_v2.py +++ b/tests/test_repository_consistency_v2.py @@ -20,7 +20,7 @@ def test_public_repository_identity_is_consistent() -> None: assert "SignalLayer Labs" in text or "SignalLayerLabs" in text, path codemeta = json.loads((ROOT / "codemeta.json").read_text(encoding="utf-8")) - assert codemeta["version"] == "0.2.0" + assert codemeta["version"] == "0.3.0" assert codemeta["codeRepository"] == "https://github.com/SignalLayerLabs/Marginal" assert codemeta["issueTracker"] == "https://github.com/SignalLayerLabs/Marginal/issues" assert codemeta["author"]["name"] == "SignalLayer Labs"