From 6e227f62a35d55e46a0b5d684d839c069958cc37 Mon Sep 17 00:00:00 2001 From: Renato Date: Thu, 13 Aug 2026 11:24:31 +0200 Subject: [PATCH 1/2] fix(codex): select a compatible Python runtime --- .github/ISSUE_TEMPLATE/bug_report.yml | 2 +- CHANGELOG.md | 9 +++ CITATION.cff | 2 +- README.md | 5 +- codemeta.json | 2 +- docs/integrations/codex.md | 4 ++ docs/operations/codex-plugin-submission.md | 16 +++-- docs/operations/codex-plugin-test-cases.json | 2 +- .../codex-plugin-smoke-2026-08-13-v0.3.3.json | 17 +++++ plugins/marginal/.codex-plugin/plugin.json | 2 +- plugins/marginal/runtime/marginal_runtime.pyz | Bin 399633 -> 399633 bytes plugins/marginal/runtime/provenance.json | 2 +- plugins/marginal/scripts/marginal_control.py | 11 ++- plugins/marginal/scripts/marginal_hook.py | 9 ++- plugins/marginal/scripts/runtime_python.py | 63 ++++++++++++++++++ plugins/marginal/skills/marginal/SKILL.md | 2 + pyproject.toml | 2 +- scripts/smoke_codex_plugin.py | 9 ++- src/marginal/__init__.py | 2 +- src/marginal/integrations/codex/identity.py | 2 +- .../codex/test_marketplace_smoke.py | 1 + tests/plugin/test_codex_plugin.py | 23 +++++++ tests/test_public_api_v2.py | 2 +- tests/test_repository_consistency_v2.py | 2 +- 24 files changed, 164 insertions(+), 27 deletions(-) create mode 100644 docs/operations/evidence/codex-plugin-smoke-2026-08-13-v0.3.3.json create mode 100644 plugins/marginal/scripts/runtime_python.py diff --git a/.github/ISSUE_TEMPLATE/bug_report.yml b/.github/ISSUE_TEMPLATE/bug_report.yml index 6ea90da..33dbf32 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.yml +++ b/.github/ISSUE_TEMPLATE/bug_report.yml @@ -10,7 +10,7 @@ body: id: version attributes: label: MARGINAL version - placeholder: "0.3.2" + placeholder: "0.3.3" validations: required: true - type: input diff --git a/CHANGELOG.md b/CHANGELOG.md index 2a22d2c..8db83a8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,15 @@ All notable changes to MARGINAL are documented here. The project follows Semanti ## [Unreleased] +## [0.3.3] - 2026-08-13 + +### Fixed + +- native hook and control launchers now replace an incompatible `python3` (including macOS Xcode + Python 3.9) with an available Python 3.10–3.13 interpreter; +- marketplace smoke now invokes the exact public `python3` launcher path and records its version, + preventing isolated test environments from masking interpreter-selection failures. + ## [0.3.2] - 2026-08-13 ### Added diff --git a/CITATION.cff b/CITATION.cff index 143fda5..b615667 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -4,7 +4,7 @@ title: "MARGINAL: Economically Disciplined Compute Allocation for AI Agents" type: software authors: - name: SignalLayer Labs -version: 0.3.2 +version: 0.3.3 date-released: 2026-08-13 license: Apache-2.0 repository-code: "https://github.com/SignalLayerLabs/Marginal" diff --git a/README.md b/README.md index f60de20..33238d8 100644 --- a/README.md +++ b/README.md @@ -213,7 +213,8 @@ codex plugin marketplace add SignalLayerLabs/Marginal --ref main && codex plugin Then open `/hooks` in Codex, review the exact commands, and grant trust only after inspection. MARGINAL never bypasses the hook trust boundary. No Python package or global executable is needed. -In Codex, ask the bundled skill directly: +One local Python 3.10–3.13 interpreter is required; the launcher finds it automatically even when +`python3` points to an older macOS/Xcode runtime. In Codex, ask the bundled skill directly: ```text Use $marginal to report whether a live hook service is active, whether hooks were observed, and whether this repository is in Shadow Mode or Tool Enforcement. @@ -237,7 +238,7 @@ marginal install codex Current tagged library install target: ```bash -pip install "marginal-ai @ git+https://github.com/SignalLayerLabs/Marginal.git@v0.3.2" +pip install "marginal-ai @ git+https://github.com/SignalLayerLabs/Marginal.git@v0.3.3" ``` Development checkout: diff --git a/codemeta.json b/codemeta.json index 796b7f9..74a0073 100644 --- a/codemeta.json +++ b/codemeta.json @@ -6,7 +6,7 @@ "codeRepository": "https://github.com/SignalLayerLabs/Marginal", "issueTracker": "https://github.com/SignalLayerLabs/Marginal/issues", "license": "https://spdx.org/licenses/Apache-2.0", - "version": "0.3.2", + "version": "0.3.3", "datePublished": "2026-08-06", "programmingLanguage": "Python", "runtimePlatform": "Python 3.10-3.13", diff --git a/docs/integrations/codex.md b/docs/integrations/codex.md index 7f57b5a..110f98c 100644 --- a/docs/integrations/codex.md +++ b/docs/integrations/codex.md @@ -14,6 +14,10 @@ Open `/hooks` in Codex and inspect the four MARGINAL lifecycle commands before g plugin never uses the bypass-trust flag. Until trust and runtime coverage are observed, MARGINAL is unobserved or Shadow-only; lack of evidence alone does not prove that hooks are disabled. +The bundle has no third-party Python dependency, but requires one local Python 3.10–3.13 runtime. +Its launchers automatically select a compatible version even when `python3` resolves to macOS/Xcode +Python 3.9. If none is available, control commands return the exact requirement and hooks fail open. + An installed Python package can perform the same native transaction: ```bash diff --git a/docs/operations/codex-plugin-submission.md b/docs/operations/codex-plugin-submission.md index 7b998b4..7c50ab8 100644 --- a/docs/operations/codex-plugin-submission.md +++ b/docs/operations/codex-plugin-submission.md @@ -3,7 +3,7 @@ ```text status: not_submitted status_date: 2026-08-13 -plugin_version: 0.3.2 +plugin_version: 0.3.3 marketplace_selector: marginal@marginal portal_submission_type: skills_only_zip portal_access: authentication_required @@ -32,9 +32,9 @@ and permits repository Tool Enforcement only after a versioned evidence gate pro reviewed false stops, and governance overhead. It never reads prompts, source, raw commands, raw outputs, transcripts, or Codex credentials for evidence. -**Release notes:** Adds a dependency-free native control plane, repository-scoped hook attestation, -local Shadow Mode, review workflow, Earned Enforcement gates, fail-open demotion, production logo, -eight reviewer cases, and a reproducible v0.3.2 archive. +**Release notes:** Adds automatic compatible-Python discovery, a dependency-free native control +plane, repository-scoped hook attestation, local Shadow Mode, review workflow, Earned Enforcement +gates, fail-open demotion, production logo, eight reviewer cases, and a reproducible v0.3.3 archive. Starter prompts are the three `interface.defaultPrompt` entries in `plugins/marginal/.codex-plugin/plugin.json`. The production logo is @@ -63,14 +63,16 @@ coverage or identity changes. - [x] Official plugin validator passes. - [x] Skill validator passes. - [x] Isolated Codex 0.147.0 marketplace add/install/remove smoke passes. +- [x] Public `python3` launcher smoke passes from macOS/Xcode Python 3.9.6 by selecting a compatible + bundled-runtime interpreter. - [x] Four-event direct lifecycle smoke passes with 100% exercised coverage. - [x] Secret-marker scan returns zero persisted occurrences. - [x] Reproducible smoke evidence is committed with runtime SHA-256 provenance. - [x] Privacy, terms, support, and eight reviewer cases exist. -- [ ] Canonical main contains v0.3.2 after merge and every public Pages URL resolves. +- [ ] Canonical main contains v0.3.3 after merge and every public Pages URL resolves. - [x] A deterministic single-root ZIP builder covers the manifest, skill, hooks, runtime, and logo. - [ ] SignalLayer Labs identity and Apps Management write permission are confirmed in Platform. -- [ ] The v0.3.2 archive is uploaded in Platform and external review is started. +- [ ] The v0.3.3 archive is uploaded in Platform and external review is started. Build the exact upload artifact with: @@ -89,4 +91,4 @@ verified SignalLayer Labs identity; none can be inferred from repository or GitH external identifier, credential, or reviewer correspondence belongs in this repository. Acceptance evidence: -[`codex-plugin-smoke-2026-08-13-v0.3.2.json`](evidence/codex-plugin-smoke-2026-08-13-v0.3.2.json). +[`codex-plugin-smoke-2026-08-13-v0.3.3.json`](evidence/codex-plugin-smoke-2026-08-13-v0.3.3.json). diff --git a/docs/operations/codex-plugin-test-cases.json b/docs/operations/codex-plugin-test-cases.json index 5f84509..e719b54 100644 --- a/docs/operations/codex-plugin-test-cases.json +++ b/docs/operations/codex-plugin-test-cases.json @@ -1,6 +1,6 @@ { "schema_version": 1, - "plugin_version": "0.3.2", + "plugin_version": "0.3.3", "positive": [ { "id": "positive-shadow-status", diff --git a/docs/operations/evidence/codex-plugin-smoke-2026-08-13-v0.3.3.json b/docs/operations/evidence/codex-plugin-smoke-2026-08-13-v0.3.3.json new file mode 100644 index 0000000..0bb8181 --- /dev/null +++ b/docs/operations/evidence/codex-plugin-smoke-2026-08-13-v0.3.3.json @@ -0,0 +1,17 @@ +{ + "codex_version": "codex-cli 0.147.0", + "completed_sessions": 1, + "directory_archive_sha256": "42edc01ce21753ce9676426f6d89b0a249539545df0d3c2073c1c060e1b39f7b", + "evidence_records": 4, + "hook_coverage": 1.0, + "installed": true, + "launcher_python_version": "Python 3.9.6", + "marketplace_selector": "marginal@marginal", + "native_control_mode": "shadow", + "native_control_observed": true, + "plugin_version": "0.3.3", + "raw_secret_occurrences": 0, + "removed": true, + "runtime_sha256": "9843b084dd481f29762870e6e6f79a301bdd9c15b7a8c2dc1c4104abc332b259", + "shadow_block_count": 0 +} diff --git a/plugins/marginal/.codex-plugin/plugin.json b/plugins/marginal/.codex-plugin/plugin.json index d1c601a..f0bfda0 100644 --- a/plugins/marginal/.codex-plugin/plugin.json +++ b/plugins/marginal/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "marginal", - "version": "0.3.2", + "version": "0.3.3", "description": "Evidence-gated compute governance for Codex, local-first and Shadow Mode by default.", "author": { "name": "SignalLayer Labs", diff --git a/plugins/marginal/runtime/marginal_runtime.pyz b/plugins/marginal/runtime/marginal_runtime.pyz index 0dafd6c1df9d3c6d9773862f3e719c353539c8a2..c980695c765904140f02744663655d4542821de2 100644 GIT binary patch delta 87 zcmbQZNn+wAi48K0EN7Dh_HI^SEaYP}Zl1)yeG)&TPyvh9qqRKSSJX3RGBO&s3pFwV hF%u9o12M~Xp+?qQ9bkRiKX9_$VS%W=z{lpu1ppPx9k>7h delta 87 zcmbQZNn+wAi48K0EWbD|?b)orSjfj{)I5oQ`y_rwp#m1KhMAMMuc&9tWMnjI7iwe# hVkRJF24a@&LXE7qI>7q2f8b=h!vax#fsf6R3jkI@A3y*A diff --git a/plugins/marginal/runtime/provenance.json b/plugins/marginal/runtime/provenance.json index 464af46..7052df1 100644 --- a/plugins/marginal/runtime/provenance.json +++ b/plugins/marginal/runtime/provenance.json @@ -1 +1 @@ -{"builder":"scripts/build_codex_plugin.py","python_requires":">=3.10","schema_version":1,"sha256":"7ea67b6a17d8f44c03b70e29ec29497a00edd0f9c4262e59cb9082f2015eafe6","source_hash":"ab9dfa18a1ddf72f65a64cc7aba86ca0bd1b4e79e0747c983078e5b8ee87e042"} +{"builder":"scripts/build_codex_plugin.py","python_requires":">=3.10","schema_version":1,"sha256":"9843b084dd481f29762870e6e6f79a301bdd9c15b7a8c2dc1c4104abc332b259","source_hash":"081703fe680da21b5d261dc1828071f8667c73ed74afe58399b33a25cf98360e"} diff --git a/plugins/marginal/scripts/marginal_control.py b/plugins/marginal/scripts/marginal_control.py index 813442d..ea49cfc 100644 --- a/plugins/marginal/scripts/marginal_control.py +++ b/plugins/marginal/scripts/marginal_control.py @@ -7,6 +7,8 @@ import sys from pathlib import Path +from runtime_python import compatible_python + def _plugin_data() -> Path: configured = os.environ.get("PLUGIN_DATA") @@ -37,6 +39,11 @@ def main(argv: list[str] | None = None) -> int: if not runtime.is_file(): print(f"MARGINAL runtime not found: {runtime}", file=sys.stderr) return 1 + try: + python = compatible_python() + except RuntimeError as exc: + print(str(exc), file=sys.stderr) + return 1 data = _plugin_data() data.mkdir(parents=True, exist_ok=True, mode=0o700) @@ -48,9 +55,9 @@ def main(argv: list[str] | None = None) -> int: environment["PLUGIN_DATA"] = str(data) environment["PLUGIN_ROOT"] = str(plugin_root) os.execve( - sys.executable, + python[0], [ - sys.executable, + *python, str(runtime), "codex", arguments[0], diff --git a/plugins/marginal/scripts/marginal_hook.py b/plugins/marginal/scripts/marginal_hook.py index a08a623..34a21df 100644 --- a/plugins/marginal/scripts/marginal_hook.py +++ b/plugins/marginal/scripts/marginal_hook.py @@ -4,9 +4,10 @@ from __future__ import annotations import os -import sys from pathlib import Path +from runtime_python import compatible_python + def main() -> int: plugin_root = os.environ.get("PLUGIN_ROOT") @@ -16,6 +17,10 @@ def main() -> int: runtime = Path(plugin_root).resolve() / "runtime" / "marginal_runtime.pyz" if not runtime.is_file(): return 0 + try: + python = compatible_python() + except RuntimeError: + return 0 environment = { name: value for name in ("PATH", "LANG", "LC_ALL", "SYSTEMROOT") @@ -23,7 +28,7 @@ def main() -> int: } environment["PLUGIN_DATA"] = str(Path(plugin_data).resolve()) environment["PLUGIN_ROOT"] = str(Path(plugin_root).resolve()) - os.execve(sys.executable, [sys.executable, str(runtime)], environment) + os.execve(python[0], [*python, str(runtime)], environment) return 0 diff --git a/plugins/marginal/scripts/runtime_python.py b/plugins/marginal/scripts/runtime_python.py new file mode 100644 index 0000000..de52a3c --- /dev/null +++ b/plugins/marginal/scripts/runtime_python.py @@ -0,0 +1,63 @@ +"""Select a Python interpreter compatible with the bundled MARGINAL runtime.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import sys + +_MINIMUM_VERSION = (3, 10) +_VERSIONED_NAMES = ("python3.13", "python3.12", "python3.11", "python3.10") + + +def _probe(command: tuple[str, ...]) -> bool: + environment = { + name: value + for name in ("PATH", "SYSTEMROOT") + if (value := os.environ.get(name)) is not None + } + environment["PYTHONNOUSERSITE"] = "1" + try: + completed = subprocess.run( + [ + *command, + "-I", + "-c", + "import sys; raise SystemExit(sys.version_info < (3, 10))", + ], + check=False, + stdin=subprocess.DEVNULL, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + env=environment, + timeout=2, + ) + except (OSError, subprocess.TimeoutExpired): + return False + return completed.returncode == 0 + + +def compatible_python() -> tuple[str, ...]: + """Return an executable command for Python 3.10+ without importing user packages.""" + + if sys.version_info[:2] >= _MINIMUM_VERSION: + return (sys.executable,) + + for name in _VERSIONED_NAMES: + executable = shutil.which(name) + if executable: + return (executable,) + + if os.name == "nt": + launcher = shutil.which("py") + if launcher: + for version in ("-3.13", "-3.12", "-3.11", "-3.10"): + command = (launcher, version) + if _probe(command): + return command + + python3 = shutil.which("python3") + if python3 and _probe((python3,)): + return (python3,) + raise RuntimeError("MARGINAL requires Python 3.10 or newer") diff --git a/plugins/marginal/skills/marginal/SKILL.md b/plugins/marginal/skills/marginal/SKILL.md index 239d996..f6953c2 100644 --- a/plugins/marginal/skills/marginal/SKILL.md +++ b/plugins/marginal/skills/marginal/SKILL.md @@ -20,6 +20,8 @@ Do not require a global `marginal` executable or a pip installation. Resolve the Pass `--workspace ` and `--json` when inspecting repository-scoped state. The launcher uses Codex's native plugin data directory, so hook evidence and control commands share one state. +It automatically replaces an older `python3` with an available Python 3.10–3.13 interpreter. If +none is installed, report that exact runtime requirement and do not claim hooks are operational. ## Workflow diff --git a/pyproject.toml b/pyproject.toml index 5c07dcb..17e9257 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "marginal-ai" -version = "0.3.2" +version = "0.3.3" description = "Universal learning-loop and compute-governance foundation for economically disciplined AI agents." readme = "README.md" requires-python = ">=3.10" diff --git a/scripts/smoke_codex_plugin.py b/scripts/smoke_codex_plugin.py index 0e213d9..781d5f0 100644 --- a/scripts/smoke_codex_plugin.py +++ b/scripts/smoke_codex_plugin.py @@ -7,7 +7,6 @@ import json import os import subprocess -import sys import time from dataclasses import asdict, dataclass from pathlib import Path @@ -23,6 +22,7 @@ class CodexPluginSmokeResult: completed_sessions: int native_control_observed: bool native_control_mode: str + launcher_python_version: str raw_secret_occurrences: int removed: bool codex_version: str @@ -131,6 +131,8 @@ def smoke_plugin( } _initialize_repository(workspace, environment) version = _run([str(codex), "--version"], environment=environment, cwd=root).stdout.strip() + launcher = _run(["python3", "--version"], environment=environment, cwd=root) + launcher_python_version = (launcher.stdout or launcher.stderr).strip() _run( [str(codex), "plugin", "marketplace", "add", str(marketplace), "--json"], environment=environment, @@ -161,7 +163,7 @@ def smoke_plugin( try: for payload in _hook_payloads(workspace, secret): result = _run( - [sys.executable, str(hook_script)], + ["python3", str(hook_script)], environment=hook_environment, cwd=workspace, input_text=json.dumps(payload), @@ -177,7 +179,7 @@ def smoke_plugin( time.sleep(0.02) control = _run( [ - sys.executable, + "python3", str(plugin_root / "scripts" / "marginal_control.py"), "status", "--workspace", @@ -219,6 +221,7 @@ def smoke_plugin( ), native_control_observed=native_control_observed, native_control_mode=native_control_mode, + launcher_python_version=launcher_python_version, raw_secret_occurrences=_count_secret(plugin_data, secret), removed=removed, codex_version=version, diff --git a/src/marginal/__init__.py b/src/marginal/__init__.py index 8234636..037a4e3 100644 --- a/src/marginal/__init__.py +++ b/src/marginal/__init__.py @@ -137,4 +137,4 @@ "validate_safe_telemetry_record", ] -__version__ = "0.3.2" +__version__ = "0.3.3" diff --git a/src/marginal/integrations/codex/identity.py b/src/marginal/integrations/codex/identity.py index 488f21b..fdda22c 100644 --- a/src/marginal/integrations/codex/identity.py +++ b/src/marginal/integrations/codex/identity.py @@ -10,7 +10,7 @@ from .installer import CommandRunner, SubprocessRunner, inspect_codex from .promotion import PromotionIdentity -PLUGIN_VERSION = "0.3.2" +PLUGIN_VERSION = "0.3.3" ADAPTER_VERSION = "1" POLICY_HASH = hashlib.sha256(b"marginal:no-progress:v1:max-same-evidence=2").hexdigest() DEFAULT_HOOK_HASH = "46b7a85a3a542957d055c615ab501f9fee284bb3193ac9ecfbe8951cce5a9942" diff --git a/tests/integrations/codex/test_marketplace_smoke.py b/tests/integrations/codex/test_marketplace_smoke.py index 66524db..c38aacb 100644 --- a/tests/integrations/codex/test_marketplace_smoke.py +++ b/tests/integrations/codex/test_marketplace_smoke.py @@ -24,5 +24,6 @@ def test_marketplace_install_and_remove(tmp_path: Path) -> None: assert result.completed_sessions == 1 assert result.native_control_observed is True assert result.native_control_mode == "shadow" + assert result.launcher_python_version.startswith("Python 3.") assert result.raw_secret_occurrences == 0 assert result.removed is True diff --git a/tests/plugin/test_codex_plugin.py b/tests/plugin/test_codex_plugin.py index 1c92fbb..35c965b 100644 --- a/tests/plugin/test_codex_plugin.py +++ b/tests/plugin/test_codex_plugin.py @@ -1,12 +1,14 @@ from __future__ import annotations import hashlib +import importlib.util import json import os import subprocess import sys import zipfile from pathlib import Path +from types import SimpleNamespace import pytest from scripts.build_codex_plugin import build_plugin_runtime @@ -245,6 +247,26 @@ def test_native_control_preserves_codex_home_for_public_cli_discovery(tmp_path: assert payload["plugins_enabled"] is True +def test_runtime_selector_replaces_an_incompatible_launcher_python(monkeypatch) -> None: + module_path = PLUGIN / "scripts" / "runtime_python.py" + spec = importlib.util.spec_from_file_location("marginal_plugin_runtime_python", module_path) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + monkeypatch.setattr( + module, + "sys", + SimpleNamespace(version_info=(3, 9, 6), executable="/usr/bin/python3"), + ) + monkeypatch.setattr( + module.shutil, + "which", + lambda name: "/opt/runtime/python3.11" if name == "python3.11" else None, + ) + + assert module.compatible_python() == ("/opt/runtime/python3.11",) + + def test_plugin_runtime_contains_no_live_repository_paths() -> None: runtime = (PLUGIN / "runtime" / "marginal_runtime.pyz").read_bytes() @@ -265,6 +287,7 @@ def test_skill_teaches_truthful_earned_enforcement_workflow() -> None: "scripts/marginal_control.py", "hooks_active", "hooks_observed", + "Python 3.10", "never claim token savings", ): assert phrase.casefold() in text.casefold() diff --git a/tests/test_public_api_v2.py b/tests/test_public_api_v2.py index 0f052f8..c170c17 100644 --- a/tests/test_public_api_v2.py +++ b/tests/test_public_api_v2.py @@ -23,7 +23,7 @@ def test_v02_public_exports_and_version() -> None: "ValueEstimate", } assert expected.issubset(set(marginal.__all__)) - assert marginal.__version__ == "0.3.2" + assert marginal.__version__ == "0.3.3" def test_json_schemas_exist_and_are_valid() -> None: diff --git a/tests/test_repository_consistency_v2.py b/tests/test_repository_consistency_v2.py index 840a7e2..3fbcf7f 100644 --- a/tests/test_repository_consistency_v2.py +++ b/tests/test_repository_consistency_v2.py @@ -20,7 +20,7 @@ def test_public_repository_identity_is_consistent() -> None: assert "SignalLayer Labs" in text or "SignalLayerLabs" in text, path codemeta = json.loads((ROOT / "codemeta.json").read_text(encoding="utf-8")) - assert codemeta["version"] == "0.3.2" + assert codemeta["version"] == "0.3.3" assert codemeta["codeRepository"] == "https://github.com/SignalLayerLabs/Marginal" assert codemeta["issueTracker"] == "https://github.com/SignalLayerLabs/Marginal/issues" assert codemeta["author"]["name"] == "SignalLayer Labs" From 41e7a7c45edd40d20b5dc0cf764906baf0ae506f Mon Sep 17 00:00:00 2001 From: Renato Date: Fri, 14 Aug 2026 10:39:18 +0200 Subject: [PATCH 2/2] Add evidence-based Codex autonomy controls --- ...6-08-14-evidence-based-autonomy-control.md | 506 +++++++++++++++ ...-evidence-based-autonomy-control-design.md | 290 +++++++++ plugins/marginal/hooks/hooks.json | 14 +- plugins/marginal/runtime/marginal_runtime.pyz | Bin 399633 -> 523971 bytes plugins/marginal/runtime/provenance.json | 2 +- schemas/decision-receipt-v1.json | 69 ++ schemas/governance-ledger-v3.json | 30 + schemas/progress-evidence-v1.json | 25 + schemas/trust-snapshot-v1.json | 33 + src/marginal/authority.py | 115 ++++ src/marginal/canonical.py | 25 + src/marginal/cli.py | 115 +++- src/marginal/diagnostics.py | 456 +++++++++++++ src/marginal/fingerprint.py | 11 +- src/marginal/governance_ledger.py | 607 ++++++++++++++++++ src/marginal/integrations/codex/__init__.py | 13 +- src/marginal/integrations/codex/autopilot.py | 301 +++++++++ src/marginal/integrations/codex/commands.py | 66 +- src/marginal/integrations/codex/events.py | 15 +- src/marginal/integrations/codex/evidence.py | 40 ++ src/marginal/integrations/codex/identity.py | 2 +- src/marginal/integrations/codex/installer.py | 69 +- src/marginal/integrations/codex/intent.py | 175 +++++ .../integrations/codex/normalization.py | 8 + src/marginal/integrations/codex/promotion.py | 71 +- src/marginal/integrations/codex/runtime.py | 111 +++- src/marginal/integrations/codex/service.py | 75 ++- src/marginal/ledger.py | 88 +-- src/marginal/policy.py | 8 +- src/marginal/reason_codes.py | 22 + src/marginal/receipts.py | 322 ++++++++++ src/marginal/schemas/decision-receipt-v1.json | 69 ++ .../schemas/governance-ledger-v3.json | 30 + .../schemas/progress-evidence-v1.json | 25 + src/marginal/schemas/trust-snapshot-v1.json | 33 + src/marginal/trust.py | 451 +++++++++++++ src/marginal/utility.py | 168 +++++ tests/integrations/codex/test_autopilot.py | 303 +++++++++ tests/integrations/codex/test_events.py | 9 +- tests/integrations/codex/test_evidence.py | 37 +- tests/integrations/codex/test_installer.py | 9 + tests/integrations/codex/test_intent.py | 141 ++++ .../codex/test_marketplace_smoke.py | 7 + .../integrations/codex/test_normalization.py | 16 +- tests/integrations/codex/test_promotion.py | 107 ++- tests/integrations/codex/test_runtime.py | 64 +- tests/integrations/codex/test_service.py | 64 +- tests/plugin/test_codex_plugin.py | 8 +- tests/test_authority.py | 93 +++ tests/test_canonical.py | 45 ++ tests/test_cli_v2.py | 18 + tests/test_diagnostics.py | 146 +++++ tests/test_governance_ledger.py | 307 +++++++++ tests/test_ledger_migration_v3.py | 139 ++++ tests/test_packaged_schemas_v2.py | 4 + tests/test_reason_codes.py | 23 + tests/test_receipts.py | 146 +++++ tests/test_trust.py | 244 +++++++ tests/test_utility.py | 92 +++ 59 files changed, 6335 insertions(+), 147 deletions(-) create mode 100644 docs/superpowers/plans/2026-08-14-evidence-based-autonomy-control.md create mode 100644 docs/superpowers/specs/2026-08-14-evidence-based-autonomy-control-design.md create mode 100644 schemas/decision-receipt-v1.json create mode 100644 schemas/governance-ledger-v3.json create mode 100644 schemas/progress-evidence-v1.json create mode 100644 schemas/trust-snapshot-v1.json create mode 100644 src/marginal/authority.py create mode 100644 src/marginal/canonical.py create mode 100644 src/marginal/diagnostics.py create mode 100644 src/marginal/governance_ledger.py create mode 100644 src/marginal/integrations/codex/autopilot.py create mode 100644 src/marginal/integrations/codex/intent.py create mode 100644 src/marginal/reason_codes.py create mode 100644 src/marginal/receipts.py create mode 100644 src/marginal/schemas/decision-receipt-v1.json create mode 100644 src/marginal/schemas/governance-ledger-v3.json create mode 100644 src/marginal/schemas/progress-evidence-v1.json create mode 100644 src/marginal/schemas/trust-snapshot-v1.json create mode 100644 src/marginal/trust.py create mode 100644 src/marginal/utility.py create mode 100644 tests/integrations/codex/test_autopilot.py create mode 100644 tests/integrations/codex/test_intent.py create mode 100644 tests/test_authority.py create mode 100644 tests/test_canonical.py create mode 100644 tests/test_diagnostics.py create mode 100644 tests/test_governance_ledger.py create mode 100644 tests/test_ledger_migration_v3.py create mode 100644 tests/test_reason_codes.py create mode 100644 tests/test_receipts.py create mode 100644 tests/test_trust.py create mode 100644 tests/test_utility.py diff --git a/docs/superpowers/plans/2026-08-14-evidence-based-autonomy-control.md b/docs/superpowers/plans/2026-08-14-evidence-based-autonomy-control.md new file mode 100644 index 0000000..8b625f8 --- /dev/null +++ b/docs/superpowers/plans/2026-08-14-evidence-based-autonomy-control.md @@ -0,0 +1,506 @@ +# Evidence-Based Autonomy Control Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development or +> superpowers:executing-plans task-by-task. Every behavior change uses red-green-refactor. + +**Goal:** Turn MARGINAL's tested compute-governance foundation into an auditable, progressive, +contextual autonomy governor with hash-chained evidence, correctness-first utility, +counterfactual/regret evaluation, safe Codex Autopilot, and fail-closed benchmark provenance. + +**Architecture:** Provider-neutral governance primitives live in focused core modules and are +consumed by Treasury, replay, diagnostics, and thin engine adapters. Codex owns only lifecycle +translation, ephemeral user-intent normalization, local service operation, and capability-limited +Tool Enforcement. Existing v2 APIs remain readable and compatible while new v3 attestations use a +single canonical serializer and hash chain. + +**Tech Stack:** Python 3.10–3.13 standard library, pytest, Ruff 0.16.2, strict mypy, JSON Schema, +Codex 0.147 lifecycle hooks, deterministic zipapp plugin runtime. + +## Global constraints + +- Correctness dominates efficiency and unknown evidence cannot justify enforcement. +- No raw prompt, prompt hash, source, raw command, raw output, transcript, auth, or credential is + persisted by governance code. +- Hook/runtime failures fail open for the agent action and fail closed for governance authority. +- Preserve current public constructors, CLI commands, v2 ledgers, frozen evidence, and Python + 3.10–3.13 compatibility. +- No mandatory runtime dependency. +- No commit, push, tag, release, package publication, marketplace submission, or external + production mutation during this execution. +- Codex reports Tool Enforcement, never Full Compute Enforcement. + +--- + +### Task 1: Canonical serialization and versioned reason codes + +**Files:** +- Create: `src/marginal/canonical.py` +- Create: `src/marginal/reason_codes.py` +- Create: `tests/test_canonical.py` +- Create: `tests/test_reason_codes.py` +- Modify: `src/marginal/fingerprint.py` +- Modify: `src/marginal/policy.py` +- Modify: `src/marginal/controls/progress.py` + +**Produces:** +- `canonical_bytes(value: Any) -> bytes` +- `canonical_hash(value: Any) -> str` +- `ReasonCode(str, Enum)` and `REASON_CODE_VERSION = "1.0"` + +- [x] Write tests proving stable hashes across key order, rejection of NaN/non-JSON values, and + exact documented values for the small reason-code registry. +- [x] Run the focused tests and confirm failure because the modules do not exist. +- [x] Implement canonical compact JSON with `sort_keys=True`, `ensure_ascii=False`, + `allow_nan=False`, and UTF-8 SHA-256. +- [x] Add only used governance codes: approval, insufficient evidence/trust, repeated action, + no progress, user-requested repeat, control-plane bypass, policy revoked, distribution shift, + integrity failure, recovery, and outcome unknown. +- [x] Replace duplicated attestation hashing where compatibility permits; preserve historical hash + formats where changing them would invalidate frozen artifacts. +- [x] Run focused tests, existing fingerprint/policy/progress tests, Ruff, and mypy. + +### Task 2: Decision Receipts and structured governance signals + +**Files:** +- Create: `src/marginal/receipts.py` +- Create: `src/marginal/utility.py` +- Create: `schemas/decision-receipt-v1.json` +- Create: `schemas/progress-evidence-v1.json` +- Mirror: `src/marginal/schemas/decision-receipt-v1.json` +- Mirror: `src/marginal/schemas/progress-evidence-v1.json` +- Create: `tests/test_receipts.py` +- Create: `tests/test_utility.py` +- Modify: `tests/test_packaged_schemas_v2.py` + +**Interfaces:** + +```python +class ProgressLevel(str, Enum): + ACTIVITY = "activity" + INFORMATION = "information" + PROGRESS = "progress" + VERIFIED_PROGRESS = "verified_progress" + + +@dataclass(frozen=True, slots=True) +class ProgressEvidence: + schema_version: str + level: ProgressLevel + state_hash: str + evidence_hash: str + confidence: float + verifier: str | None + + +@dataclass(frozen=True, slots=True) +class GovernanceCost: + wall_clock_ms: float + cpu_ms: float | None + memory_peak_bytes: int | None + storage_bytes: int + tokens: int + model_calls: int + additional_tool_calls: int + + +@dataclass(frozen=True, slots=True) +class DecisionReceipt: + schema_version: str + decision_id: str + timestamp: str + context: Mapping[str, str] + decision: str + reason_code: str + state_hash: str | None + evidence_hash: str | None + trajectory_hash: str | None + policy_hash: str + decision_hash: str + confidence: float + expected_utility: Mapping[str, Any] | None + estimated_cost: Mapping[str, Any] | None + enforcement_level: str + trust_snapshot: Mapping[str, Any] + governance_cost: GovernanceCost +``` + +- [x] Write schema and round-trip tests, including explicit `None` for unavailable measurements, + immutable mappings, invalid confidence, and tampered receipt hash. +- [x] Confirm RED. +- [x] Implement correctness-first `UtilityVector` comparison and `MarginalUtilityEstimate` that + returns a structured scorecard when scalar EMU is scientifically unavailable. +- [x] Implement Decision Receipt canonical payload/hash/verification without arbitrary object + string representations. +- [x] Verify root and packaged schemas are byte-identical and validate real examples. +- [x] Run focused tests, schema suite, Ruff, and mypy. + +### Task 3: Hash-chained governance ledger and v2 migration + +**Files:** +- Create: `src/marginal/governance_ledger.py` +- Create: `schemas/governance-ledger-v3.json` +- Mirror: `src/marginal/schemas/governance-ledger-v3.json` +- Create: `tests/test_governance_ledger.py` +- Create: `tests/test_ledger_migration_v3.py` +- Modify: `src/marginal/ledger.py` +- Modify: `src/marginal/cli.py` + +**Interfaces:** + +```python +@dataclass(frozen=True, slots=True) +class LedgerVerificationReport: + valid: bool + records: int + root_hash: str | None + first_invalid_sequence: int | None + error_codes: tuple[str, ...] + + +class GovernanceLedger: + def append(self, payload: Mapping[str, Any]) -> str: ... + def verify(self, *, expected_root: str | None = None) -> LedgerVerificationReport: ... + + +def migrate_v2_to_v3(source: Path, destination: Path) -> LedgerVerificationReport: ... +def quarantine_invalid_records(source: Path, destination: Path) -> Path: ... +``` + +- [x] Write tests for contiguous append, previous-hash and record-hash linkage, fsync/owner-only + permissions, symlink refusal, tampered payload, deleted middle record, incompatible schema, + expected-root mismatch, non-destructive quarantine, and deterministic v2 migration. +- [x] Confirm RED for each corruption class. +- [x] Implement v3 as a separate ledger so v2 read/write compatibility remains intact. +- [x] Use an OS file lock where available, re-read the tail under lock, append one canonical line, + flush, and fsync. Fail safely on platforms without the lock primitive rather than claiming + multi-process safety. +- [x] Add `marginal verify LEDGER [--expected-root HASH] [--json]` and + `marginal ledger-migrate SOURCE DESTINATION`. +- [x] Run focused tests, existing ledger/privacy tests, CLI tests, Ruff, and mypy. + +### Task 4: Progressive authority and contextual Trust Engine + +**Files:** +- Create: `src/marginal/authority.py` +- Create: `src/marginal/trust.py` +- Create: `schemas/trust-snapshot-v1.json` +- Mirror: `src/marginal/schemas/trust-snapshot-v1.json` +- Create: `tests/test_authority.py` +- Create: `tests/test_trust.py` + +**Interfaces:** + +```python +class AuthorityLevel(IntEnum): + OBSERVE = 0 + ADVISE = 1 + SOFT_INTERVENE = 2 + TOOL_GATE = 3 + COMPUTE_GOVERN = 4 + + +@dataclass(frozen=True, slots=True) +class TrustContext: + repository: str + agent: str + model: str + task_class: str + policy_version: str + + +@dataclass(frozen=True, slots=True) +class TrustEvidence: + observed: int + evaluable: int + covered: int + coverable: int + beneficial: int + neutral: int + harmful: int + indeterminate: int + governance_tax_ratio: float | None + mean_regret: float | None + integrity_valid: bool + last_observed_at: str | None + + +class TrustEngine: + def evaluate( + self, + context: TrustContext, + evidence: TrustEvidence, + current: AuthorityLevel, + *, + capabilities: int, + shift_reasons: tuple[str, ...] = (), + ) -> TrustSnapshot: ... +``` + +- [x] Write transition-table tests for minimum samples, coverage, harm, regret, tax, explicit + capability ceilings, promotion hysteresis, one-level soft decay, critical reset, model/policy + shift, large repository shift, inactivity, and anti-flapping. +- [x] Confirm RED. +- [x] Implement transparent component calculations and blocker lists; do not hide sample size in a + single score. +- [x] Implement transition receipts bound to the evidence-ledger root. +- [x] Run focused tests, Ruff, and mypy. + +### Task 5: Progress proof, governance tax, and Treasury EMU allocation + +**Files:** +- Modify: `src/marginal/controls/progress.py` +- Modify: `src/marginal/controls/governance.py` +- Modify: `src/marginal/models.py` +- Modify: `src/marginal/treasury.py` +- Create: `tests/controls/test_progress_evidence.py` +- Modify: `tests/controls/test_treasury_governance.py` +- Create: `tests/test_treasury_utility.py` + +**Produces:** +- conversion from no-progress observations to `ProgressEvidence`; +- governance p50/p95, CPU, memory, storage, model/tool call accounting; +- `Treasury.rank_candidates(actions, estimates) -> tuple[AllocationScore, ...]` with structured + utility and uncertainty. + +- [ ] Write tests proving activity/new information/progress/verified progress remain distinct, + failure/unknown never create verified progress, and changed evidence resets repetition. +- [ ] Write tests for governance-tax distributions and net scorecards without dividing by zero or + inventing unavailable fields. +- [ ] Write ranking tests where correctness/verification beats cheaper low-value work, uncertainty + lowers confidence, and existing `fund_best` behavior remains unchanged. +- [ ] Confirm RED, implement minimal behavior, and refactor shared validation only after GREEN. +- [ ] Run Treasury, controls, policy, benchmark, Ruff, and mypy suites. + +### Task 6: Counterfactual engine and Intervention Regret + +**Files:** +- Create: `src/marginal/counterfactual.py` +- Create: `schemas/intervention-evaluation-v1.json` +- Mirror: `src/marginal/schemas/intervention-evaluation-v1.json` +- Create: `tests/test_counterfactual.py` +- Create: `tests/test_intervention_regret.py` +- Modify: `src/marginal/outcomes.py` + +**Interfaces:** + +```python +class CounterfactualMode(str, Enum): + LIVE_PAIRED = "live_paired" + REPLAY_APPROXIMATION = "replay_approximation" + + +class InterventionCategory(str, Enum): + BENEFICIAL = "beneficial" + NEUTRAL = "neutral" + HARMFUL = "harmful" + INDETERMINATE = "indeterminate" + + +class PairedBranchRunner(Protocol): + def run_pair(self, spec: PairedBranchSpec) -> PairedBranchResult: ... + + +def evaluate_intervention( + governed: CounterfactualOutcome, comparison: CounterfactualOutcome +) -> InterventionEvaluation: ... +def summarize_regret(items: Sequence[InterventionEvaluation]) -> RegretSummary: ... +``` + +- [ ] Write tests for all four categories, correctness dominance, unavailable outcomes, + structured regret, scalar regret only for comparable measurements, median/mean/high-regret + summaries, and live-pair attestation mismatch rejection. +- [ ] Confirm RED. +- [ ] Implement the provider-neutral live runner contract plus deterministic reference runner for + tests; document that Codex cannot supply live session cloning. +- [ ] Persist evaluations as v3 ledger payloads and validate schemas. +- [ ] Run focused tests, outcome/ledger tests, Ruff, and mypy. + +### Task 7: Replay Lab and offline policy lifecycle + +**Files:** +- Modify: `src/marginal/replay.py` +- Create: `src/marginal/policy_evaluation.py` +- Create: `tests/test_replay_lab.py` +- Create: `tests/test_policy_evaluation.py` +- Modify: `src/marginal/cli.py` + +**Produces:** +- event timeline and candidate/actual intervention report; +- `compare_policies(...) -> PolicyComparison`; +- immutable candidate artifacts and hash-verified promotion/rollback state. + +- [ ] Write replay tests for progress/evidence timeline, earliest/highest-confidence candidates, + later invalidating evidence, explicit replay approximation labels, and two-policy comparison. +- [ ] Write policy tests proving no promotion on missing quality, unknown policy, harmful-rate or + integrity failure; valid candidate promotion writes a receipt; rollback restores the prior hash. +- [ ] Confirm RED. +- [ ] Implement `marginal policy evaluate|compare|promote|rollback` without direct online mutation + of an enforced estimator. +- [ ] Keep `Treasury.observe_value` backward compatible but mark its estimator state as a candidate + training fingerprint unless an explicitly active policy consumes it. +- [ ] Run replay/policy/CLI tests, Ruff, and mypy. + +### Task 8: Codex user-intent normalization and control-plane bypass + +**Files:** +- Create: `src/marginal/integrations/codex/intent.py` +- Modify: `src/marginal/integrations/codex/events.py` +- Modify: `src/marginal/integrations/codex/normalization.py` +- Modify: `plugins/marginal/hooks/hooks.json` +- Create: `tests/integrations/codex/test_intent.py` +- Modify: `tests/integrations/codex/test_events.py` +- Modify: `tests/integrations/codex/test_normalization.py` +- Modify: `tests/integrations/codex/test_marketplace_smoke.py` + +**Interfaces:** + +```python +@dataclass(frozen=True, slots=True) +class UserIntent: + repeat_requested: bool = False + force_run: bool = False + pause_marginal: bool = False + resume_marginal: bool = False + status_requested: bool = False + + +def normalize_user_prompt(prompt: str) -> UserIntent: ... +def is_control_plane_action(event: PreToolUseEvent, plugin_root: Path) -> bool: ... +``` + +- [x] Write Italian/English NFKC/case/whitespace/synonym tests, ambiguous language fail-open tests, + and assertions that prompt text/hash never enters evidence serialization. +- [x] Write control-plane tests accepting only the resolved trusted plugin script path and + rejecting lookalike repository paths, traversal, symlinks, and shell injection. +- [x] Confirm RED. +- [x] Parse official `UserPromptSubmit` events, retain intent only in the authenticated in-memory + session, and add the hook to the deterministic plugin bundle. +- [x] Mark trusted control-plane actions as non-coverable bypass decisions with a stable reason and + no pending workload reservation. +- [x] Run Codex event/normalization/privacy/marketplace tests, rebuild the zipapp, and run check. + +### Task 9: Codex Autopilot, receipt-bound evidence, and recovery + +**Files:** +- Create: `src/marginal/integrations/codex/autopilot.py` +- Modify: `src/marginal/integrations/codex/evidence.py` +- Modify: `src/marginal/integrations/codex/promotion.py` +- Modify: `src/marginal/integrations/codex/runtime.py` +- Modify: `src/marginal/integrations/codex/service.py` +- Modify: `src/marginal/integrations/codex/commands.py` +- Create: `tests/integrations/codex/test_autopilot.py` +- Modify: `tests/integrations/codex/test_evidence.py` +- Modify: `tests/integrations/codex/test_promotion.py` +- Modify: `tests/integrations/codex/test_runtime.py` +- Modify: `tests/integrations/codex/test_service.py` + +**Behavior:** +- deferred one-time Autopilot consent; +- first-session quick receipt bound to a verified evidence root; +- L3 only for exact proven-success no-progress local actions; +- user-requested repeat, polling/waiting, failure/unknown, changed state/evidence, and uncovered + families pass; +- one immediate retry is allowed as recovery and demotes; +- errors, recovery, integrity/capability/identity drift demote and fail open. + +- [x] Write end-to-end state-machine tests for consent, warmup, auto-eligibility, receipt creation, + third-repeat deny, user-intent bypass, recovery, auto-demotion, concurrent pending actions, and + restart persistence. +- [x] Confirm RED. +- [x] Make Codex evidence use or anchor to the v3 chain; promotion verifies root/range before + activation. +- [x] Separate active workload pending counts from unrelated concurrent actions so concurrency does + not spuriously demote while unresolved eligible-family outcomes still fail closed. +- [x] Add actual `avoided_actions` and `recoveries` counters; do not estimate tokens. +- [x] Run the full Codex integration suite, plugin build/check, Ruff, and mypy. + +### Task 10: Unified diagnostics, explainability, and privacy inspection + +**Files:** +- Create: `src/marginal/diagnostics.py` +- Modify: `src/marginal/cli.py` +- Modify: `src/marginal/integrations/codex/commands.py` +- Modify: `src/marginal/integrations/codex/installer.py` +- Create: `tests/test_diagnostics.py` +- Modify: `tests/test_cli_v2.py` +- Modify: `tests/integrations/codex/test_commands.py` +- Modify: `tests/integrations/codex/test_installer.py` + +**Commands:** +- `marginal status [--json]` +- `marginal doctor [--json]` +- `marginal explain DECISION_ID [--json]` +- `marginal privacy inspect [--json]` + +- [x] Write output-contract tests for authority/current eligibility, trust components, exact next + promotion blockers, ledger integrity, plugin/runtime provenance, permissions, benchmark + readiness, deterministic decision explanation, and persisted data categories. +- [x] Confirm RED. +- [x] Implement typed report objects shared by human/JSON renderers. +- [x] Keep `marginal codex status|doctor` as compatible aliases. +- [x] Add install-time Autopilot consent configuration without allowing repository configuration to + raise user-level authority. +- [x] Run CLI/installer/privacy/integration tests, Ruff, and mypy. + +### Task 11: Benchmark provenance and correctness-first public metrics + +**Files:** +- Modify: `benchmark/codex_adapter/evidence.py` +- Modify: `benchmark/codex_adapter/runner.py` +- Modify: `benchmark/codex_adapter/container_runner.py` +- Modify: `benchmarks/swebench_lite/protocol.py` +- Modify: `benchmarks/swebench_lite/merge_results.py` +- Modify: `src/marginal/public_eval.py` +- Create: `benchmark/schemas/evidence-bundle-v2.json` +- Modify: `tests/evaluation/test_swebench_lite_protocol.py` +- Modify: `tests/evaluation/test_swebench_lite_merge.py` +- Modify: `tests/test_public_eval_v2.py` + +- [ ] Write fail-closed tests for run records, provenance, verifier digests, merged rows, public + artifacts, patch/configuration hashes, pair mismatch, starting SHA, task mismatch, execution + order, and missing files. +- [ ] Confirm RED without altering frozen evidence. +- [ ] Add a versioned evidence-DAG manifest for future runs and a compatibility validator for the + existing frozen v1 smoke. +- [ ] Add input/cached/output/reasoning metrics, paired classes, intervention categories, regret, + candidate counts, and provenance hashes to public evaluation. +- [ ] Report `insufficient_correctness_evidence` for zero-resolved paired samples and preserve + `pass_through`. +- [ ] Generate public summaries programmatically and verify the frozen JSON remains traceable. +- [ ] Run benchmark/evaluation tests, Ruff, and mypy. + +### Task 12: Security, performance, documentation, and release readiness + +**Files:** +- Create: `tests/performance/test_governor_performance.py` +- Create: `tests/adversarial/test_productive_repetition.py` +- Create: `tests/adversarial/test_partial_streams.py` +- Modify: `.github/workflows/release.yml` +- Modify: `README.md`, `CHANGELOG.md`, `SECURITY.md`, `ROADMAP.md` +- Create: `docs/adr/0001-earned-authority.md` +- Create: `docs/adr/0002-decision-receipt-ledger.md` +- Create: `docs/adr/0003-contextual-trust.md` +- Create: `docs/adr/0004-counterfactual-and-regret.md` +- Create: `docs/adr/0005-plugin-core-boundary.md` +- Create: `docs/product/decision-receipts.md` +- Create: `docs/product/trust-and-authority.md` +- Create: `docs/evaluation/counterfactual-and-regret.md` +- Create: `docs/evaluation/replay-and-policy-lab.md` +- Create: `docs/reference/reason-codes.md` +- Modify: `docs/integrations/codex.md` +- Modify: `docs/integrations/codex-benchmark-readiness.md` +- Modify: `benchmark/analysis/summary.md` + +- [ ] Add adversarial tests for productive repetition, long debugging, slow eventual progress, + changed-state verification, oscillating exploration, new repository, policy/model shift, + corrupted ledger, clock anomaly, and partial streams. +- [ ] Add a deterministic governor microbenchmark measuring throughput, p50/p95 event and decision + latency, CPU, memory peak, storage, and ledger growth with conservative local regression limits. +- [ ] Change release creation to explicit manual/tag-gated approval; do not create a release. +- [ ] Update product positioning to Evidence-Based Agent Governance and document every implemented + CLI, schema, authority rule, privacy category, security boundary, migration, and external blocker. +- [ ] Reconcile 10-versus-20 canary wording and stale zero-run analysis while preserving historical + provenance. +- [ ] Regenerate plugin runtime and submission archive locally without uploading it. +- [ ] Run the full final gate: Ruff format/check, strict mypy, full pytest, plugin build check, + package build, Twine check, CLI smoke, governor benchmark, secret/path scan, and git diff review. diff --git a/docs/superpowers/specs/2026-08-14-evidence-based-autonomy-control-design.md b/docs/superpowers/specs/2026-08-14-evidence-based-autonomy-control-design.md new file mode 100644 index 0000000..14e3ca0 --- /dev/null +++ b/docs/superpowers/specs/2026-08-14-evidence-based-autonomy-control-design.md @@ -0,0 +1,290 @@ +# Evidence-Based Autonomy Control Design + +**Status:** Approved execution direction +**Date:** 2026-08-14 +**Scope:** Local implementation only; no commit, push, release, package publication, or marketplace submission + +## 1. Product contract + +MARGINAL is an evidence-based governor for AI agents. It observes trajectories, measures verified +progress relative to compute, records auditable decisions, evaluates its own interventions, and +earns only the authority supported by local evidence. + +The governing sequence is: + +```text +Observe → Measure → Prove → Earn Authority → Revalidate → Revoke on degradation +``` + +Correctness dominates savings. Unknown evidence fails toward observation. The Codex integration +can claim Tool Enforcement only; hosted and specialized tool paths remain outside its complete +control surface. + +## 2. Verified baseline + +Before this design, the isolated plugin worktree passed: + +- 449 pytest tests; +- Ruff format and lint; +- strict mypy over 43 source files; +- deterministic Codex plugin runtime verification; +- Codex doctor for CLI 0.147.0 with hooks and plugins available. + +The frozen three-task SWE-bench smoke remains an integration observation: both lanes resolved 0/3, +ON used 24.93% fewer measured tokens, no denial occurred, and the intervention status is +`pass_through`. It is not efficacy evidence. + +## 3. Existing architecture to preserve + +The implementation extends, rather than duplicates: + +- `protocol.py` for provider-neutral agent lifecycle values; +- `models.py`, `policy.py`, and `treasury.py` for actions, cost, decisions, and allocation; +- `controls/` for repetition, progress, and governance overhead; +- `ledger.py` for the existing v2 decision ledger and privacy exports; +- `replay.py` and `public_eval.py` for explicitly non-causal replay and paired evaluation; +- `integrations/codex/` for the production Codex anti-corruption layer; +- `benchmark/` and `benchmarks/swebench_lite/` for frozen experiment orchestration and evidence. + +Compatibility requirements: + +- Python 3.10–3.13; +- no mandatory third-party runtime dependency; +- v0.1/v0.2 constructors and v2 ledger readers remain supported; +- raw prompts, source, commands, outputs, transcripts, auth, and credentials are never persisted; +- hook/runtime failures allow the agent action but revoke governance authority; +- evidence or provenance corruption fails closed for promotion and claims. + +## 4. Target module boundaries + +New provider-neutral modules: + +- `canonical.py`: one deterministic JSON serializer and SHA-256 primitive; +- `reason_codes.py`: small, versioned reason-code registry; +- `authority.py`: L0–L4 semantics, transitions, and hysteresis policy; +- `trust.py`: contextual evidence, confidence components, eligibility, decay, and shift handling; +- `receipts.py`: immutable Decision Receipts and transition receipts; +- `governance_ledger.py`: append-only hash-chained v3 ledger and verification reports; +- `utility.py`: progress evidence, structured correctness-first utility, and EMU estimates; +- `counterfactual.py`: live/replay branch contracts, intervention evaluation, and regret; +- `policy_evaluation.py`: offline candidate comparison, promotion receipts, active state, rollback; +- `diagnostics.py`: shared status, doctor, explain, verify, and privacy inspection models. + +Codex-specific additions stay under `integrations/codex/`: + +- `intent.py`: ephemeral normalization of user control and repeat intent; +- `autopilot.py`: first-session quick eligibility, deferred opt-in, recovery, and revocation; +- existing event, service, evidence, promotion, and command modules adapt to core authority/receipt + contracts without reimplementing them. + +## 5. Universal event and receipt model + +Existing `AgentEvent` remains valid. New stable entities complement it: + +- `EvidenceSignal` identifies a derived evidence kind and hash; +- `ProgressEvidence` distinguishes activity, information, progress, and verified progress; +- `VerificationSignal` records an observable verifier result or explicit unavailable state; +- `GovernanceCost` records wall-clock, CPU, memory peak, storage delta, tokens, model calls, and + added tool calls where measurable; +- `TrustContext` keys evidence by repository, agent, model, task class, and policy version, using + explicit `unknown` values where unavailable; +- `TrustSnapshot` exposes components, sample sizes, confidence band, eligible/current authority, + and blockers; +- `DecisionReceipt` binds decision, hashes, trust, cost, and enforcement level; +- `CounterfactualOutcome` and `InterventionEvaluation` bind governed and comparison outcomes. + +Every hash uses canonical UTF-8 JSON with sorted keys, compact separators, no NaN, and explicit +schema versions. Arbitrary `repr` output is never attestation material. + +## 6. Hash-chained governance ledger + +The existing v2 ledger remains readable. A new v3 governance ledger stores receipts, outcomes, +authority transitions, policy transitions, counterfactual references, and governance cost. + +Each line contains: + +```text +schema_version +sequence +timestamp +previous_hash +payload +record_hash = SHA256(canonical(envelope without record_hash)) +``` + +The verifier checks contiguous sequences, previous-hash linkage, record hashes, receipt hashes, +schema compatibility, context/provenance identity, and optional expected root. Invalid ledgers are +never silently truncated. A quarantine command copies invalid records plus a report into an +owner-only quarantine directory and preserves the source. + +Promotion receipts bind to the verified evidence window root, first/last sequence, record count, +identity, trust snapshot, criteria, and allowed enforcement scope. Editing evidence invalidates the +receipt. + +## 7. Progressive authority and contextual trust + +Authority levels: + +- L0 `OBSERVE`: record counterfactual decisions only; +- L1 `ADVISE`: surface structured advice without changing execution; +- L2 `SOFT_INTERVENE`: request reconsideration; agent retains control; +- L3 `TOOL_GATE`: deny explicitly eligible local tool classes; +- L4 `COMPUTE_GOVERN`: allocate/deny candidate compute through an adapter that proves the needed + capabilities. Codex is not eligible for L4. + +Trust is a scorecard, not an opaque scalar. It records observed/evaluable decisions, coverage, +verified outcomes, beneficial/harmful/neutral/indeterminate interventions, false stops, regret, +governance tax, calibration error, recency, and shift indicators. + +Promotion and demotion use different thresholds. Promotion requires minimum samples, coverage, +bounded harm/regret/tax, integrity, and capability. Demotion occurs at lower evidence quality, +identity changes, integrity failure, harmful recovery, or prolonged inactivity. Critical integrity +or capability drift returns directly to L0; non-critical decay steps down one level. + +## 8. Progress and verified utility per compute + +`ProgressEvidence` never equates activity with progress. It reports separately: + +- activity completed; +- new information acquired; +- task state advanced; +- verified requirement satisfied. + +`UtilityVector` is ordered lexicographically: + +1. verified correctness; +2. task completion; +3. safety/risk; +4. latency; +5. tokens and monetary cost; +6. governance overhead. + +Unknown correctness cannot be traded for lower compute. `MarginalUtilityEstimate` carries expected +incremental verified utility, cost, uncertainty, confidence, and provenance. A scalar ratio is +reported only when its inputs are commensurable; otherwise a structured scorecard is returned. + +Treasury keeps its current API and gains an auditable candidate-ranking path that consumes these +estimates. Existing scalar policy behavior remains backward compatible. + +## 9. Counterfactual evaluation and regret + +Two explicitly labelled modes are supported: + +- `LIVE_PAIRED`: a provider supplies two independently continued branches from an attested common + start. The core validates comparability but does not pretend Codex hooks can clone a session. +- `REPLAY_APPROXIMATION`: recorded proposals are re-evaluated. It is non-causal and cannot simulate + missing state changes. + +`InterventionEvaluation` compares governed and counterfactual utility correctness-first and emits +`BENEFICIAL`, `NEUTRAL`, `HARMFUL`, or `INDETERMINATE`. Regret preserves a structured delta and only +provides a scalar when justified. Aggregate reports include reviewed count, category rates, +mean/median scalar regret when available, and high-regret incidents. + +## 10. Offline policy lifecycle + +Production observations never mutate an enforced policy directly. Calibration produces an +immutable candidate artifact with source-ledger root, training fingerprint, configuration hash, +and evaluation results. + +The lifecycle is: + +```text +evidence → candidate → replay → benchmark/counterfactual gates → explicit or pre-authorized +promotion → active policy → rollback +``` + +Policy promotion writes a hash-chained receipt. Rollback restores the previous validated policy. +Newer versions receive no automatic authority. + +## 11. Codex Autopilot and user intent + +Codex remains the reference adapter and uses official lifecycle hooks only. + +The plugin adds `UserPromptSubmit`. User text is normalized in memory using Unicode NFKC, +case-folding, whitespace collapse, and a bounded Italian/English control vocabulary. Only ephemeral +turn flags are retained: + +- `repeat_requested`; +- `force_run`; +- `pause_marginal`; +- `resume_marginal`; +- `status_requested`. + +Neither prompt nor prompt hash is persisted. Ambiguity fails open. + +Autopilot lifecycle: + +1. user grants hook trust and one-time deferred promotion consent; +2. repository starts at L0 and observes a short clean window; +3. the quick profile may reach L3 only for exact successful no-progress repetitions, with 100% + observed coverage inside that eligible family, observable outcomes, bounded latency, no + failures, and an integrity-valid receipt; +4. the third equivalent action may be denied only when input, state, completion evidence, and user + intent all support it; +5. polling, waiting, monitoring, unknown/failure outcomes, changed state/evidence, verification + requested by the user, and uncovered tools always pass; +6. an immediate identical retry after denial is allowed once as recovery and demotes authority; +7. integration failure, false stop, recovery, integrity drift, identity drift, or coverage loss + demotes and fails open. + +MARGINAL control-plane commands are recognized in memory from the trusted installed script path +and bypass workload governance so promotion cannot block itself. They are not candidates or +pending workload actions. + +Normal users do not run review/promote commands. Advanced manual Earned Enforcement remains +available. Status reports actual avoided actions and recoveries, never invented token savings. + +## 12. Diagnostics and CLI + +The coherent CLI surface is: + +```text +marginal status +marginal doctor +marginal explain +marginal verify +marginal replay [--profile ...] [--compare-profile ...] +marginal policy evaluate|compare|promote|rollback +marginal privacy inspect +``` + +Existing commands remain aliases/compatible. Human and JSON output share typed report objects. +Doctor checks runtime, plugin artifact/provenance, hooks, permissions, schemas, ledger integrity, +policy identity, authority/trust, and benchmark readiness. + +## 13. Benchmark and provenance + +The frozen smoke is immutable. New validation verifies the full evidence DAG: run records, +provenance, verifier reports, merged verified rows, predictions, metrics, public JSON/Markdown, +patch hashes, configuration hashes, pair identity, and execution-order evidence. + +Public evaluation adds token breakdown, paired outcome classes, intervention categories, regret, +candidate counts, and provenance hashes. At 0/3 vs 0/3 it reports insufficient correctness evidence, +not an easily misread quality-preserved boolean. + +Governor performance benchmarks measure decision/event p50/p95, CPU, memory peak, storage/ledger +growth, and throughput. Thresholds guard regressions but are not generalized product claims. + +## 14. Security, privacy, and configuration + +- Repositories are untrusted input; governance never executes repository code by itself. +- Paths reject symlinks and traversal; ledgers are owner-only and durably fsynced. +- Global user policy caps repository policy authority. Repository configuration can reduce, never + increase, authority. +- Hook commands remain reviewable and never bypass Codex trust. +- Local evidence uses derived enums, counts, and hashes with documented dictionary-attack limits; + secret-bearing raw values are excluded. +- Release workflows require an explicit dispatch/tag and approval; no push to `main` creates a + release automatically. + +## 15. Testing and completion evidence + +Behavior-critical work follows red-green-refactor. Required suites cover canonicalization, +schemas, chain corruption/missing records/migration, authority transitions/hysteresis/decay, +context shifts, structured utility, counterfactual/regret, offline policy promotion/rollback, +productive repetition, user-intent bypass, Autopilot recovery, control-plane bypass, partial event +streams, concurrency, privacy, provenance, and performance. + +Final local gates are pytest, Ruff, strict mypy, plugin runtime reproducibility, build, Twine check, +CLI smoke, governor benchmark, and dirty-worktree review. External inference, marketplace review, +push, tag, release, and publication remain explicit blockers rather than simulated completion. diff --git a/plugins/marginal/hooks/hooks.json b/plugins/marginal/hooks/hooks.json index 5cee9db..8702011 100644 --- a/plugins/marginal/hooks/hooks.json +++ b/plugins/marginal/hooks/hooks.json @@ -14,6 +14,19 @@ ] } ], + "UserPromptSubmit": [ + { + "hooks": [ + { + "type": "command", + "command": "python3 \"$PLUGIN_ROOT/scripts/marginal_hook.py\"", + "commandWindows": "py -3 \"%PLUGIN_ROOT%\\scripts\\marginal_hook.py\"", + "timeout": 5, + "statusMessage": "Reading explicit MARGINAL controls" + } + ] + } + ], "PreToolUse": [ { "matcher": "Bash|apply_patch|MCP|Read|Write|Edit", @@ -57,4 +70,3 @@ ] } } - diff --git a/plugins/marginal/runtime/marginal_runtime.pyz b/plugins/marginal/runtime/marginal_runtime.pyz index c980695c765904140f02744663655d4542821de2..2118d4ba4b65a5f0937317a083930ce9bd36b41f 100644 GIT binary patch delta 88186 zcmd4433yybl`rndb>xU-{F*`(J2%>(O(st@oEqnL_`SPnj}hB%C-HjfIDo zhsRQf;)!VLc*E#%zu(`Th#yQulF8`Nh%XWwiYEplBav9jHyk+{8BY4bu|eOVaPm-n ze|!u-t;)VsA{g#Fg>}~IB@9GTowr+3R)e_p%)|2kVv#;6bU(wL$ zFAi;QyBZG|()E#8GLj1T8&|AcwR+9kb?Y~T`v(RiL;hfKaq+-#IGOadnDy$Q6$xNQ zZLzVDU{kRVe^|iw*gzr@PDWz~eWUSX7;7r-3lFF8BOLRE4`MBsCVlD1b&-KF8gK)u z6m#9KD|*{{_OxLjjr`oQWlwuAJ+9!#-ma~Eq4v%`D$_}?SMuw=uC9*Iww6A6vx?ue zcJ1ok-Pab{)`f}EBv)hXn1e5rh#VPUV~Kq75l^kG z=o4iH$wT3lYu4iHqJxeas1P_B9v+L-`C{RbNRYdV4goG=>V42x+w5Dr+83vZ9uEv0 z3Maw?H0@YC<%`CAyvTqjpZx`Ixk-6?pO_~PREm%xzvK27>eZ4hpYj9?)Pp=17Yk&^^X_TZ*JH2X&3((o zoaxEP@KA%Ul$mv^LSFV$_q5COi@iAjc{p;s&ga)3eR%Q*_XFUwO39WsQ7YemuiIN* z=NmGG_ntuM+Qg~-{!GberToQD-8E)(;!#hz^!>~|tw=W<%!ohLR+??C%idP`!1E}+ z?wjrzGX1!F`s|#~!}2qYo&_1-=cdI>1B^ol6dd?Q#*!&te7IbA7O!;s7v6!2tbIqg5(@6u) zb#&001Wg1df6~$f>5T9R2q&GX)f^++%0sjR08$qN&J89Ub=|?<76?2(R%Epfn!8Y1rESLjC48 zmnfGtKXK0^)LSSI{76if&wSM5nVyIYM53c=8RWCSb1$Cecq30Q^OR5J5fLVrwFL!p z<%=H_B`qKyJ(1K{BE}faBA>pYXk>UW=^KcTj6{I~2Yvm=eFqLWnmlyi0OLA&;cKF2 zy9RZC6R_<-Nt-5W(kT5lI^If_lX z**Tq>T9JWxVlZhc=GpoNOp%J@)HnO#mmj%Jl%#X7=6kd%jNv4QD2q*fZ=reKL7w>;-WFX~B zM6i%a&F6mVqT$UUF5kPqU_Qw zOrig>SaWM<6l>mUG0Qu8!M!HI2>{DpuXr26KsKN3WP6Ao-zu+o+EXFZUvw9ytyV+* z$H5yZ?hUXVJFfPKURs;ft2>Ecf2h}bo@gpcTo&_gzQG8Xypd=uifscUKYZL5PWpED zZLMDql!K;hhDxoMeDq_Y+^hJ$nU9O);s&r_@xe&IKb9J*U(cL2ftx){k_6-{Cq31d zWh1+6{?Hd6!WgYuJX$qGh2~B_E*8ynPB|Eqv#)l|Esp*4H#?qcn=<98w$GNl?H`K^ z6r5+y-&k{n`suIp9f}N(Hv82hdfC;|v#q_erNg=63K#i~uF|JVH*?RsFUdS|lV^o` zYzPkyh9G&3kxUiv*Vjj`8^yto4AwI~RAgV*yl_*yXC!(sky-fe`U?61Qf3WBJ!c{} zEnG?5tYhcxaWH;UO=?7`Tge=HPeY;kefN6;pFRKXWz#aZ-&T{!WEPfywTq8NhvO;i zHKd_fDs%SMip<|;7H96hb;W1reej4o%`Fqektm>6Cqz6k@%c904ETxh9!p}-{^X(f zF`H=wxlMrask1onIwqx%n+BsX2%mljat`Dv zyx&uZy&MilV+1)-{oC3v{mfWykmRC49Gn4g74ewCdm3(Z$s9;Ag+tN7>~a~5(drq4 zvwo{lc?%NJqv3($G^q5b`fUvgEDb@QUdT}rbdo6&h;WRsoX`qHuob-7x@w5{3FrWl z;Ij2A9%yo`ALtJlDH#}qxYZ0Mfx9*3(Pbi%`S^`9(zKlBqv_vj%M?#?qpHP#Ul+56 z&9G!fi**dA=-f{-U|rcPXXd*%t)b1zoRNQ&E4GQs%x$-qR9l&N5+eV)>gY8B*#1T&kZp@#I-LwdQhKSBx}j3h#b#;~2) zU(})1E%KZ)kKVCiol;UzyR_gSbP2$~!3a)xDiNUdqVuW0V5>Ugu?Vr#^76H!s+Fo8 zizf~zN5g=w3H7W`j*7nSaOzOQz%g(jygiu@-B%@t*NGDORJ)kTGW}Rmk21g=6AEoM zL#VU+YqpvXHl*ShIT&2~xYN)g*F{s<+(8VXVz%ltYpwqu_E}bB zAVfaaE~;i*hhJ@cICFY?MP~Co-n8{0I#{Q+<&D=M3)bJ%g80BIX+~?We~9#c{{0Y$ ze0{}Nbx+Vwe0&l8|Ozq#V%gnlCW9Gnp-Wd#&0o`NfhWl!! zTKki^;eXY9_WlnIyRL19@=g&-f9}z-iD8bqaGRw^>|n*=q@r*tbIaMeneX<^s%NES zvoE(M1AK#`s=)>|jzDqd?2hV6-T}rR*e#~*tjHlXLuUEC^W@1diDG$Rk;_Bvx9axQ zc=kASJ4%fc*c!(2*7+LMfinLF{Ow(#ww|7@9)Dd>HtrOcEM#F9auJV3D?|0^AYojd z`I)=R!Dr_-~<(oc4xbIYkClZ=nSq5yV} z2rRJ*ut5;hmU-ag$20fbTd+9`J^6JbgFb)Od+d?=V)tm|{>W9Juq!n;9BWDN1m2Qw&dEgq?zD(mQ1`~|$mKKN;3=7cYj>S^R5V+CMNFXm z%<9p8F-v*|#A@bj{jmMOq>zGOZgokja@EHbZJn*#Av^92UD46ix)W2jyRpBuUES8Y zo9$Isw4hMumeB6bmOU-)9W7UMw3%@@itlOPf_B>~Lt9$5g*v-nB-+*8(YC7%rlKt< zo)MpK2=}rerf7;K+qaN*5Ox7*NR=o>oL|J3jD<(RVy7-6EvS;9lVIecWG=*KaJ8zm z0RtsoZ(Jv;3Y8jbzkJ|s&tg(fR+ZS=L_^QT=a&z$KsRAD1P}G`7&HinBCwbp(q{z+ zBn*x0*g=w+lYCqhi)y};V$GKTFhoZnxg3fN99BvP`In3-t);Ox65l%W1}pFd6HD!TkCh5{hoyR)shiH*Mi5!m(uDDs%RSzM%Q8)UnENqZikAl6h-D z4zbwQQkTr@hs`Ws`DWjVVp;NWSNUub9<|DqJCJ9;;VCau9G!m851_-RuW-$k>%Zx# zk$>1EX3StUtr=9N<+nBSskgm2m`a8PL)h%l9Lr~pc*>@-R+`GmbF*Fb1Kmm_Y*#`j zRw~c>D#$V`$C4~S^lwp^j&1A_sPADjhB*<(0f+8bJ*Jt6j~>Sv937*h0LwBs^F-=6FOEK! ziYxVn2L_-?k*{3pt*I~y?kv$RARqGsLH}{1YnlvwUwGxvd&OM&$c?bX-MmIjJ;2() zyolq9vFI?2s5lRris|#197}+PMXa6DMs93W0#py)0dc6hNz76xV^foBncTA$l;Y2; zU6;y#Un}NRgIW%c0n;$g-0b)*AE?J*#X7Nhks?*F{OOs(WMmO9L!{*0jjpO4O3F)w zkAW>=t_@HcA~I~A0|tqOl0*tsd9Y1n15&LM^Bf&)%6#M8{0dVz;Ak;eEbH4vZH2|3 zn9=bIxhW{hU4fu{-?zo$w5hKqr`xwUY+~1(eJaVyuQ5Ksj{0iG-fSAkW!!bKQOM8D zR+pGxZ6eYzfqe*3F$JQS%JFq#$3iBoQ035KIh7p1u%zIT_dh2VZNQNPNd}(}R}pAM z$RHLTwx79fRuQo5M6XqKS-T!!*1BF?C4YBZ%u3rk#LK4k;ZQt&81$W6L3=qp`YzX^ zI?n%4^xzn9F*6g?bIz*(!KdEb$j}gkG8#I2v_J)k7ismz7>TF}yByJPc95|(o3v;4 zu_h;i6j%!TO-}#KInQ=My;bQ6d-aA}MeqRE(I-W0HS1)tcF;D(h}1fX3HT-uo*`I-=>KHLA&d{g z^5sWFwY+zuxJ3T=5!l&(uu&|Q#m!=w^qv=G3qV!)XdY(+39%Z3Hpmpqh3Cb5(ImV7 zRm^jN=zR0MsKw($KNY?PdyV`g{9c9CxZqV@-RrjA43s6=3RLrNAPM;UmczpR= zqKY2Ro)=5-_{zVD#nT)O%Lgz<`I&zc)20z!)D7Ej3su=u%t*{P^p~bht4M##mN96d zi+Gb*DL>vM*2teUiJ57?f>v8>!LB&r*=m8+C;oUpC>c21Jmt-&oEpGTA@j+BBK)co z`Hsu5BcHxZERo;8Oe`$5j;Ym(U#2&RimPbeSqIm#JQnWI$P;BP-d`gjb*OW(D3)U~ zuxR#M#|h2Z3rUMfz8TJwi~>1t*t(p?d;4^%Q*2F(+p)gB`H~@X_ECjZ4^|u27enbU zr#|rKH^+~9$vTPl8gezrHzzP@bKdZndjwC}O@Se~E(Kj~ zd?Zf%gBh0GzeChr>Q{(_8q8h_re1ihX$2lx6SUE2g7ka6!**{xaEmnA4bUsr_pHya z!m&%{{|?X~70+Bt#iIw{)N5?M-&J{;A^nQ2r+FWZunQo12x(^D@it(})A+h~i;Bh6 z1F>%-QAMEwjI=F_lOp)?*?QO97523K{ZLxLpqn_%h*KG(XqOm`qD7^TlegR^YSJNF z2s83r(79oTj-s0Rsd_rad!)_By$;dLC_RnIJzL?-wU$|mUirI?uGuoNMwE*>`S6Wm zp#+4@n3*bmLpY^N54*U|a{0Rv<>A;0N&RZS|8rPaZH zLp{Y-IL`P;0YNMX^bzDfL0TIl8d)fiZSK%q#9_c;wF^1hK<5Jo+(GR>W@{{>LvrLT zz&GFA=$$HW#II?N(aS4Wx#r2x{qQCI=eM{POZOVr650K-s9rNEASv)=sSB{)=2S=q zOH>{)c`Y95)zJ@nrU7!cw5?9+sWsg6?9v>c6V;^XD;-Y zvZg-sda|f)g5C*+bfsH?)}~I=9WeGZ1^LV_PeA_aGH=y%Q^-%I)|O+UW)`_nnzyJb z-~Udru-CkBvh?}*SsFAPW1C2VfR$q}ek-SDGaGXHz)QfByt2+)CeMD~Ti{jdP(E@C zid-{fPmiZm?)gJjL>cDp!%*)NdMpyvj9Ic3F@2{@goF z4Prqx%3blim^T}Hr^c<8*H9Vdnwh4_GglH(;8QQd+1mNfeMuNj(}K zRKGyKNk>;dl!&ki#Np+rbU{Cy@2yw}vkvy?6!qZ#Op&%>@iHA#|KtVl1Em36Keg?ia#$>s1Eyzv7tjcmTRSblqxs}^XDm7{z2 zn`U!lEiG5&YoDhDaa^hKbwd<|k*(O`s+RAj)}k2MyHvHGWS+Q*JuT{F=phXA{7O-i zDZjHi^UNKy*O^)*Q<|zt#eB8cTwqmE0CdIdVU0*!{c?3b+Z<^kVSJ}vwVHrU6T$U zje`?_^BogKW>XM)6pBR-f*?a1ggzBl*#MuGfIqi59dLsK$s?CEDC8Xl8RKHjSS5GD zkxDmVl(7>lGCXF1Z|**G30eCUQQ-mx{^UF{Kl9{M1&tJy#h-ka!+}LFsDV5s$IoqBjq@<>RO&h5t03Yb@s!5|Fi$@uWm2*`n~y%7;r%J+&FomKKl-T)If zs5Dg?4Zut>NUT1rgY3Ag9!UKNH?cra0VJ8L3{2WwwxteI+Ep7NK&ajPFTy9Uo#QG< zFH@(_9xkhTnQ2yElp2AR*&3fHIoJ{Sv#WfjUM)ZLuywvb=E$OQ`OuFbTin|YM&_TM zgGSs!V5fdA*2>|ur)Y&L><3it>SC*KPg`qSdpE@bDjv1YK8URLwa?;W-BX^*yV`qu z+dH?xeMl+&UEuFG{8+5qXdNU_Z>CX`x6#`NQ+)_6?P}?3-A*|MkfZ={GNO?T^;SM*M!LQ-A7TLj#1DiY{Td$bKP=)igDe7A}KNyz0RTF*9!E^ zGu|>b>_beleYu)})>*%9Q*96^1&G#WkO@f*LYa^#v`9Y_kTcIrQ_b6!bKX~HZG$DO zboaDvZNEAf3am|AMIdI8UK_Ic;3!Hb&(wqBKk+1Z=TkkRh+S1-AtLfeUKtv3=oW3t zlbQXEVm}A+hWx0GLsaD= z4VVGuZA8^w4tbRWCYjUH4W0yUclMw_hBctF@f)t1Ost`@ho`_pwb&nO(qclaCUhC| zjJ3!r2W-TU2pWCuy9FGa@W)-j$9i-We0*TYPh`RbM{C5;bb?wrrK}cJK9?NtmduVf_;GbBx}kz_&IIcM?T-W>ONY#I{V6i@+^F3 zx{@FJwtGR9V-o5dFH$p-vu;G>-6Z4Q(%#a6K~s}N5=@Hp@vn+RvI*PlYo z4R75_9I~@9NA?BMauLDF-<$@72EOzWjK&dArYq|Q|uB#HrCOFcC7!Ue5sOB<{iz`OJYPT=qjrXpZBX12IM7c_GL za0^pwc=uxYJx?kj<+UUBf6JNVJJHWEF^zH=2@SYThy1LttpksG;cUP>ZuQ#-%9c*r2 zSI;$}ty@qCAQ1<#BC?I2It>9!7onlS;?S<0D6eYguRL;?;#Kn2X6LkY$0V6V6jp|i zK;D0lkj92n6llo0^yB1$7Iy51$s=)e6(zZdI&yn}dmYILcRxxT`GCWwwwqEGrg_;7Q~Z+8N)#wK3c=aVb?ARQ5ds+?>6{uX9g*t%f>rRiLU5Hhz+0vhRF5-M9wgB`ro;_L3} zZNJ(V9y}TaYj&K>nuo1eT5`FI1JcMb2Ag{fd2z554Zi)|8s8+1a~qGC5H5SC!o?dt ze9$BLhQiTd-@q_zDuaMGgwus#jDb*$$Up?i9`p?);>lz^Bz^ME^N}_&bzI*6Rc~eH zJ-?W?SPT7tc(^S>u|yHI5NcHEQuS#Zx zXkxHA4GfP_G9P*NL3g?Q_#N)*N}{6F1F2{2nh7qEdCOOe7m?&d+FORvI$wwiD~JzK zU5oVP5AqaH1(tH;=HGj(ZW|FL%P@Rm71a31Q*qL;2YKOr?$Ws?U<7@U zGE;X>%H9?7%6;y#`J7WlNlhd)ahJ^R5OU01`PEW4hAUs%=vg2;H;W?K{e;UaJ9m3Z zra7=v<`-_{P56L&y88T!ce$rd#eq}7DpGs^CIPPZ!Hjzg|EhvyX!i`#X`bcjN_faV zcEmNUBnCQ!^L;&}eYHihO}D#%E16cOI{wzcgH!L2+fSKR6@MUoS! zMbWT)HYzIDV7>cUyKWjX{}1t52;uN%HOaBtvgG#|_t7KBDRb>t3*_dnd%c&C=3EUf zXAbsQ<^4+lLchG*T~rtvixIdgAgwzrmMyk0f&x57Aay=?KR6}v%$bvA$}DZO(<~@jo7FJtO^D*)zS-OX4c?X zLg~f3>l|lFww&=)?3iqjlSg?e6;G(`YFL^mID!>=b-`ff%-Nc`S?4$#%OafSvP{=k zpof`+u1(fu$R;BJ;Eo{rvM`-GbifCbpai=l2|Ij3cWB)Qt!>tVQ2Zd_2T}@Z#GQSj zNi~vHBh&Jg^2$l3!`=+Hp7K=AFo=P1m#n?YQ)qB7|~b-*rJ0NBT^etYr{krQ_)dH=DxRb zW`RNs@@xbS=L-&rf;kW#Vd~)YJL+39)AEDTO!rp`;*%}cWDxAvdeIOV$gUn7nU>CN z2%B-hF`$z`?|<&3u7&#_E0Pq+%|5CSjYi>p#YF;#BN5WcC`+U{#QlHi5?t&MWAa2_ zxx1imBLM)oq|D~8RJkz5V;M z_e&pZ2GYM0y4azOQ&|`TaA=-MKShY#%=IG3BJ=E57eJyV1ZJZN=B}7{k(-L<-I7el z7!4?M#AI8m6H@IZO_U1%L4KC~b)3CTTP`%F=uwjN8q!Y9a_CR3+650d3c5Lvsco4lj z{{2CD{v~L8l=py+FSUYM11dh%Xr>$ist%pUWP4_WDFpH4ro@4g2cUp_X%qr_bfqt` zRFMqAdKAPHiySli0!yw#e3Z|B)ZLjT_Xft!N`tNt2uISg#Vc2)V#ro&XV+Akb||Kd zyD{-`Cj6G0Q8lp8`~=Z~y_ryIMpiM|d7u00O?Jb!keAzF?zohD1=E$x1$UVUWx{96 zs-RiISs76$M<0KZIOyYc{V`zF6agm+wO1gBSYl#itt3Jkedu+E7t#6K-9ZnXXL-xF z+*PZ|kI9G}1`4qmu=^o34@tpB0^z`xM5=j3ut~1iDN0x75zN$}Enp=(BJDPgJ%7S(Xuez#v)oz$TMlWW8iBXXE$_Su$@ zMv4pdNstd8ae-Z>C7B?a$nPIbxHkc%Y+|s$S!hpqu zSS5|w!g`?=c@VZPS_QzMmUjq-SVqR-a)`MTz#ewkQi?*FNogn5zGds;ClP&EU@9^K zlRUG;b&0(Hi117*Uw>qYYu#=8Jf5_o3W!T2KYoM2F3wfsROYGW;I8w1wZvm$rtRgg z!xCqg-0at;D z9!H*eXs3ZAV^B{a&?-jGI*btY&V#5jCP-f#ZFhrHwNP{`g;5f7YSa?Ru_0K1nYRdM zEwv^&M(+dl6_7h=a#W|bhip>*@}ur;|Nq0)mrS!)&L@xD=U#)TWrqeRmztREP^1)K zX;wQvG=${HbgX!jj@!y0RRu8WB+u*=yxpNB9A;URBtPE0%ZQ*KqLl;*w1;T>t7P@d zUJd)QUeQogwjR5-J?f} zvJrV0c6VCdz1c%;+HWJmynU!+vz`tO{_^*jb+n)v;^W6|P<8T$_j@V|RC+skp+yAe z@ESAC;9_``n_oq^8|k+y^@#1y>IDnTx`g3>9c8oxJ$J1?zfRgNn)kayzxIa~QEbNm z>6zH7ha~4VAx+p!x%nk;v3%x(Zcm9;P1%S;etL<^w=?fr86akDa4uhBypSvvkYUf| znHxMl*^MmNnVVjkBfGB_b;ULS%gerxc=)qdL%Dd@GSBq+AzFIvd8&yjIdLg%t-NxN zsGe%6_~a`c2ya+eFT9!8H&0t+H)?6GO(tN-(aP;szI@D66)_h0+oxK z6%2B1wYc0J1eY>9g|rl#YecofMbIm68G_q%bxO?OPx2GhqI4}cM6^Nu086BPfDa0A zs3+xKL+*3rO{Tz4}HD2{D#3|-!T0&p{QuV+3nOuQT%P+BO4`U)YGoZ zx<@lA-ba=V?L4W*C_@@r!~`SHULfHQY}(koH(Y;xWBrEx!N8{Gy)E@`*&igI^iUfA zVuA|%Nea>aE08^i<%JNyNgAms(mJ5Az1QlNE=5(VJ7nfhMuy_Ul>e1hkSb=mFVYZg ze<|*XHoBHp+_xk*43WJ2bd;(Lf6iSYV^4Tz$xljemHd?S7EU+GQmC@LfV?Vf@S&$X z8CB>t{rQni7kLmI(XWoT&KCJKOfGgF7kWc~Pat15@G%s_@@fTVJ5K}e){*@w)GGpPE~p;W zY1ZH0G^Q*dOu4^F&b-O><}~XX*U{Vz>rM1nrBQ_Lltu-ux=nm^1^a~>sCu?|3~Wer z8xWjHd7}@eGZ0W|mQ7b`qS=T>FPl8u>n)pM;T-T8$=jMe^X2{5c?)JUEi{dQD#<^* z%RN7xqmZ%(Yxrab{W7nVPRdbK&D8l?u_8FEbX$-5%sJe%#sJ>e?yebGv-Nvw2jrtf17*^P@-eJa@#sKNg2CD=Md23en6xFI8k-UO#<~85xE2AW&gfx1aQUv%$y0Gbdz~mv7zYsj;ie zv->e+RrE`1$MrKT6AUka zOH&P-Xsy1ID*gL;#Rg1=-7VL2bhRM<)Kn!_%rb1|R8|KFa0T8dZ#h+77vl}yZaUIL z)9NRG)3m?#MA;2O&xKn{>*40zXPw@Sd&h#2{Do|`Bl_3Rd6l*z=R&3-INI5R$7v$>WJ`~$LWbTBS@pv`5M?Y3 zLS7I=;J9><{7~1f&iY8+H89t+r{csr$ao#hXVnJR&cgMK>a%P;9tyCKBcF2DycO;% zJ?aiHC^~)Negu4tLE;^TsfSYOlO|Bb@WyaKRw4|cA&J)jm*nvgu#ksg^jA&WT2CxJ z!4g%acrVkYykVbM;MARn&$=rv(H+|0>ThU(e{M`wCmA!z@a?Sl@TVYl7^&JGlx@e- ztr^z_9WBBGQWt(Hpe2dGs+8U5-WA0A+KLL(6rwj&4L6pMSmP$-e}sP@$<(?GysX&2 z+1f_UAa6yaOJ_W_m9k~GQM>?*m1S~%ziaw*sz5H%y&au9WyAgM6}hvWcL*yFqaYT& zpR!}r)6(mO_sf#|-8F8wyK>d74n>kyukt3UU9zr%2_vEbXMoh6SS_P{z?NwvpZw)X z*U}&_nR7mq%|*WWoS1usmhniGGfe_0bIp|0sL{*Adn?QNQ*K@^KxwR6Szt?0%O2$N z+0VG=x4!AY+1m)v%<9f;!Bn~IEuLbEY02G+xia)=_ne&lSS!EupnK_rF97Ie@*|&i z2WN+JoSNl@PrGZKjr`%$?#(b%!`qT(AwPMwXKr3I0e#I`qGpFJH;-v+I1(OmsJwZ- z>c~a9E7Y@vysY7?jgGgRyKKor{-)`xKANUqNP4+m5escf$ zr^}5^m?(8dp!2qzSza|^zcMW~Wzx6TGg}^d4c7}i{3NnRV9VB#TL-e7nALz5y1=zS z*8Kx6w4D8byI^LvU6GeUK6%fEyKzUG3v1g z7?e*w=nm#gw_%Pxr*=8VMA9&!WjO&moI6<%Q0hqgN+5K5QNPmOwmyY!2M&$J4kLvi z>4NvNnu)wuk??@F|JWg*M@o*mHDVuEuAyLp2IUw@9o;Lr%(tR(D zN4p+$*T`FjT~lXJZA+q(jq5#&cR&p0wX36}>niIrLNeAiHrkJ98?83mUD%+c z3K3+^oy1Dcs|efpjs?s*fpO+BHwT~s5P>T)vmK(!rm|o`tWgnfh|h*1qUNMn zK}qJt#G^xtKcQ25`Rzl+JCrs{?pON&EOwPwTM~e7$N~`tjLeA}rp=kax;M!O z?sZqq%!WbP@_WSW%wFwT&oicdH-8pTIy%b(lnNBTenu2~-t72f=7?B%shW=sRMr6A zh|*t#RFDCP)T5{@N<-4Iuv;GU0I_$U^Z zl499{SazV6gYRjYK}uw%cCBa@DwwuUT4F$NNe2n~oBuN7hyzV0C4luSVPF_>^Cx<|vR) ze}zm&p0Z2L>g|mrThU7KslQ7an@z*-h>#Q#QHfg2m#re8)^zYuRoj zuI^?ddhb|2Vj$a)rdNJ%rmJ=vKo>H+i6N$P?feD~Wl6Xp(^UqDZNz085uKC^-Yh1P zE<~a6$jSoa{P@$0#ctuZ>uZYvzj5{~mo62{Dko`Mwl5V+%Ja3Kk_VTHZ8QG-#50M9 z>!wVh{{pghu~@jxiso0h`ShqznfiXx>41FT9M=<#hO<@_A*3xPHyLXcl+v;ltjKi~ ziM?5V;maZ@-+PT~C2sNY$`Ablel52J#Nvg;>bk$l7E47ZUr9aa{w)mM2k%Apv#QtG zy?q_4ocF#$li;}T-Z(Wvg$Ms~=;h(v3euA>>Kj{G4ajbZ4Q=k#8-jbG~pW?r2%G zhLgs?7qNuotf$1HDhodCtnvI=&WgI4ji6np>&r5F<^94w2Ojyv@!Odfl&x z%4zs&;6U#ElGiUczJg68vHgV7q}u}^a6O3N=;KQDgCeX3qyqG(I?0+*Ghfz1=bn~@ zq025wiVViWJU9Xtkal%rND1yqQDHb{`DLxUpj57X#5-q}ZY_kHUooITdG8P13s?{xj) zELN%^1lt&c@+66?s2@7t)@Ob0^=lq{red^AZYlIsz9nl3?06<8jsh2NGm|t5wpN%m zg=KuBr(Awa zh>C@@8syWHU_%&?7vdCwDUAu%j!G2C%(I)}FblsHdmQUjxRrCZoOi|qktD&0(2OZk zl)+Mid}hVhtL5km-r_WQor9aJCmSRM9K)%ST%q}T2tr_VJDdX+EywdHUvZ>bryjCE z$O2|eJI_%B7fLvj5`wKYqeEbO{BT4ERzuQmQqY+HAT|1GPF>__G zlym}HBDjA5{!=3(5nNP9*&?(y4F5m5~Dl)_?v*Lx)io7(dPG0z}D9y}z+M9X$8UHG3PZ9Gb-)@`d;gK=( zX^}RYa?bj#sM3Tgpz@P)&S%p%5zac_tuMIdwyKtGL5FTCK=iHiN)j4?-&qx2+0IA# z@;F>JK6*bg6x>qeD$ksIv6AliDuz^IEfX4aK$kjhj?a$62IkT6!FVjt;peS)euH&Z z4Of=@*jqYJVS&AVM-kphmUN!}gDBr&K$d1if9NY`2OOvre4hoAgr6zV$?<^dFvvY_ zG(GCCl)iPp4^Db3EU=+oc;fnO_=$T)4x4NI^4#n2G5%$bccveELl770hT9&XJ>Xp2 zIzD{z%%H1m*yae$PHP`0yH$H5FFH5SKO`yQ-a@9gtZvl=W#>cQ!iAB^L5NKvG`95Q zVgw6)a4pd5Pl9*u9)PYYutWw?V!~KIgpf6Gn)W9qt#qRS^Msk-ohZpX zbDd||MFyMG4XRy-wx4o>(=(xc1rKo*R#9V zWNxLUKLq4W$!fCB&sMnjgD@&Dl*f)YhZ$uG$1p_d+_cfs|dJ zN)b1rKPz;=Oq6mJ31(V;yl_F*SBo}tcp+RJWLUVW7UyqijBl#LX=Im(l`?#zt7e*G z&@%i-=xpCKa0hH4$vt>k`yITsrJ$lB62q-^5kv?S?7k_%`nY$GjAheQJ_APgd;q`ZLF0n3vh~V1fM3Z^cYaF@y3m z?tg`nS80X6Yjj(H%IlFgJnF5GyLKZ$siDkOJJpiQ<<4Jv7pjC(bWcX1LYGZ)mKRz8&NGGU zGG@)r9bu;A`3m(JcWBJiBdo+#8T&Tk>iBz#T!moT=E}2e!gJe8-b(qCxlm!4Sh*4z zdJpbd-hA&g`=)sw^v+zF6f2+2Wm|A;xF3VAkTw-}GV{eK3gqPM1yGm}tj)^7Aes69 zwpc#vaxGe&i>6q4#L`BCIgYtS_RH7Hz4M8i$S&t(*x&#yOR8OEw|QJk<@bwR)64Q< zaFX1sf)0Ettb%L#GfCioNKXncfkM@!NFg==*YC(sJgCo<= z)<_9$^0Xb_ls*9IVZ;h1g}6GDg~fb#<=7xmDPMn16m#fS)^W^Qea4^RB(WHQD;;`G z*7zWeoHZ_5DGpuL0=*+6Ob*alGiI{Y8dJoR-;x+k{;+vu18SsBizH5WcxVRp%Wpm^ z7IrH*%_IJh>!udHEM<;EWY_8=n;3$zAlX<7m#bnCZZgi*h$61TTp*~_q3jtapS;;! zI4k!Hb0+tDT}9!&maO;&@`!^h6AGFrOQ@xoFf3YynoivEl9z1IA~>t407E6F;0$Y$ zXpkE#Zs4mcA1fD2ihzsg2IGL74k3GIRx_-F;Fv3Ttr7dEUl_=DzEj-2Q_z;R#b6{J z9ubAJ$v|j929AR~*CVPn(tasz6q%3JFAR2ap~)s~U{xADm&#c`^%l%A%19-FlAb}9 z{KUOj-n$CQq60q?(`ICiSJn=?sxvnXcxJ?)d$RNba>cj3wQ|=XQM8k%<A3JERI zaMdiWZ&vM;lixzRM`%rXBl#wmX&|UNp?)Z{@OZf_IEbL7v)}c23UMaF$YTnvHxvP# zJ8^Hy{_SFp{K3z}{93pfCH92&X;9dEluLs};hPNbtPhw;~_g<1CpCqCl-O{XTa^ zr5TPBJk7^C^D!6B2r>JuODeYXv~TT$CiQ$-fvO<8-h=Cw^sP{fnN33i$$aZ0)8E2I zcD~b7Fx#eUN^T>~@*D(0fL7vCd5cq!)eQ)L8@k3*EuZ?XXeJ%C%_mjUgD~n zOLGoTJbT07*vP2LOr}{g`TqI1Kj-EKQCJj2bSj0O$uqAbwePIMqD%f~1Fo83*jN6{ zpC56pyp(@7;O^A;V1WGxgT>uD-JUXJx2C_^?(cYi{poA{Q>I+&C+o3aes8Jx3fV!Y z!m%)XFct?JHjp&d;;f5Li4lmw(5Eg1Q>4M%W0@i=>WU%`Z$_Rq`7Z&{BA>bejCylW zynFn|L2<_n*|bv>Uq*P`s>La6nICH=enXMuBr@_xnlUlBLcv*uXdAx@W*)!2QeLxM z%$9pw!GV8yt0?i1e4e?mc3$S`yNWUg?(@#zd&uC2#<_YkH{3T*eseo?tnX^ZSa<+g zbNU|}!1%A4>3VCdA?8U+Ak9tkW!K1At8%%}(%sz%%DJ>%EM1Uwh_ddpCz!GBw7<5| zRWCb#BBrLPZC*&-fz8Ab6YA($bD>K6<_P_xcC*>w?B*0dnoTHldDu0VX5Vu^>D}ov ze1ofO{HAvCaj^<9(VUKk;kMZ#*6LCKhfU2r>=q*h?7@(d^(FQVPphZprcGJ;FD2H*xDMegk9V7MYj; zKeq(AU8DlbmI23)#A6gWOL+~N<%-=%PpTep0a%)HCB_5BxTB zY{`@<^xsl>ZkMQ)RR_eYd@i?4SOs%Bx8-CJuk&?Gl4iD8K6!(v8uy0C3%>|waQeh}t#-Ux%e2xRDn1bYbeM_1|mIu>*)6l53C4WW(P0V;8FLP_pY z1jqLt6wMwhfIC za7*(b*+ewiYASyUIx8^5^(aFm+1gB#mPF;atAXedILz**l~rfU_Mw-befc1<*_etgwd; zTBsdx&Q!^HHEkZ{07X$0Q~nLrkrcyXeTS|}45ews-Zuh8G^9T$MpvB5L zUbJFNlvS9M%O}$8L6Bi&2JeHJ@whxJUN6)7^DK!CJ=S?K z2Msbs9Awe}9LP2=1S^kj&3th-?^KwKB4LL!=QLRpa`hfeqnRgZ%^R!t2&-=wH{kYSr4+8yfr9tzHve zyJAD5zC=u(ZE@9%|NgW%G;PM0UTb*jbnTQW^j|>nN zlf2~`*V6IPn?=>sasow5@oL!=(S>2gOqPEh!#ZP9^P|L`_lB_mcO~ zgf_>b zwYHasmVrZCTKZbZ8=H-(h%r6wh1bmRN0IUD;~x>#i!XDCj@skIcPy;36lfz77|E9B zaXA=1t9#tDu6<>XE5-X_#BlUhj$8T`5c#%qP)1aJ6n7W#Kj%Io{HtuHZ^AM(Pk5sW zkALMOqR~~U8maO*)Y4 zA~U6~Y6^{a8b!O33r)9nU?i%19kOr@z&nNpslBvYEJsq%?&tt80W@VmiaLUWj<8ZB z=$!98)(ap7wdkB#lrZ5Sw1@qi&lV_y^a6uK7<|asj#x5s%E1^=3Cbq;O7cR$;~OF0 z{Sm|udG^vE9qnJ z!>t<4OK`*Z6KhAs}ch{8}}k{}`PTH@M7`CR3P zz9+7lHvtwk%)z>h?85-LJvoez&BHN+00&)SRhXwyFhLswPSe+qjbblh;`JC%C}?m{ z@3wsZ7vaYG;bBo)Gx^fV2Tr)=%dQ{6LTFtvN*7VfSDzClhc)oCA}sxEgdv%DpJp`D=0_Qm!?!x!po+lD(;Stm&&qw?}b3^l#4Kd`B3FupH&DURaF z!FX&b|C>4)1HY9B;yu1YyN6&0{0?r?*YhG(ATsKcSI$X*Vf-Wy@PtuBX<*N{pj<2+Dmz|o~r7E!0+6b&~KF&{B(e=$}W7f-iiH@mv&f3%k;7_;yN zsAF_KBb!ByeE>^`Us2<#GgtWH;7*&>1c`x1qcM+)pIfqwH0Eg6At7~J?cIb>OMI{ z300_p!6^QTjm3|~fm79pXkDO0vknZc;hUKOS3qkR2~MjSlufAsTPKyJTTKdIRB1f} z*ryTOqbQP5;^|QXj+n%0l1Br#O7H7uVuM%1OE+P917p{tr-((_CQhDP8)PB}5D*2HJ$PMA4tPI8SGSAW1EiY?7j= zpmd38Z8_L^UDn!x>}D&KAs)_9Q)T}t_^z{<#0Q81NI`r7@{}ve`lXZ3kRQ)Pk!z1m zQN>I`68AO^l)dD+AP;&BX52_4gOchwt%Jq-pXfr33eL zP8G;Bbe9x;An`^MGMHAc!yDyFsdC#y`JGB_W6j7x#nd&4sAPS_bq8q1ieTGBYGGZ9 zC_ccz3>}N4Kj5WG2$1!PrOpiQBV%VW8Gl|B4%-zhj+7ZvTMtHu8eOV^2lHi4LC-COU}Ox=H8Fo@MUt-*S!U*|J&~M%vz)Rr*+HMf(5ud2wplw zl_1b})!Z~supvA+7*K7fHK&5~r-w_6&y83NG(}I2}fAyT0Cq0|RY_Nw8KS%>-#JM*NjSUZvgj3+|tVVwA z*XnAGZF5?tV`06D`A+x`MTSSw63I1C9zuvQSHw>eyB(0TUlbKHz(x*_0Y|dgp72-s zP8`J+1P>q~wN|0191K|3qflc}yg%K&OX2_r`3=L;I6;r#Q~AAQ_${t^nfa#+<-O&u z*=`J4KJ*DOx6zzjB_h(M`TJpEa@el@NlXZ{;T=51r&*SMLNv&$_lP+Q^trSH4zm;j z()UT2T;VlYxdf1P6x*3R1pS3(zOA~X@w<+dq>Pifk4*q~N+wD5=pw&W8vPud1oPyo zFu0Mlnml|?6cSiThlpCtX#}P+s5bm;N8`^ca2W8eew81ti%5o{ut)ZgFd0r!uUp4q zv(Kvr6^0u#2sF&79IL>3SnkN##C{Fc%CA*_F}0l8Cy^2qKxdLm<`z_R$U3zN`TOVK zDYM{RqH3A-5;LJ=K%{bF!%RRy@ib|&oL=cY2G5m1#^b|B5njpKE3M!r@S#7GrdSsy z?XntV(oeq)&uEAoY07d(#uSnSq=H72|LHJK+~qlYE^NO_9#O?K+&{2$*BmWu_8iq`|1y zoq$DRVU`hAe}t9&O5yt7X9$9oEvju1>t2UVDJt|vY>nLV(eyk(XL3BtbN z*);z0*TjQ@f@Ei2P7OiV#ubbOnOD_j&uJ9??8Z!u%rm$id;D?FdRdGdm*YSGy7-1m z?)Zs-cR$(%AD=LOW5$H8Y~b7jkx97jWOtO^PctApyM^rUXaZ@J>=rBuVf@RFi487! zV2;Z(-8rqy`PXYM$!>#PH)zLHGwP%K^GDpZ^3U%TGsitoik;q_vnNsuu@|@tAFK|6OiJEqPoaTn-;T* zb4MdNEWtV58|zp*934tJkqgA;b?#8~Gq0u5;pJPO7?%DtFtf zw&`nQY)>9*<4MHI8aY}gHv&HTOqv~ov-!s*NX|l+q&iGcG%mGKJWMViOAd+BX-FIf zs1FiqkVo!BifFu`p(qvsGQ09~%tcoLN6$g+-1H{AicG7Vzfm==T^R}6(2E_;&=um1 zXz13u=UWkLz0Jy)7YYX08fRg_$=8JTiUz?qXQ0oEmtd;b(sQn0@f%35*fTOzQXK+LK-7^SHBsecs5VR~8kQjdy z$SeK>A&%>RDAFsl!6^64XmB@&Mp^XfiiMi&3`eFO?ZbHHhoaskuihwTwPJ;Umd&9t zr5IY($Y?$3Zz3)?l?a`Wf9Vm++-$j>!;X30}O6N{wW40o)@Hi(t-mP%Jq=9!0z##cQrrn$yPek%UpDlQe(ADFg! z%9LrV$B(=qel6y6ju3E0NIhs=L3{XJD>;a7n(11*$g&HY0xiayp;eh}k;zHEcd=MM zKsvc@)~nlz!%}aM4GU*JKWCOdb(lozu+VtLuSD9#*ONHp$0qs6dqp+hR%Q(uSL#g| ztZaD+QJ#;#+g(aQMWy3kd`ZNGm*r47|6LG#H@{0Plizqn6qj4=fJ$1wIg!#h5-IHx zHPBlo*t;Kkv1XrBCpmuouf+zzIgMhDQnp8^UDy_(z*Y7rHKsC`)SFybtU0xmSYQ=l z@*pLaDu%!{AOYk;VQw{4j)pN~!^=jpY(3mugX$QlxOxsmu$$1kEHkwBsc?n zSj~n^5d$4+FVr2Q?f(V8z$fRU`R+dnPv-24^Ej}&N(qa)H3uBa3$0>a=Hc?fV!M6W zask=WD7Z;J{X6X2wO=ieo4<}Y@uF8nkk|9cAK+;&B53%j?^XRH@v69l_bfB|&vWJd zPhjgielM!L?1Ux%;Z>lPzx}-^r%IW02lCX>_RjiOXu@BA71}zp+hqXZ^MCNpmuLUr zEq0FUny+~XT2yYU3EU0E_7AB)s_qh0ZgW= zC9J{jhtSP3+{42IE{fs7@F+2>Dju4xN*$|g1qihW=SJgjxcRs6f&2O=#H{(uc7ek^ zfxCXfgCsoD|5+*KKSdtyw$f`gY#@sIq#!teG9x58XzB+};5sr~S_*H#6S(desZ7+P zt>t22mu^p_4?t$QO(|KPYr(0Hh-QP`kNRk}E6*jxf)>-5tt@68r0GKQl-=l#uV$pC zS>tCe-3i!(qj0~*oRT>@J9aQ3A6I1sgw;>}@web;Zaw2Eu@|cI=LmayqFl@ye`}%Z zxXA2(s_J~^w% zwZP@~=VT_MH!l?wRE+_lyiS3Z!M*Xt5q@Z~GZ>GDx`q$%2V>4Y> zPYmSx;!IaGE5NWx?wgH$H zyUMB?nA&7>Uc73G|5)%2U^rw%^vK`Mag~g}v)uK7Oa5!6tLzd|(7q`yfm4rkS0SUw zf1}>CM?ijVsB(Qw{SrJXVoxeasLPI@44;wZm&G> zc2Pe5(gN4J+~c?UT}3l(c&4b&)*#G1+0w5=O=7FmE*2aqkNl0Ne3ye|Zw5jIE+Ctm zlMxR%^pF-gYi==2rY$7~isj#Ic3mPb-zL^_sE{n#03YVzMpw~*a~04}zO^wg9o-Sl z;IpL}RRq2ogt?|V-bG96QpO_|p_A?k~C<3e&`fqrh+2BVciIdHMaKqCUH`>y98DHP( z`Ug>-7jw!h|HfT9{@NB`!ct3cppnr1-Vwj7{bw=PGVNAb>>CwO>R;k19mRImxJqjB z6=QV;JOuhn<>J+@xi=MUca0ZqcduPP^`?;rW`1PDlqnw>-*$sLStb8?yL$s)rLy%sYxV-U`VRL-nYzR6laKVe=49?VTOq%8hkLGk zf7n$hU%UhNTXj_;bp7cao+7#SPWPHj(`$3&0%R$X@3|9-kh3@9`t7s-?w%n(d#8J| zeBm-&U#GPr1#o?4#Tvk&kOX0bM8j@ z;yK)&_VRL1iQM{OcdHt~$3E;{JbvNB?(b|Z{+F7{o1R68^|MRl13!0PFJn(*gxxQ= zI~Rmwal|Mf`azr5;8&quD2>DA|KzP4|I`ca@63=>fA1c?B>QtfBYc)O6zU4<*c9LY zsqIVPq^z!dyZWoIn(n5X-kWZEsV+dep*IxKARsMvpp{KPG*PK8x(n#g8@+(IOxZE$ z%c4kXhC7)oNXD2+9JeGBO~yz};>)5>@l70GOrAUaUsXe1 z@_X&Cr0VYX+;h)8_uR9b>*n45^|zfzzW#RjPfOxIUGlm1f62|sp%2V3a`X#xhumXQ zrY8+Dwvlxbv_xQu0uSf=5`91}@Oz$|A_?g!ZA|D(K@iATCK?Zgt6K{;cXuIzH*}mX z?T+K2W(&0I!57CCx3h^~A3SG_YX(I;{9~s&wM8DwQlbbQIRu~oqE8bJ~jeILW)PTx?H4UIFKfH?u5M+&Qg-|q@=Aj?{r@! zcE6cXgINyQ#b)*}3PTO4g+)cBnjj84dA*Y5EH{Z%W_X0uw|9X@7^i$ZUuE@NN=5mg zGTOj+6D1`?*=RQL4^;sMkfyVWsUZ1tS!2`2bv>)!r+qn7*-;8CnL2=WiPZMXVwd)8 z*}A#sk}FBK3Ti5FRzgMl;UR5o0ES?|ium~Mft_j?L8oxtRG7pj(mg80M*vBZEo(oz zR0NW5cFr4xH63U-r9D&+~Vd@kMJpbo^Wqtn2wwxULkbe;infZlmTq1Ih zX8T&BWD_pqHrQy~pQo@&T-O^~CjM9&o}D|EOdtQ7dU5Dw5W&=jP$YfRk4nYTL*cHJ z*?QTxY5}l+b$8FEttz~8#s=`*Wo{#fFo?x0<3MK=gtgJ;o^3D#itFGWTe~-}7w>5f zH@MgkyJI5$DCp^SxNi_u3&P|DpH&CPl?sx0ZbztU7N33yS;V|iVo-FAxbWZYow5XE za=~RM$va18imwnZsdfqAGE_asliAXI^>9L^ATY7ZAqgq|QX`Y_!S5@?DZWAw7Ys`^ z5xbz5steyt#y(rtcAb0v()il$tK(bNb#GX+l?+f2cQ=lZckmia>6iAb>DlVH>B@W{ zqf=XwuuW~LM$Ud<9a`*IXfPdBXU9>GJlR{5JD}?r=+TJPUij{J0 z0uznu_j2Q{m5P{WV6Ut)@?f#tZw+57 z{`c2Hv(lY!M+=Owzh69Dg?ruFRJanhYucoOUx7)w(b?V^U2eS%^wTe#7!(J1j_9op zl@=us<^&Lfp9{JwCVm9u|LuMpwW)t~8fu0ohhX5EWsvFkRh8oRw}#i9c}z7KW70zu zs~&SY>rCf3m)nMuy8w3FKxMDum-``w#4utFG&C})CHHOyM67*`q-El84Fu~aUT94J z_$S5UCtnMd07hlv*;9x)df{Wv`^(r(p^0WiOkBF}%^S&^Tvd8-d8s(@&2W>b+8U}z z@tKSTf^}*$KD-nGE$2Qg(O`T$iR&hol9YwXDra&B3=W$wUcTa*MB5E8xw@{U`QjDV zbhW+zx)zd>#qmG1Cxx33ybX<`UNe1Rq~$LVt6>W(<_to1Ao?xuE7i?H^tPixnqa@kwVz#0##tYbQdJ*GznN@!Dcemj&W?&=ShMWp#4c^<3~ z$44=Wy>)mBslK^XtUBz>a)5xM^dP9k=xL{+09+l-l@e>$hU=>7NW*0TEEF4|>Ma0; z?FH$(-!4qqqsJ*F*;kMa?ZN~KH3(u!f#sXVZVvj>W24_|kVjhHXP&0H87lezdO6TC zoD0_<*m__z*(|4tn``1|ICimyD6g!?M$l?2EpH4)5nu)1A=xA&kU056Kx@O6P)Vuk ziIn(Tqd=kS(o=6XGGxWko-h%Wa#438T%3{v$UYNZ7`0+Nl}xo8S-mx?`txmEnWx7~ z2BsOv-Xboj2^DPsYOfv{#$m%UIoMR&(W&G$c1Hq{v*14fI)N~ug!ek$0$^w$+a4f^ zoq`?iB;xQo5L<&vV*S_QxrtS&S z?V<}7Q*MU;U^y`ou+FjhijTj1?p(;C_y|^qFBeuFjmb0kCb7LeyhL34t}{bKz8@;C zB#CPVrzxVJbn4TM*O#?U;s!KIs!=!=%t#Hfkpm=UeKzQ9`gbQ*!I1-7{bLhTUkpc= z$(Vvec=d+62!8=rNNI@u$DiP9!=*2eN32MA&gm>BT_h&7Qw438yw3 zKcgAfWCK<0)d*D!#B<*ZZ7ySgX`N91L67zo;+HQ%zOd=bVfbNP4!`1X@I50;kJ#+x z`@=`wV)5B$@KpD{`*TagZ+`}@^B)`uwWmDIw$+ec8H`z zy~54c%TI^i>JV>#2jZfiJ{gV{`-UD9FF)?ghxICRj*LgXe4vBxH~jIizYn~A&xpvW ziYq$8OJ>OLa2$K=ac6eD-JZ-rL#jwi`r%#WB|%^iVv1ip?#x-pgUA3D>Sv^hCth`G zF7ZK)*f$9;`bL@@3KH})D~MuX4PJTNX`Tl+{jlO3Q<^=*q9FPz{}bE>A9&3vyizS( z(8n2TMZz>dx-GGh2R`%%q>M-2e3S~EK9U)?Xmd2C{169qqaEN&^zN>-^cu3NqyjkO z3T4w@Ch+v;%YbUDdZO+@HKs65q04t%7%5UO!n-r+lDam~bL_lZ4})Wh4@(;ZA&>V_8<3~ORc z?o2#`;az8FbXl}br_m`5>b^`f43a@Xt~3ag?T9_rE@TSl-FK*txjV zQ($VMAtb}G^TlJ2hiX@tA)0I+q<4JSWJ+7F#39QX#O+#Ku^C=-?EWyb0n(4_#|3zN zg7eS!7p$ax_O}rXoE|DaStfTlhxb{%m+g zn0||?g*eYfPdQ=n^s7#>_`$v5;yFG~ry(F^ zzlfZZ(lwYb`fF|^f$^x>?bJiMM%`n`ZIC8g(9{?I^@dZeXB+x%xY2;&b zU)id3qPBZ{Lf|hxsc5n11U6Z#waG;L^X|MIfda+_*&1WdM4OS= zKb9~Ux@y!Fq2|L*G!dtQ_uaDSrHYVlmI)%XE4QZ0&na>gE3;Gi0vho8+1u;eDZRaH zh+3^WX$^EDPJ9hc`t{D&ymw$A;8z~{65_(#FdQnaQf1ryG+r#O3zw%r*DYOmGY&W* zGbj}!tp_IeR!Ng)l~Z_+Qr<0#Lu#_kipsSf;}OMCtgH)1Dy??)2oU0>>FIEFo*KJI zd=D|rcfk(Jq(FDAQat)lsC0&&AICBL#6f4~Wpol)jRuYoYsfYQ^h^S*RFOLB&b~_b z;O!l&jo^_o&HRJK00&j&E9(!}5M=+zM`EMH?{d{fkp4kGQ)xje%NoB3IO}ZLv%xlnAG?PV zee8e_ZsD}xIBPBVJ=DNd`94Gk+K^4MI9w5_edPP??#gXNCu{!Yot%8UN8-~WrsB}z z-IQ|km8FmPV3C7-dQltpn~w1k@RA*N@6r6q@z&bs=e@H+FG+w_9}q zhHMQZSVjXWW0l0{9=EN?e}^f(-L0GDU!VyI*3^(YQxuK31y}p7=xR|JI$Aj5u@V`# zZo9bea*Yhp^=i1b)pw(o>&uO2yB>=>N8Ck4zVV2?w_z7H47;`S)Np*jkrEpVfK;+G zbC|wLtl^Lx3)Grvk6%oUy61`^0d1-u54o)co(w>A4ZEEcSx1kY`Qq<~-Sfmp_PI-o zR@MLh*&Cdk9QrI|YEt2?1Uv;($OIVniA10K!R$Gs|IMv$SYy~*<$av!45aOF8BRd% z+zY+iyFY-in}Z*A>%@Z}cAHk?goW#0gi2vXN^L52#Hba6YCLlSz9Jx#Xqc14p#yGb z16@^^pNO3Y4IqtqY;O%GH;Qj$(aSFt{e$%|j5T&$K7SWTx#5^%Y#~NfTwN zoxX~A?n4On_03^e2|cvWEqNsGpnFel(eD-%j^tyF=_A*`8%f}0G@*uom&~m!+?tWh))KTj(uX+i!#GBWON$#~=p6G|#vtU3chnNKFw+_O3 zTm1k(^=K(ewaHE@`4uE41wceqJ15<_M|Q!&?eAtljt!FBV(47nQauypE}parumx;O~Y`)+rCHm~Td74ILoA}=S0J_~r~$oplRlFjf*T+FADPJ%l$fU(N07W>tPW@*k``l_6lLFI%q)L9wdffr^<(xzu*NF?25bHT|mIMJK% zC;Ce4ATc}MVXCM&4w_-O{074dp@8HP-u!?&2eDOPg-C~`x#RR}Z*FwM$G7K?rz-%1j{x5ihgVWNHZ6C=*MJjU^7ynCKOHoHp_NTHR$9C?5ML zl%lSQ1rJGm&_Gp9zFqMiI3prxnnZmou9v>*9js&AkiI{qEA@J!9*uB^LP=&m_^wu2 z($WWq06n_P$8Pi}D(O_JiOJJeN?^_MN(3mki>M*e5j2liqe8<{wjfu;19g)xJKIc{ zNS-%6#j|_2W;})v0XPQdni*T?Yqk}1(2f}S0sI4iYE+zw!oD)%acJ)ewixk~W}4lcebdW0GpxVSD#DAC25M*fVl;rlYp8(@ zU8Fk+XX;d~XQR%)w0r$V)?1x!5NN|!#y=XJnS$nP=D2)y(Q|Non4s78nFfl2@L{hM z{o+^kY`ApeRhR*x0AoXQsy}gHe8qX4Ed{IBZCrhMe9PtCSH-t(TnE5m+PI)a>B1A+!YAHD_Ss}a1pXLT3+(34g#Kue0+evL;y^KVSW`Vymw z9sE4Pw9Nd9_roO(6EEg~De?mqVBVI#$cqhp`H{}$MszZLS2xB6;b|-h%g;FT&weuQwFW8g|AYs%r85C zwgz8|?3t*7Ziy8p)#H)Xr1--J!?j`!6lBDww!lluCu(4WaN>ofS}m}qkVG8xLJwiI z_NB4pdFK?a?y;|E0#yE@_H#7ZgtAs?ntA~uXYC=TRK?iPpjMc zROm?cmNmwFsr=3%pER^A9wIw`^u;#ez6J$XzAXP;WX)W~JK0Xw*s3tiMnn>@1%- zS9Ne1Q7b7WYcn|bMw%S%M`S5-Jpm;UAPHQ(_U-mGo8aAnF~E~VdfU@`<%FsRIsX7= z@Z$3BUluLXzr9T>etf>$AT~cAUJgOAz0J7=ALZ~Wd2~c;NkLP=U=d*wXojs$H0>K? zh8j=%L3)8)8_M_I?4P#sSRrRR)f*EY2w@EXL@D`3ZGv3Y(6F|6@buxEYFZUXDzBcB-B~tWh&IvlI4r+zc^)Pg^tApS4QzFpmWTjU zFc1%BD7eSO(eH;UcLUtP0v^zVISTZ_WVHanWNR4^lbqh|7b22D;H;Z&+lbTc;qp>5 ziRenXEyT-(@QvHsHPf4)BwX~qoxGl0>>CpYvG`S|yjo3HH^4uEWa4B;xCu|>fH|s| zCvw$8#oTfN= zb>kUB)P-Q~aPkcpB!2e(aK)x<7!FD%QC(_0W-9`gCtGr?h3!kB1zES)Y>7Z~P_fdR zg;+4}7yFSAo9+!q5NO`oG3n8NTaXHDA~`vzZtAFW2Iv_FE(1Ea)4W9jG`U2Uon|>d zN!QdC&Qs4ebPtx0$x=t0T<*rs&yq1=4VnC)kM+SyFbV^oHkbiG;dWP2?A-<*H_Kmj zW|U{y-sxGjQE|c3FkDSN1EcttzV1fOQLYx{{+~8PV>`|ovfwIgpk{-@3LMaS~jpODxRl|VyQq>>vRP*tyn)=2?JXx^x61? zWWmJZ=iR0{m1A($Dlj3aZh={aaaK~Yf0v#J*Nb(p!AyAjYfg2MULcUc8*dh$?g-71 z9*j&2>EfHKV*%z!t1#(U4EUWs*)d!E=3 z89C&HkO}mqU;1^mxV$t}P^9f7n2{F0{Ejm>-;6{Q{>*7yokblxfLGel4laXP1&TL| z({DmvzVXR$evA>(3~p`uLU$5z^JB!3j77P}-9Ly4nQ^0;ZFt>W;9oC}JKIE7G9 zTyZpfez7kJDsWga9AQ!{3mMBKUi!^zw9l4^&MwPer! zs$ub9XSg99eY7x7b#){c{s(Jccj}m`m1bamy12HS-5b&IQ9(Vv&3We;HXQ+O*i<2ka>J|^#DBc=J=2fM%KwaqyiK(M* z`L$XMty{9i>;Y~(3l1=os*an%fn8)+;vqwe7=9#Ny(gh*5Qc+;QkKrZ97}+2n!RuP#ZBG8ROv4ai3*pwxan?;# z0JoaJr3ULw)&)4lAQy#<6yfUdfFVTzh)RP9e;_GPvQX(`lj6Zcp}9pc_y#_xn|%6D zFE^3jO?TD)3@>xdDA_)71HDF39tlhcD=F1~^X4pZ_+V}&v9McFKqB~bw{8LsgA;u7 zeh^%0ahkg~n>Hd}Y`hKX7UTtVAwWv(Em_thAR_r6lIVSnP5#+;ar(i>YSN#1rTQwh zGvNBeJ3_1|Uz!n8`Q6G!$wPaErPdM1=cw5GL$?x2L`s~8n?K{ISC4J^#Y2A^s!3s* z+A$A{Qg|eO#)ZHFZ67>0_D>>G82MltCTAlC+of^IK2SdFFAo{a!2pXX$cQur_1G7* zD~g|o%gVK{}Cs;*pAK4FHv;0YFwrO&yZRaMQY+TwXX(pg|6dNmo$_Z3YSJ zxHuLHHB?Cs#iEQ7U}$hbbtd&wnR~-E1s*iT(PXG@ruREJo<6y%LfrL5xCwdu({P+} z`384p`m#q0$}9-s?QRelA-=!fowI#8)0r+!LAaOpjhAfc-n^}w4;H&|+@y#q85W&Z zBTy9~OgCK#uK`j@P6x?Camfa^GPWZ*u|EmxJJmMBDo_}QTSF_kb^zLvZx1-rKJx)5 zK6<5Fa}F)2at1;tn)(Xa5~G%G%baO5?PNOGw&P~nmV|4x10+!bWX4GWm*z%)3R=KT z|4b2l#=~iku*u!(?OeE0KAHw|>$RuZY;qN@&RaAtYj|i1pMcWr4no_J{a*WYXEBc1 zO+8M9w^1oZNcNlBx+0u>fOOMM+1G$ytfhz34!ci8S8I+pj*}sMc(Aa=)TS1&%8J|oBh>;03$U12 z59^$o=rpene9ujdj{xa-ycm?fW;Dw?f;SfhR1KZrsLujleHVrc6z!j06jV|70U6if|hW=;Q4!fEE5Tn8BXzO*YcBjc$a2xkX@f2Pc-+H(P_! zWAJt~mP79dcDmdRgbsqeeRO2B*&qsW(cX??g!Fn?E_)EY?>LpIEKS*KGX0CRK`)I~ zSV08rooJ>@U^{6{Ku$5ZQ&CFUw z*XyCmDlf8%k{!~4E%Wl)Ws+FY6q++bk4>5dFPjxANJ$Ar@K}|qQNED`$t7V3ky?+U z`HD;bgi6=SmqUvy`B=o=X!0iGDEvp+U8#zo+Yy#Jh;E&`JJPQj@gGop~|g)pM)5l=~af ztG0qUTMxnfrL60(LN7#QB{UP?stmPU=FQZ@I?aLvFKEO*dUsh~7rtCciKgg%4ckZy z_7Kkmdw`>JTGrLPS@9?dl-HfWcmY%#uO}UWB*>=l-0L}=k_9=Hv4FO=!3}yPq4$aH zzNYotiPf}u@bES5rDis5-c_&}cwb%C?t8Q`6Y(ByOp+eJB5!WK2OL7;lzw}|gFWDq z@m66PoBq855RJo~&KSKavD^S)vs=~T1fCqp8E{gY{MN)o3mkdFvn^aK0mb{r5p1s` zdfw8`P6J+pCE~i*oY`6-f$VX#YAw+p+^2P`5Wbjk>oh-JdQWe5S+fw(kZYKoTU}yW zQE0;NjVXF>w0e+eg~UJ?-a6T(z0^Q){V-{@tdar+Q)*r9R#j4JOm#1c=}i`Bl?F&#gdk_W2g+D4Z+3rHvgr7=Thkgh zFW+{tY_ES<$4$#aRaFK>QNes~6{jx_m21hc=zJ=-N}MPOg;Vyn_vX4hnpGFh&sA!w zXc!GJMgv2WL-3DDVm%;9d>qF9gGsuYORV)jpW-PcWw6dvI!hse(E;-(=jV-cjqOpMA%qMSuxOGN(mDT3{6FHOA=@OXy4onyev$yJIw?_Jm%$B|QzRFdE2`u1%2r$>>V`rho$>?Pzlvgu-ypl3FSjeH|9 z&Od`CB+LCJJjP$aV$7X}ebF==gZ}@*V|I2s`wX{B!NrJb< z=+hoc;nJb#EG~$i5^*+7YD0dh84|=1CrI3(+Aq`Q2O8F7lge+Tw)hVX8nQKAi(Y}1 z)>v)INOLM|_sAMFkU}*b)6g?-x_+H%YUJ$XDK#B=@->@{P6YvF&Njv%4<9AZDjg)3rkM?o9MWe5W$BAT_O&IY0rjY;e-PgaVCZ5uGe z*4(jxBt_-6Y`R7K$I4Jcih7+^pT-IxNytY8SB_Ig)ilEiWHK1YERkeNXB8wEEzK*TE}zpEpfR$}46b$&I!;;(($8(J*fXte;e@&a zdVS~bVfC)Tn)M<O{G}Ey2+W8lAsRmW{VUtZWL^F3~G~X4!%@j9!!7J`W;kh4L5s3)G- zhDZNHw}f;N;@w zuNk$uoroW7a|{0R;m_B@m(yNJ5P^3S3|}TJ=LFUwy;6Fs+s2OYBm|*|(S-un5A?y? zY+@9a98gX~6rH{VO#9)9WsE%Mnq3N4*ZW~s7VRG)?R?mbMZNv#>0wAx%&zZDO;W{A z!SB1;M~p4f4gwNic)Ec`PZn)`iWh=VRbvC=IQ8^yG}wl4bFlwEklkUZ<@Zzfxr%7? z5O8$@4uX{ny^Kb!J0_6}@^)&vmOgTg^1j%9e{NnedH6HajET+{!xeCwMw1dx{KhSx z8M7uKHoY2NSnL}@j0HRSaorhf+Ow~+@+{j^Jp3W2+GTHLqWAOR>hzA$h9gfqB`!}_ z+?~Xo97@}6L{1-?BrWMf&(%QTvbK<)MP<5p^HjJw{mOF6@MOx(WLQ>$GZN?!4HtW3Ekv;+zFP)PiYtq*>F@%yWPr5nUxKw*kx}u zfYG|aK=O1=F-76g8I&0w-=iBbY2$m&-hj6?bagM~47Ej9cc{FTHW<-S3t{}!lsAo3 zY!r)+grvO?4$IYMTUnFr=Q?dlrYlISOh2)+T*O}u&y2`+#OWuUn)EYQ=Xo}A zD=y-!Txc;`UB9tXZb%CB?04N>wt^mabRG`@)Sq@t5zb^L%((d8z7iW zhBTKd9l-{Dwu*23n^T>>^|y1DKwSfoU5PoJ&iO4?3#1f2xmlJ@dqsr4>r{#NJ&urJ z&}}Va$}dWucJeFLM6*!TbAn#CvJ+Rs<}y-{E-wGv|gY31d07Vi!(Cq3)t z8jdtk&bAS(ldrC-dy?WLMVEW!v{PPWn&RLSC9gYeT#I7HrAwZ2(~rM3Q`~nRT%tXtASMKR} za%htAg067YFkW%3SHs@hTwP9ue#Eh1#2ae3ONjKKbNpJi%cB_(ixLf5c zGn(m;1ImP*=e*sVQ4Bfe1r?Hs(@Wiofd5MC8m|c;i4B_x)RFoD*=5@a3|CBwGK=?-bRLVlC!*js6@`(Da$kQz+*ugu$-VK#^r0`u#5;wNg}MLmR{DXbo5b3g_ST)kK|9|38P7ez|y|M=`HW5-)_a_H0SE4=@8ZnS;| z7Hwp7U~oh<6-FxR|K0u71KVRcIrQxB#s`S&r5(-}1ZAt&pqHXTGoX z#-+KUt2nZ>{+}NHb$UM{h0@30-fKNvq)~jiI8qayTK(H1aiTa<b%@FfxDlOg;9fYP#)SM2x3#i05WuHE#ItoZc*|O3>S@7rpkuPb|&Jp^txP zFPu>IdrKm9_2r?Oj~6b<$)OKGVgBx!`f8pd;ri8*NNv3r6+T$bUG9jk(ny0SERCG! zJaQZV+UPuSm#Y3yX=JH$_x+CekJ3nUy@}t87`|nuH`<3kd^9ZD%OWeB&;KechUv>1 zge8BVEVA5bs&>UMU&2L-r{)ED4L0#clPfyPQS*0eT`^J~S?OH(m@6JCk6i3L^Q0>p z?}Rh!yo$&|r>rU>mQ+M8an|mPhyxXo*y5iC0=Md+-7~d+XT;leQo$SyRKF-KiBz@+ zLu7i@b`1Ec@$-n>sh@-^TjzfsiHJulFt8<`iiqD=V9b|&DI!)~2B&cwDlx*mZ$-q7 zl}NthyAkoPm60{htnWp{%0Bq@tFOYu);<*x+p8j*ov%KplAo%IT;TlZrHFWTAN&$_ zRHKo#Zy+xP2Y|O#0VRBXbM-*+?>$q|UvRlO+J)}3Rgs$2KF658;=6uhO-&j_k9T;4 z!g889RTTlm54t75V)=K^RLk!pqUsYXvoPsG8?d|i{*(&zW{fiH^58Ta-Y9KUssX&dEOH12|?06KA{~K3IYi^ zJF9g*)tVDnS~mDY+^ zl+UBvYPDOBTCLO_>tdJIdhAnm@u)|s)_T<{2;Rbq?j+G-xMge){v@@bcJoYPAF$(b zyv0~ywwi2aOI42EGR>;(>1$LLW64uvNkdD+1HNJ3o(kL$ecx@WB_v4|dK)uhYyLlp{jZMXXZl4El0&xyhg}HN9bsuh3VOn)SsA z8X4Sd#47N-gjdl`wP*?tb1tBH-r+v$AH0G&&e7uFs6-+m=_1m>Xc_4fG2Upkn#|?T zR~pN08dZhfVm0cfo6R$HMrNV;m(VK+{&N{cK)Z|tLGBea!h7_Ybz9cOaGaw(@%N7a zychQO!ig{?k7R)&1bM>lJmL$3LQpR#3PDFfhQTC`-VVgGy`d-_4F^*U zPJsm)B7@3khE^40QB07j*aM4Wks3mcSOzo}y$IK0v3kWs*6Tx962l1@3=0TJ07{6T z!!FkwY^F+A0+1bxlrSd=d16%#%8f-d)x!5>6K0#|U>xxYVTYLgsO~>(0 zIGr4W<#Mn_qact+BOf`tJ^EsuzODzThGI<80F??79> z0Dq2iv>4|tp%eP!G#;i6!b9n|gK(H9%sM0l(z56AeBOI|;ps)QB00{{o?L>+QMiki z=Hn{}4urCEa48g})6d4>R34N?*q2tngjGD9JqbSwfEP+cAMeBqTfJs`bDX2Oe9)E( zd=L#z?_V@PpP8ZWr_V@F%gPu~oRX1Y=%1RQFUm+sEy*e|8q+evW$k3 zB4cJ&MykPJ%+P0LrKi!07W}?kc4|h=hKWj!o2Z1vSJ*{b`4+B$m}>ko^d5kH=-O(m z^aKSFqwC+tzfsi!JXnOSOK>;@uf$qvSc-MLy-p{=+Ev(J{<3A3QKNExLK)ybbT8l( z99ED|_+Fgri7J>fY?WyMV=zb4xTxmwElC05*BIbIhNOfeU*l?a;{2L$Al{DSZ zFE*7=b25*)oNnR?W23c=r!zN4aopx8dPE|(5VU!cSlE96_lC(bCIO=_p(!*~PO339 zX@vmT?j`usbRSZIpyO>xg!i%~FPfwvV-Q{MM-mYo)r;s6J**@)Op%H#^@P&zB@fz| zNRA*lbQTBG4H_~D!H1c|lfJ1XLlBI;%P6c@H=~RH_9Zf^OC|3hdN6}5L3CXvu@Kq* zMOQbzH-zKf8v-myHO21TzSB^~is@)ZJO zi=d!=-XvdQL0w@ns-b3MAWKbLi8 zaaU{ec9|}{B@>!4>B`yNrOS}lG<9rz8pXB9s^YRxU_a|P0 zy5qdo-myFg4m-wF3&mjA=CI=vY^J=BctV$nk=Ljb`~c9M6@p>L^Vkd4o?x?op+^UUTA1)Y3^efuO2Sa6bIqw-k|@X1EepZVb7>M6b#{C1Mhv^Q7s_HL~Q zkbbKVNrH|j(H9B=g%HQsQ!oCmec5)J&wZLXJUY#%J2D>P8nc|?2RUPm+$gHx@EKkP zXaCLivf?C`!{sx4u5*9w+m?Gc_CCeo2FicpwQ&9yBC}5&B6kJa{JZGzH-r%u8ZY&> z-)LtWag6r@pBCQFQJ2UeZx8aI+7|x6aM_IT7oRi4ahxFz9!CjpKvOom{7pfk0B@;< z@$|l0xa{p?FqfB@rsvVluM*nzsFA3V|#@WTDX-2Cp9->E-(rl;*Kk0W8@4JHBIH>6N}yB(4W z5tHF|NfG`>xy!^N*v;>%V3^!4B|vjKn^O}FOFEcx@EuZExLYmjHb-QYNr%+iQQFF3 z>viVetxk2ru8nhH`J2q8oAA5A=8ti~Cft-l!rcVl4feCn1!MXz2YQpO>?Z7P*!pT0 zmg&44yqGkG_lPvd#k*n0*SWAvw!OgC%CK(o>W01jsSC^0b|ALxaCIi!mck->g!Rw< zk2WG$dRxj0@6jHPi*<85;hM`+hs1cy=MD>@M*{wg9pMGP-eNC+iGs`-F6^666ztu) zo((ktV_ksH??`DlqR`cObVuq>9@;SM*-Df!?yjVW_=8Cqo3di;k7q+o%3K0BckuzR z_^y;lChR8Acvs3Hoj(zHcvs3M$IlTMbWe(j=(&o2cCqSr0&m`9lQe!j?71i9lG1m1 zcyv$7BX|)HgYUC1!_V=M93(1X^?kO?9IXHc?n`Qz`9KOKTUf8XUJP_)*x)dErwj1n zhhl)sRJcP_uzJq*Y^dooN&rPCTju-(hU#Psm*@mo+9{17SJ@ZMK~V#R>`p>(7jqS5 zaGI2$I3zN%>|)C=Ha#(2KPq~|7Y@^oN&#BBq?bulwE#J7Vi?T*P4t0D|7C92{SGCa z?+P&Bp_B?utS9`fR^ZB3G)u9rg%Z`$vjMKzBtVv0ihbG}*@LE$b{xmA0a)AYTw231+arLPiPeJssWu0kkz0ga7~l diff --git a/plugins/marginal/runtime/provenance.json b/plugins/marginal/runtime/provenance.json index 7052df1..e9d2c69 100644 --- a/plugins/marginal/runtime/provenance.json +++ b/plugins/marginal/runtime/provenance.json @@ -1 +1 @@ -{"builder":"scripts/build_codex_plugin.py","python_requires":">=3.10","schema_version":1,"sha256":"9843b084dd481f29762870e6e6f79a301bdd9c15b7a8c2dc1c4104abc332b259","source_hash":"081703fe680da21b5d261dc1828071f8667c73ed74afe58399b33a25cf98360e"} +{"builder":"scripts/build_codex_plugin.py","python_requires":">=3.10","schema_version":1,"sha256":"73f73a7578f15d9bec3f4a0fcc0882514d8f5c370865bf3bbd6c90d72ef4b5dd","source_hash":"c96b9c73d8cd9191af890a6aa3f71568ceb86bf155d41cd1e94eee2b23e0a6e0"} diff --git a/schemas/decision-receipt-v1.json b/schemas/decision-receipt-v1.json new file mode 100644 index 0000000..ebe948d --- /dev/null +++ b/schemas/decision-receipt-v1.json @@ -0,0 +1,69 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/decision-receipt-v1.json", + "title": "MARGINAL Decision Receipt v1", + "type": "object", + "required": [ + "schema_version", + "decision_id", + "timestamp", + "context", + "decision", + "reason_code", + "state_hash", + "evidence_hash", + "trajectory_hash", + "policy_hash", + "decision_hash", + "confidence", + "expected_utility", + "estimated_cost", + "enforcement_level", + "trust_snapshot", + "governance_cost" + ], + "properties": { + "schema_version": {"const": "1.0"}, + "decision_id": {"type": "string", "minLength": 1}, + "timestamp": {"type": "string", "format": "date-time"}, + "context": { + "type": "object", + "additionalProperties": {"type": "string", "minLength": 1} + }, + "decision": {"type": "string", "minLength": 1}, + "reason_code": {"type": "string", "minLength": 1}, + "state_hash": {"type": ["string", "null"]}, + "evidence_hash": {"type": ["string", "null"]}, + "trajectory_hash": {"type": ["string", "null"]}, + "policy_hash": {"type": "string", "minLength": 1}, + "decision_hash": {"type": "string", "minLength": 1}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + "expected_utility": {"type": ["object", "null"]}, + "estimated_cost": {"type": ["object", "null"]}, + "enforcement_level": {"type": "string", "minLength": 1}, + "trust_snapshot": {"type": "object"}, + "governance_cost": { + "type": "object", + "required": [ + "wall_clock_ms", + "cpu_ms", + "memory_peak_bytes", + "storage_bytes", + "tokens", + "model_calls", + "additional_tool_calls" + ], + "properties": { + "wall_clock_ms": {"type": "number", "minimum": 0}, + "cpu_ms": {"type": ["number", "null"], "minimum": 0}, + "memory_peak_bytes": {"type": ["integer", "null"], "minimum": 0}, + "storage_bytes": {"type": "integer", "minimum": 0}, + "tokens": {"type": "integer", "minimum": 0}, + "model_calls": {"type": "integer", "minimum": 0}, + "additional_tool_calls": {"type": "integer", "minimum": 0} + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/schemas/governance-ledger-v3.json b/schemas/governance-ledger-v3.json new file mode 100644 index 0000000..5edbe82 --- /dev/null +++ b/schemas/governance-ledger-v3.json @@ -0,0 +1,30 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/governance-ledger-v3.json", + "title": "MARGINAL Governance Ledger Record v3", + "type": "object", + "required": [ + "schema_version", + "sequence", + "timestamp", + "previous_hash", + "payload", + "payload_hash", + "record_hash" + ], + "properties": { + "schema_version": {"const": "3.0"}, + "sequence": {"type": "integer", "minimum": 1}, + "timestamp": {"type": "string", "format": "date-time"}, + "previous_hash": { + "oneOf": [ + {"type": "null"}, + {"type": "string", "pattern": "^[0-9a-f]{64}$"} + ] + }, + "payload": {"type": "object"}, + "payload_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "record_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + }, + "additionalProperties": false +} diff --git a/schemas/progress-evidence-v1.json b/schemas/progress-evidence-v1.json new file mode 100644 index 0000000..3effc9c --- /dev/null +++ b/schemas/progress-evidence-v1.json @@ -0,0 +1,25 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/progress-evidence-v1.json", + "title": "MARGINAL Progress Evidence v1", + "type": "object", + "required": [ + "schema_version", + "level", + "state_hash", + "evidence_hash", + "confidence", + "verifier" + ], + "properties": { + "schema_version": {"const": "1.0"}, + "level": { + "enum": ["activity", "information", "progress", "verified_progress"] + }, + "state_hash": {"type": "string", "minLength": 1}, + "evidence_hash": {"type": "string", "minLength": 1}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + "verifier": {"type": ["string", "null"], "minLength": 1} + }, + "additionalProperties": false +} diff --git a/schemas/trust-snapshot-v1.json b/schemas/trust-snapshot-v1.json new file mode 100644 index 0000000..e9d8e85 --- /dev/null +++ b/schemas/trust-snapshot-v1.json @@ -0,0 +1,33 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/trust-snapshot-v1.json", + "title": "MARGINAL Trust Snapshot v1", + "type": "object", + "required": ["schema_version", "context", "components", "confidence_band", "eligible_authority", "current_authority", "authority", "blockers", "transition_receipt"], + "properties": { + "schema_version": {"const": "1.0"}, + "context": {"type": "object", "additionalProperties": {"type": "string", "minLength": 1}}, + "components": {"type": "object", "additionalProperties": {"type": ["number", "null"]}}, + "confidence_band": {"enum": ["low", "medium", "high"]}, + "eligible_authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "current_authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "blockers": {"type": "array", "items": {"type": "string", "minLength": 1}}, + "transition_receipt": { + "type": ["object", "null"], + "required": ["schema_version", "context", "previous", "current", "evidence_ledger_root", "ledger_records", "blockers", "receipt_hash"], + "properties": { + "schema_version": {"const": "1.0"}, + "context": {"type": "object", "additionalProperties": {"type": "string", "minLength": 1}}, + "previous": {"type": "integer", "minimum": 0, "maximum": 4}, + "current": {"type": "integer", "minimum": 0, "maximum": 4}, + "evidence_ledger_root": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "ledger_records": {"type": "integer", "minimum": 0}, + "blockers": {"type": "array", "items": {"type": "string", "minLength": 1}}, + "receipt_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/src/marginal/authority.py b/src/marginal/authority.py new file mode 100644 index 0000000..4a445bc --- /dev/null +++ b/src/marginal/authority.py @@ -0,0 +1,115 @@ +"""Progressive enforcement levels and hash-bound authority transitions.""" + +from __future__ import annotations + +import hmac +from collections.abc import Mapping +from dataclasses import dataclass +from enum import IntEnum +from types import MappingProxyType + +from .canonical import canonical_hash +from .governance_ledger import LedgerVerificationReport + +AUTHORITY_TRANSITION_SCHEMA_VERSION = "1.0" +_HEX = frozenset("0123456789abcdef") + + +class AuthorityLevel(IntEnum): + """Increasing power to alter an agent's execution.""" + + OBSERVE = 0 + ADVISE = 1 + SOFT_INTERVENE = 2 + TOOL_GATE = 3 + COMPUTE_GOVERN = 4 + + +def _required_text(value: str, name: str) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a string") + if not value: + raise ValueError(f"{name} must not be empty") + return value + + +def _sha256(value: str, name: str) -> str: + _required_text(value, name) + if len(value) != 64 or any(character not in _HEX for character in value): + raise ValueError(f"{name} must be a lowercase SHA-256 digest") + return value + + +@dataclass(frozen=True, slots=True) +class AuthorityTransitionReceipt: + """An immutable transition attestation anchored to the evidence-ledger root.""" + + schema_version: str + context: Mapping[str, str] + previous: AuthorityLevel + current: AuthorityLevel + evidence_ledger_root: str + ledger_verification: LedgerVerificationReport + blockers: tuple[str, ...] + receipt_hash: str + + def __post_init__(self) -> None: + if self.schema_version != AUTHORITY_TRANSITION_SCHEMA_VERSION: + raise ValueError("unsupported authority transition schema version") + if not isinstance(self.context, Mapping): + raise TypeError("context must be a mapping") + normalized_context: dict[str, str] = {} + for key, value in self.context.items(): + normalized_context[_required_text(key, "context key")] = _required_text( + value, f"context[{key!r}]" + ) + object.__setattr__(self, "context", MappingProxyType(normalized_context)) + if not isinstance(self.previous, AuthorityLevel) or not isinstance( + self.current, AuthorityLevel + ): + raise TypeError("previous and current must be AuthorityLevel") + _sha256(self.evidence_ledger_root, "evidence_ledger_root") + if not isinstance(self.ledger_verification, LedgerVerificationReport): + raise TypeError("ledger_verification must be LedgerVerificationReport") + if ( + not self.ledger_verification.valid + or self.ledger_verification.root_hash != self.evidence_ledger_root + ): + raise ValueError("ledger_verification must validate evidence_ledger_root") + if not isinstance(self.blockers, tuple) or not all( + isinstance(item, str) and item for item in self.blockers + ): + raise TypeError("blockers must be a tuple of non-empty strings") + if not isinstance(self.receipt_hash, str): + raise TypeError("receipt_hash must be a string") + + def payload(self) -> dict[str, object]: + """Return the canonical fields committed by ``receipt_hash``.""" + + return { + "schema_version": self.schema_version, + "context": dict(self.context), + "previous": int(self.previous), + "current": int(self.current), + "evidence_ledger_root": self.evidence_ledger_root, + "ledger_records": self.ledger_verification.records, + "blockers": list(self.blockers), + } + + +def transition_receipt_hash(receipt: AuthorityTransitionReceipt) -> str: + """Hash a transition without trusting arbitrary object representations.""" + + if not isinstance(receipt, AuthorityTransitionReceipt): + raise TypeError("receipt must be AuthorityTransitionReceipt") + return canonical_hash(receipt.payload()) + + +def verify_transition_receipt(receipt: AuthorityTransitionReceipt) -> bool: + """Return whether a receipt still binds its exact canonical transition payload.""" + + if not isinstance(receipt, AuthorityTransitionReceipt): + return False + if len(receipt.receipt_hash) != 64 or any(char not in _HEX for char in receipt.receipt_hash): + return False + return hmac.compare_digest(receipt.receipt_hash, transition_receipt_hash(receipt)) diff --git a/src/marginal/canonical.py b/src/marginal/canonical.py new file mode 100644 index 0000000..91bf427 --- /dev/null +++ b/src/marginal/canonical.py @@ -0,0 +1,25 @@ +"""Canonical JSON serialization for governance attestations.""" + +from __future__ import annotations + +import hashlib +import json +from typing import Any + + +def canonical_bytes(value: Any) -> bytes: + """Serialize a JSON-compatible value deterministically as UTF-8 bytes.""" + + return json.dumps( + value, + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ).encode("utf-8") + + +def canonical_hash(value: Any) -> str: + """Return the SHA-256 digest of a canonical JSON value.""" + + return hashlib.sha256(canonical_bytes(value)).hexdigest() diff --git a/src/marginal/cli.py b/src/marginal/cli.py index 5fa1f66..3802051 100644 --- a/src/marginal/cli.py +++ b/src/marginal/cli.py @@ -75,6 +75,17 @@ def _build_parser() -> argparse.ArgumentParser: ) ledger_validate.add_argument("ledger", type=Path) + verify = subparsers.add_parser("verify", help="verify a MARGINAL governance ledger v3") + verify.add_argument("ledger", type=Path) + verify.add_argument("--expected-root") + verify.add_argument("--json", action="store_true", dest="as_json") + + ledger_migrate = subparsers.add_parser( + "ledger-migrate", help="migrate a MARGINAL decision ledger v2 to governance ledger v3" + ) + ledger_migrate.add_argument("source", type=Path) + ledger_migrate.add_argument("destination", type=Path) + ledger_export = subparsers.add_parser( "ledger-export", help="export a decision ledger with a privacy-preserving profile", @@ -146,6 +157,8 @@ def _build_parser() -> argparse.ArgumentParser: install_parser.add_argument("target", choices=["codex"]) install_parser.add_argument("--repository", default="SignalLayerLabs/Marginal") install_parser.add_argument("--ref", default="main") + install_parser.add_argument("--data-dir", type=Path) + install_parser.add_argument("--autopilot-consent", action="store_true") install_parser.add_argument("--json", action="store_true", dest="as_json") uninstall_parser = subparsers.add_parser("uninstall", help="remove a native integration") @@ -162,6 +175,24 @@ def _build_parser() -> argparse.ArgumentParser: codex.add_argument("--candidate") codex.add_argument("--verdict", choices=["helpful", "waste"]) codex.add_argument("--json", action="store_true", dest="as_json") + + for name, help_text in ( + ("status", "show authority, trust, evidence, and readiness"), + ("doctor", "diagnose the local Codex integration"), + ): + diagnostic = subparsers.add_parser(name, help=help_text) + diagnostic.add_argument("--data-dir", type=Path) + diagnostic.add_argument("--workspace", type=Path) + diagnostic.add_argument("--json", action="store_true", dest="as_json") + explain = subparsers.add_parser("explain", help="explain a redacted decision receipt") + explain.add_argument("decision_id") + explain.add_argument("--data-dir", type=Path) + explain.add_argument("--workspace", type=Path) + explain.add_argument("--json", action="store_true", dest="as_json") + privacy = subparsers.add_parser("privacy", help="inspect local persistence categories") + privacy_commands = privacy.add_subparsers(dest="privacy_command", required=True) + privacy_inspect = privacy_commands.add_parser("inspect", help="show persisted data categories") + privacy_inspect.add_argument("--json", action="store_true", dest="as_json") return parser @@ -172,7 +203,12 @@ def main(argv: Sequence[str] | None = None) -> int: if args.command == "install": from .integrations.codex.installer import install - result = install(repository=args.repository, ref=args.ref) + result = install( + repository=args.repository, + ref=args.ref, + data_dir=args.data_dir, + autopilot_consent=args.autopilot_consent, + ) payload = result.to_dict() if args.as_json: print(json.dumps(payload, sort_keys=True)) @@ -208,6 +244,38 @@ def main(argv: Sequence[str] | None = None) -> int: as_json=args.as_json, ) + if args.command in {"status", "doctor", "explain", "privacy"}: + from .diagnostics import ( + decision_explanation, + doctor_report, + inspect_privacy, + render_human, + status_report, + ) + from .integrations.codex.commands import default_data_dir + + data_dir = getattr(args, "data_dir", None) or default_data_dir() + workspace = getattr(args, "workspace", None) or Path.cwd() + if args.command == "status": + payload = status_report(data_root=data_dir, workspace=workspace).to_dict() + exit_code = 0 + elif args.command == "doctor": + payload = doctor_report(data_root=data_dir, workspace=workspace).to_dict() + exit_code = 0 + elif args.command == "explain": + payload = decision_explanation( + args.decision_id, data_root=data_dir, workspace=workspace + ).to_dict() + exit_code = 0 if payload["found"] is True else 1 + else: + payload = inspect_privacy().to_dict() + exit_code = 0 + if args.as_json: + print(json.dumps(payload, sort_keys=True)) + else: + print(render_human(payload), end="") + return exit_code + if args.command == "ledger-export": from .ledger import export_decision_ledger @@ -225,6 +293,51 @@ def main(argv: Sequence[str] | None = None) -> int: print(f"exported {exported} records to {args.destination} with {args.privacy_profile}") return 0 + if args.command == "verify": + from .governance_ledger import GovernanceLedger, LedgerVerificationReport + + try: + verification_report = GovernanceLedger(args.ledger).verify( + expected_root=args.expected_root + ) + except (OSError, ValueError): + verification_report = LedgerVerificationReport(False, 0, None, None, ("IO_ERROR",)) + payload = { + "valid": verification_report.valid, + "records": verification_report.records, + "root_hash": verification_report.root_hash, + "first_invalid_sequence": verification_report.first_invalid_sequence, + "error_codes": list(verification_report.error_codes), + } + if args.as_json: + print(json.dumps(payload, sort_keys=True)) + elif verification_report.valid: + print( + "valid governance ledger: " + f"{verification_report.records} records; root {verification_report.root_hash}" + ) + else: + print( + "invalid governance ledger: " + ", ".join(verification_report.error_codes), + file=sys.stderr, + ) + return 0 if verification_report.valid else 1 + + if args.command == "ledger-migrate": + from .governance_ledger import migrate_v2_to_v3 + + try: + migration_report = migrate_v2_to_v3(args.source, args.destination) + except (OSError, ValueError) as exc: + print(str(exc), file=sys.stderr) + return 1 + print( + "migrated " + f"{migration_report.records} records to {args.destination}; " + f"root {migration_report.root_hash}", + ) + return 0 if migration_report.valid else 1 + if args.command in {"ledger-report", "ledger-validate"}: from .ledger import read_decision_ledger, summarize_decision_ledger diff --git a/src/marginal/diagnostics.py b/src/marginal/diagnostics.py new file mode 100644 index 0000000..a7ba027 --- /dev/null +++ b/src/marginal/diagnostics.py @@ -0,0 +1,456 @@ +"""Typed, privacy-safe reports shared by the user-facing diagnostics commands.""" + +from __future__ import annotations + +import json +import os +import stat +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +from .governance_ledger import GovernanceLedger +from .integrations.codex.evidence import ( + EvidenceStore, + summarize_evidence, + summarize_verified_evidence, +) +from .integrations.codex.identity import current_promotion_identity +from .integrations.codex.installer import inspect_codex +from .integrations.codex.promotion import ( + PromotionCriteria, + PromotionIdentity, + evaluate_promotion, + read_promotion_receipt, +) +from .integrations.codex.service import read_mode + +_PERSISTED_CATEGORIES = ( + "derived_enums", + "counts_and_metrics", + "pseudonymous_hashes", + "integrity_receipts", +) +_NEVER_PERSISTED = ( + "prompt", + "prompt_hash", + "source", + "command", + "tool_input", + "tool_response", + "transcript", + "credentials", + "auth", +) +_BENCHMARK_BLOCKERS = ( + "BENCHMARK_EXECUTION_BACKEND_UNAVAILABLE", + "BENCHMARK_EVIDENCE_DAG_NOT_IMPLEMENTED", +) + + +@dataclass(frozen=True, slots=True) +class StatusReport: + """A complete local state snapshot; all values originate in local evidence.""" + + payload: dict[str, object] + + def to_dict(self) -> dict[str, object]: + return dict(self.payload) + + +@dataclass(frozen=True, slots=True) +class DoctorReport: + """A read-only health report that augments the status report with runtime checks.""" + + payload: dict[str, object] + + def to_dict(self) -> dict[str, object]: + return dict(self.payload) + + +@dataclass(frozen=True, slots=True) +class DecisionExplanationReport: + """The deterministic, redacted explanation for one persisted decision.""" + + decision_id: str + found: bool + reason_code: str + decision: dict[str, object] | None = None + + def to_dict(self) -> dict[str, object]: + base: dict[str, object] = { + "decision_id": self.decision_id, + "found": self.found, + "reason_code": self.reason_code, + } + if self.decision is not None: + base["decision"] = dict(self.decision) + return base + + +@dataclass(frozen=True, slots=True) +class PrivacyInspectionReport: + """The local persistence contract, without inspecting user content.""" + + def to_dict(self) -> dict[str, object]: + return { + "persisted_categories": list(_PERSISTED_CATEGORIES), + "never_persisted": list(_NEVER_PERSISTED), + "local_only": True, + "dictionary_attack_limit": ( + "Derived hashes can reveal low-entropy inputs to a party with local ledger access." + ), + } + + +def status_report( + *, + data_root: str | Path, + workspace: str | Path, + plugin_root: str | Path | None = None, +) -> StatusReport: + """Build the shared status surface without executing repository code.""" + + root = Path(data_root).resolve() + selected_workspace = Path(workspace).resolve() + identity = current_promotion_identity(selected_workspace, plugin_root=plugin_root) + evidence_store = EvidenceStore(root / "evidence" / identity.repository_hash) + summary, ledger = summarize_verified_evidence(evidence_store) + raw_records = _safe_records(evidence_store) + if not ledger.valid: + summary = summarize_evidence(raw_records) + receipt = evaluate_promotion( + summary, + PromotionCriteria(), + identity=identity, + evidence_root=ledger.root_hash if ledger.valid else "", + ledger_records=ledger.records, + ledger_path=evidence_store.governance_ledger_path, + ) + state = read_mode(root, repository_hash=identity.repository_hash) + counters = _autopilot_counters(root, identity.repository_hash) + active_sessions, stale_sessions = _active_session_counts(root, identity.repository_hash) + hooks_observed = any( + record.get("event") in {"session_start", "decision", "outcome", "session_end"} + for record in raw_records + ) + hooks_active = active_sessions > 0 + configured_mode = _configured_mode(state) + effective_level, effective_blockers = _effective_authority( + root, + identity, + configured_mode=configured_mode, + ledger_path=evidence_store.governance_ledger_path, + ) + eligible_level = "L3" if receipt.is_ready else "L0" + coverage_ratio = ( + summary.covered_actions / summary.coverable_actions if summary.coverable_actions else 0.0 + ) + trust_components: dict[str, int | float | bool] = { + "covered_actions": summary.covered_actions, + "coverable_actions": summary.coverable_actions, + "coverage_ratio": coverage_ratio, + "completed_sessions": summary.completed_sessions, + "reviewed_candidates": summary.reviewed_candidates, + "intervention_candidates": summary.intervention_candidates, + "false_stops": summary.false_stops, + "integration_failures": summary.integration_failures, + "pending_actions": summary.pending_actions, + "unknown_enforceable_outcomes": summary.unknown_enforceable_outcomes, + "enforceable_outcomes_observable": summary.enforceable_outcomes_observable, + } + payload: dict[str, object] = { + **state, + "capability": "Tool Enforcement", + "repository_hash": identity.repository_hash, + "hook_state": "active" + if hooks_active + else "observed" + if hooks_observed + else "not_observed", + "hooks_observed": hooks_observed, + "hooks_active": hooks_active, + "active_hook_sessions": active_sessions, + "stale_session_receipts": stale_sessions, + "evidence_records": _evidence_record_count(evidence_store), + "covered_actions": summary.covered_actions, + "coverable_actions": summary.coverable_actions, + "coverage_ratio": coverage_ratio, + "authority": { + "configured_mode": configured_mode, + "current": effective_level, + "effective": effective_level, + "effective_blockers": list(effective_blockers), + "eligible": eligible_level, + "ceiling": "L3", + }, + "trust": {"components": trust_components}, + "next_promotion_blockers": list(receipt.blocking_reasons), + "ledger": { + "valid": ledger.valid, + "records": ledger.records, + "root_hash": ledger.root_hash, + "first_invalid_sequence": ledger.first_invalid_sequence, + "error_codes": list(ledger.error_codes), + }, + "plugin_runtime_provenance": _runtime_provenance(plugin_root), + "permissions": _permissions(root, evidence_store, identity.repository_hash), + "benchmark_readiness": {"ready": False, "blocking_reasons": list(_BENCHMARK_BLOCKERS)}, + "counters": counters, + } + return StatusReport(payload) + + +def doctor_report( + *, data_root: str | Path, workspace: str | Path, plugin_root: str | Path | None = None +) -> DoctorReport: + """Combine public Codex capability discovery with the same typed local status.""" + + runtime = inspect_codex().to_dict() + root = Path(data_root).resolve() + selected_workspace = Path(workspace).resolve() + identity = current_promotion_identity(selected_workspace, plugin_root=plugin_root) + evidence_store = EvidenceStore(root / "evidence" / identity.repository_hash) + state = read_mode(root, repository_hash=identity.repository_hash) + configured_mode = _configured_mode(state) + effective_level, effective_blockers = _effective_authority( + root, + identity, + configured_mode=configured_mode, + ledger_path=evidence_store.governance_ledger_path, + ) + provenance = _runtime_provenance(plugin_root) + return DoctorReport( + { + **runtime, + "status": status_report( + data_root=data_root, workspace=workspace, plugin_root=plugin_root + ).to_dict(), + "schemas": _schema_validation(), + "effective_policy": { + "configured_mode": configured_mode, + "effective": effective_level == "L3", + "identity": asdict(identity), + "provenance": provenance, + "blocking_reasons": list(effective_blockers), + }, + "permissions": _permissions(root, evidence_store, identity.repository_hash), + } + ) + + +def decision_explanation( + decision_id: str, + *, + data_root: str | Path, + workspace: str | Path, + plugin_root: str | Path | None = None, +) -> DecisionExplanationReport: + """Find one action hash in verified evidence and expose only its allowed fields.""" + + if not isinstance(decision_id, str) or not decision_id: + raise ValueError("decision_id must be a non-empty string") + identity = current_promotion_identity(Path(workspace).resolve(), plugin_root=plugin_root) + store = EvidenceStore(Path(data_root).resolve() / "evidence" / identity.repository_hash) + records, ledger = store.verified_records() + if not ledger.valid: + reason = ( + "EVIDENCE_INTEGRITY_INVALID" + if store.governance_ledger_path.exists() + else "DECISION_NOT_FOUND" + ) + return DecisionExplanationReport(decision_id, False, reason) + for record in records: + if record.get("event") != "decision" or record.get("action_hash") != decision_id: + continue + decision: dict[str, object] = { + key: record[key] + for key in ( + "action_hash", + "semantic_key", + "state_hash", + "evidence_hash", + "outcome", + "reason_code", + "latency_ms", + "covered", + "coverable", + "recommended_stop", + "reviewed", + "false_stop", + "pending", + ) + if key in record + } + return DecisionExplanationReport( + decision_id, + True, + str(record.get("reason_code", "UNSPECIFIED")), + decision, + ) + return DecisionExplanationReport(decision_id, False, "DECISION_NOT_FOUND") + + +def inspect_privacy() -> PrivacyInspectionReport: + return PrivacyInspectionReport() + + +def render_human(payload: dict[str, object]) -> str: + """Render any typed report with a stable, readable, content-free representation.""" + + return "\n".join(f"{key}: {_human_value(value)}" for key, value in payload.items()) + "\n" + + +def _human_value(value: object) -> str: + if isinstance(value, (dict, list)): + return json.dumps(value, sort_keys=True) + return str(value) + + +def _safe_records(store: EvidenceStore) -> list[dict[str, Any]]: + try: + return store.read_all() + except (OSError, ValueError, json.JSONDecodeError): + return [] + + +def _evidence_record_count(store: EvidenceStore) -> int: + return len(_safe_records(store)) + + +def _active_session_counts(root: Path, repository_hash: str) -> tuple[int, int]: + from .integrations.codex.commands import _active_hook_sessions + + return _active_hook_sessions(root, repository_hash=repository_hash) + + +def _autopilot_counters(root: Path, repository_hash: str) -> dict[str, int]: + path = root / "autopilot" / f"{repository_hash}.json" + try: + state = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError, json.JSONDecodeError): + state = {} + return { + "avoided_actions": _non_negative_int(state.get("avoided_actions")), + "recoveries": _non_negative_int(state.get("recoveries")), + } + + +def _non_negative_int(value: object) -> int: + return value if isinstance(value, int) and not isinstance(value, bool) and value >= 0 else 0 + + +def _runtime_provenance(plugin_root: str | Path | None) -> dict[str, object]: + root = ( + Path(plugin_root).resolve() if plugin_root is not None else _plugin_root_from_environment() + ) + if root is None: + return {"present": False, "reason": "PLUGIN_ARTIFACT_NOT_DISCOVERED"} + path = root / "runtime" / "provenance.json" + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError, json.JSONDecodeError): + return {"present": False, "reason": "RUNTIME_PROVENANCE_UNAVAILABLE"} + if not isinstance(payload, dict): + return {"present": False, "reason": "RUNTIME_PROVENANCE_INVALID"} + return {"present": True, "provenance": payload} + + +def _plugin_root_from_environment() -> Path | None: + configured = os.environ.get("PLUGIN_ROOT") + if configured: + return Path(configured).resolve() + candidate = Path.cwd() / "plugins" / "marginal" + return candidate if candidate.is_dir() else None + + +def _effective_authority( + root: Path, + identity: PromotionIdentity, + *, + configured_mode: str, + ledger_path: Path, +) -> tuple[str, tuple[str, ...]]: + """Validate configured enforcement without mutating its persisted state.""" + + if configured_mode != "enforce": + return "L0", () + try: + receipt = read_promotion_receipt(root, identity.repository_hash) + except (OSError, ValueError, KeyError, TypeError, json.JSONDecodeError): + return "L0", ("PROMOTION_RECEIPT_INVALID",) + if receipt is None: + return "L0", ("PROMOTION_RECEIPT_MISSING",) + state = read_mode(root, repository_hash=identity.repository_hash) + if state.get("receipt_hash") != receipt.receipt_hash: + return "L0", ("PROMOTION_RECEIPT_STATE_MISMATCH",) + if not receipt.is_ready or not receipt.verify_hash(): + return "L0", ("PROMOTION_RECEIPT_INVALID",) + if receipt.identity != identity: + return "L0", ("POLICY_IDENTITY_MISMATCH",) + report = GovernanceLedger(ledger_path).verify_prefix( + receipt.ledger_records, + expected_root=receipt.evidence_root, + ) + if not report.valid or report.root_hash != receipt.evidence_root: + return "L0", ("EVIDENCE_PREFIX_INVALID",) + return "L3", () + + +def _configured_mode(state: dict[str, Any]) -> str: + value = state.get("mode") + return value if isinstance(value, str) else "shadow" + + +def _schema_validation() -> dict[str, object]: + root = Path(__file__).resolve().parents[2] / "schemas" + packaged = Path(__file__).resolve().parent / "schemas" + root_names = {path.name for path in root.glob("*.json")} + packaged_names = {path.name for path in packaged.glob("*.json")} + names = sorted(root_names | packaged_names) + errors: list[str] = [] + for name in names: + root_path = root / name + packaged_path = packaged / name + if not root_path.is_file(): + errors.append(f"ROOT_SCHEMA_MISSING:{name}") + continue + if not packaged_path.is_file(): + errors.append(f"PACKAGED_SCHEMA_MISSING:{name}") + continue + try: + root_schema = json.loads(root_path.read_text(encoding="utf-8")) + packaged_schema = json.loads(packaged_path.read_text(encoding="utf-8")) + except (OSError, ValueError, json.JSONDecodeError): + errors.append(f"SCHEMA_JSON_INVALID:{name}") + continue + if not isinstance(root_schema, dict) or not isinstance(packaged_schema, dict): + errors.append(f"SCHEMA_NOT_OBJECT:{name}") + elif root_schema != packaged_schema: + errors.append(f"SCHEMA_CONTENT_MISMATCH:{name}") + elif not isinstance(root_schema.get("$schema"), str): + errors.append(f"SCHEMA_DIALECT_MISSING:{name}") + return {"valid": not errors, "names": names, "error_codes": errors} + + +def _permissions(root: Path, store: EvidenceStore, repository_hash: str) -> dict[str, str]: + return { + "data_root": _permission_status(root), + "evidence": _permission_status(store.path), + "governance_ledger": _permission_status(store.governance_ledger_path), + "autopilot_consent": _permission_status(root / "user-config.json"), + "enforcement_receipt": _permission_status( + root / "repositories" / f"{repository_hash}.receipt.json" + ), + "enforcement_state": _permission_status(root / "repositories" / f"{repository_hash}.json"), + } + + +def _permission_status(path: Path) -> str: + try: + mode = stat.S_IMODE(path.stat().st_mode) + except OSError: + return "not_created" + if os.name != "posix": + return "platform_not_posix" + return "owner_only" if mode & 0o077 == 0 else "too_permissive" diff --git a/src/marginal/fingerprint.py b/src/marginal/fingerprint.py index 2b5677e..97e3180 100644 --- a/src/marginal/fingerprint.py +++ b/src/marginal/fingerprint.py @@ -3,7 +3,6 @@ from __future__ import annotations import base64 -import hashlib import json import math from collections.abc import Callable, Mapping, Sequence, Set @@ -11,6 +10,7 @@ from pathlib import Path from typing import Any +from .canonical import canonical_hash from .models import Action @@ -51,14 +51,7 @@ def _canonicalize(value: Any) -> Any: def _digest(payload: Mapping[str, Any]) -> str: - canonical = json.dumps( - _canonicalize(payload), - sort_keys=True, - separators=(",", ":"), - ensure_ascii=False, - allow_nan=False, - ) - return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + return canonical_hash(_canonicalize(payload)) def fingerprint_action(action: Action) -> str: diff --git a/src/marginal/governance_ledger.py b/src/marginal/governance_ledger.py new file mode 100644 index 0000000..be70566 --- /dev/null +++ b/src/marginal/governance_ledger.py @@ -0,0 +1,607 @@ +"""Append-only, hash-chained governance evidence ledger (schema v3).""" + +from __future__ import annotations + +import json +import os +import re +import stat +from collections.abc import Mapping +from dataclasses import dataclass +from datetime import datetime, timezone +from io import StringIO +from pathlib import Path +from typing import Any + +from .canonical import canonical_bytes, canonical_hash + +GOVERNANCE_LEDGER_SCHEMA_VERSION = "3.0" +_OWNER_ONLY_MASK = 0o077 +_OPEN_SUPPORTS_DIRECTORY_FD = os.open in os.supports_dir_fd +_MKDIR_SUPPORTS_DIRECTORY_FD = os.mkdir in os.supports_dir_fd + + +@dataclass(frozen=True, slots=True) +class LedgerVerificationReport: + """The result of verifying a governance-ledger v3 chain.""" + + valid: bool + records: int + root_hash: str | None + first_invalid_sequence: int | None + error_codes: tuple[str, ...] + + +class GovernanceLedger: + """A separately-versioned, append-only hash chain for governance payloads.""" + + def __init__(self, path: str | Path) -> None: + self.path = Path(path) + _validate_file_path(self.path, allow_missing=True) + + def append(self, payload: Mapping[str, Any]) -> str: + """Append one canonical payload and return its record hash. + + A POSIX advisory lock serializes cooperating writers. Platforms without + that primitive fail closed instead of claiming cross-process safety. + """ + + normalized_payload = _canonical_payload(payload) + descriptor = _open_append_descriptor(self.path) + try: + _lock_exclusive(descriptor) + try: + data = _read_descriptor(descriptor) + report, records, _ = _verify_data(data) + if not report.valid: + raise ValueError("cannot append to invalid governance ledger") + sequence = len(records) + 1 + previous_hash = report.root_hash + record: dict[str, Any] = { + "schema_version": GOVERNANCE_LEDGER_SCHEMA_VERSION, + "sequence": sequence, + "timestamp": datetime.now(timezone.utc).isoformat(), + "previous_hash": previous_hash, + "payload": normalized_payload, + "payload_hash": canonical_hash(normalized_payload), + } + record["record_hash"] = canonical_hash(record) + os.lseek(descriptor, 0, os.SEEK_END) + encoded = canonical_bytes(record) + b"\n" + _write_all(descriptor, encoded) + os.fsync(descriptor) + return str(record["record_hash"]) + finally: + _unlock(descriptor) + finally: + os.close(descriptor) + + def verify(self, *, expected_root: str | None = None) -> LedgerVerificationReport: + """Verify every canonical record and optionally require a known root.""" + + if expected_root is not None and not _is_hash(expected_root): + return LedgerVerificationReport(False, 0, None, None, ("INVALID_EXPECTED_ROOT",)) + try: + data = _read_path(self.path) + except (OSError, ValueError) as exc: + return LedgerVerificationReport(False, 0, None, None, (_error_code(exc),)) + report, _, _ = _verify_data(data) + if report.valid and expected_root is not None and report.root_hash != expected_root: + return LedgerVerificationReport( + False, + report.records, + report.root_hash, + None, + ("EXPECTED_ROOT_MISMATCH",), + ) + return report + + def verify_prefix( + self, + records: int, + *, + expected_root: str, + ) -> LedgerVerificationReport: + """Verify an immutable prefix retained by a receipt. + + A later append changes the current root, so receipts validate the exact + prefix they were issued from rather than comparing against the live tip. + """ + + if isinstance(records, bool) or not isinstance(records, int) or records < 1: + return LedgerVerificationReport(False, 0, None, None, ("INVALID_RECORD_RANGE",)) + if not _is_hash(expected_root): + return LedgerVerificationReport(False, 0, None, None, ("INVALID_EXPECTED_ROOT",)) + try: + lines = _read_path(self.path).splitlines(keepends=True) + except (OSError, ValueError) as exc: + return LedgerVerificationReport(False, 0, None, None, (_error_code(exc),)) + if len(lines) < records: + return LedgerVerificationReport(False, len(lines), None, None, ("RANGE_UNAVAILABLE",)) + report, _, _ = _verify_data(b"".join(lines[:records])) + if report.valid and report.root_hash != expected_root: + return LedgerVerificationReport( + False, report.records, report.root_hash, None, ("EXPECTED_ROOT_MISMATCH",) + ) + return report + + def read_verified_payloads(self) -> tuple[LedgerVerificationReport, tuple[dict[str, Any], ...]]: + """Return payloads only when the complete v3 chain verifies.""" + + try: + data = _read_path(self.path) + except (OSError, ValueError) as exc: + return LedgerVerificationReport(False, 0, None, None, (_error_code(exc),)), () + report, records, _ = _verify_data(data) + if not report.valid: + return report, () + return report, tuple(dict(record["payload"]) for record in records) + + +def migrate_v2_to_v3(source: Path, destination: Path) -> LedgerVerificationReport: + """Deterministically convert safe v2 records into a new verified v3 chain.""" + + source_path = Path(source) + destination_path = Path(destination) + _validate_file_path(source_path, allow_missing=False) + _validate_parent_components(destination_path) + if _path_exists(destination_path): + raise FileExistsError(f"destination already exists: {destination_path}") + _validate_file_path(destination_path, allow_missing=True) + + from .ledger import _read_decision_ledger_stream + + records = _read_decision_ledger_stream(StringIO(_read_path(source_path).decode("utf-8"))) + converted: list[dict[str, Any]] = [] + previous_hash: str | None = None + for sequence, record in enumerate(records, start=1): + payload = _canonical_payload(record) + converted_record: dict[str, Any] = { + "schema_version": GOVERNANCE_LEDGER_SCHEMA_VERSION, + "sequence": sequence, + "timestamp": record["timestamp"], + "previous_hash": previous_hash, + "payload": payload, + "payload_hash": canonical_hash(payload), + } + converted_record["record_hash"] = canonical_hash(converted_record) + previous_hash = str(converted_record["record_hash"]) + converted.append(converted_record) + + descriptor = _open_new_descriptor(destination_path) + try: + for record in converted: + _write_all(descriptor, canonical_bytes(record) + b"\n") + os.fsync(descriptor) + data = _read_descriptor(descriptor) + finally: + os.close(descriptor) + verification_report, _, _ = _verify_data(data) + if verification_report.valid and verification_report.root_hash != previous_hash: + return LedgerVerificationReport( + False, + verification_report.records, + verification_report.root_hash, + None, + ("EXPECTED_ROOT_MISMATCH",), + ) + return verification_report + + +def quarantine_invalid_records(source: Path, destination: Path) -> Path: + """Copy the invalid suffix and a verification report without altering source.""" + + source_path = Path(source) + destination_path = Path(destination) + _validate_file_path(source_path, allow_missing=False) + _validate_parent_components(destination_path) + if _path_exists(destination_path): + raise FileExistsError(f"quarantine destination already exists: {destination_path}") + data = _read_path(source_path) + report, _, invalid_offset = _verify_data(data) + if report.valid: + raise ValueError("governance ledger is valid; nothing to quarantine") + if invalid_offset is None: + invalid_offset = 0 + quarantine_descriptor = _create_quarantine_directory(destination_path) + try: + _write_owner_only_at(quarantine_descriptor, "invalid-records.jsonl", data[invalid_offset:]) + _write_owner_only_at( + quarantine_descriptor, + "report.json", + canonical_bytes( + { + "valid": report.valid, + "records": report.records, + "root_hash": report.root_hash, + "first_invalid_sequence": report.first_invalid_sequence, + "error_codes": list(report.error_codes), + } + ) + + b"\n", + ) + finally: + os.close(quarantine_descriptor) + return destination_path + + +def _canonical_payload(payload: Mapping[str, Any]) -> dict[str, Any]: + if not isinstance(payload, Mapping): + raise TypeError("governance payload must be a mapping") + normalized = _json_value(payload, "payload") + if not isinstance(normalized, dict): + raise TypeError("governance payload must be an object") + return normalized + + +def _json_value(value: Any, name: str) -> Any: + if value is None or isinstance(value, (str, bool)): + return value + if isinstance(value, int) and not isinstance(value, bool): + return value + if isinstance(value, float): + if value != value or value in (float("inf"), float("-inf")): + raise ValueError(f"{name} must not contain non-finite numbers") + return value + if isinstance(value, (list, tuple)): + return [_json_value(item, name) for item in value] + if isinstance(value, Mapping): + normalized: dict[str, Any] = {} + for key, item in value.items(): + if not isinstance(key, str): + raise TypeError(f"{name} mapping keys must be strings") + if _private_key(key): + raise ValueError(f"{name} must not contain private field {key!r}") + normalized[key] = _json_value(item, name) + return normalized + raise TypeError(f"{name} must contain only canonical JSON values") + + +def _private_key(key: str) -> bool: + separated = re.sub(r"(?<=[a-z0-9])(?=[A-Z])", "_", key) + terms = {part for part in re.split(r"[^a-z0-9]+", separated.casefold()) if part} + return bool( + terms + & { + "auth", + "authorization", + "command", + "commands", + "credential", + "credentials", + "output", + "outputs", + "prompt", + "prompts", + "raw", + "response", + "responses", + "secret", + "secrets", + "source", + "sources", + "transcript", + "transcripts", + } + ) + + +def _verify_data( + data: bytes, +) -> tuple[LedgerVerificationReport, list[dict[str, Any]], int | None]: + if not data: + return LedgerVerificationReport(True, 0, None, None, ()), [], None + records: list[dict[str, Any]] = [] + previous_hash: str | None = None + offset = 0 + for expected_sequence, raw_line in enumerate(data.splitlines(keepends=True), start=1): + line_offset = offset + offset += len(raw_line) + if not raw_line.endswith(b"\n"): + return _invalid( + records, previous_hash, expected_sequence, "NONCANONICAL_LINE", line_offset + ) + raw_value = raw_line[:-1] + try: + value = json.loads(raw_value) + except (UnicodeDecodeError, json.JSONDecodeError): + return _invalid(records, previous_hash, expected_sequence, "INVALID_JSON", line_offset) + if not isinstance(value, dict): + return _invalid( + records, previous_hash, expected_sequence, "INVALID_RECORD", line_offset + ) + try: + if canonical_bytes(value) != raw_value: + return _invalid( + records, previous_hash, expected_sequence, "NONCANONICAL_LINE", line_offset + ) + _validate_record(value, expected_sequence, previous_hash) + except (TypeError, ValueError) as exc: + return _invalid( + records, previous_hash, expected_sequence, _error_code(exc), line_offset + ) + records.append(value) + previous_hash = str(value["record_hash"]) + return LedgerVerificationReport(True, len(records), previous_hash, None, ()), records, None + + +def _invalid( + records: list[dict[str, Any]], + root_hash: str | None, + sequence: int, + code: str, + offset: int, +) -> tuple[LedgerVerificationReport, list[dict[str, Any]], int]: + return ( + LedgerVerificationReport(False, len(records), root_hash, sequence, (code,)), + records, + offset, + ) + + +def _validate_record(record: dict[str, Any], sequence: int, previous_hash: str | None) -> None: + expected_keys = { + "schema_version", + "sequence", + "timestamp", + "previous_hash", + "payload", + "payload_hash", + "record_hash", + } + if set(record) != expected_keys: + raise ValueError("INVALID_RECORD") + if ( + type(record["schema_version"]) is not str + or record["schema_version"] != GOVERNANCE_LEDGER_SCHEMA_VERSION + ): + raise ValueError("UNSUPPORTED_SCHEMA") + if type(record["sequence"]) is not int: + raise ValueError("INVALID_RECORD") + if record["sequence"] != sequence: + raise ValueError("SEQUENCE_GAP") + if record["previous_hash"] is not None and not _is_hash(record["previous_hash"]): + raise ValueError("INVALID_RECORD") + if record["previous_hash"] != previous_hash: + raise ValueError("PREVIOUS_HASH_MISMATCH") + timestamp = record["timestamp"] + if not isinstance(timestamp, str): + raise ValueError("INVALID_TIMESTAMP") + try: + parsed_timestamp = datetime.fromisoformat(timestamp) + except ValueError as exc: + raise ValueError("INVALID_TIMESTAMP") from exc + if parsed_timestamp.tzinfo is None: + raise ValueError("INVALID_TIMESTAMP") + payload = _canonical_payload(record["payload"]) + if payload != record["payload"]: + raise ValueError("INVALID_PAYLOAD") + if not _is_hash(record["payload_hash"]) or not _is_hash(record["record_hash"]): + raise ValueError("INVALID_RECORD") + if record["payload_hash"] != canonical_hash(payload): + raise ValueError("PAYLOAD_HASH_MISMATCH") + without_hash = {key: value for key, value in record.items() if key != "record_hash"} + if record["record_hash"] != canonical_hash(without_hash): + raise ValueError("RECORD_HASH_MISMATCH") + + +def _is_hash(value: object) -> bool: + return ( + isinstance(value, str) + and len(value) == 64 + and all(char in "0123456789abcdef" for char in value) + ) + + +def _validate_parent_components(path: Path) -> None: + """Require an absolute lexical path with no traversal or symlinked parent.""" + + if not path.is_absolute(): + raise ValueError("governance ledger path must be absolute") + if ".." in path.parts: + raise ValueError("governance ledger path must not contain traversal components") + current = Path(path.anchor) + for component in path.parts[1:-1]: + current /= component + try: + details = current.lstat() + except FileNotFoundError: + return + if stat.S_ISLNK(details.st_mode): + raise ValueError("governance ledger path must not have a symbolic-link parent") + if not stat.S_ISDIR(details.st_mode): + raise ValueError("governance ledger path parent must be a directory") + + +def _path_exists(path: Path) -> bool: + try: + path.lstat() + except FileNotFoundError: + return False + return True + + +def _validate_file_path(path: Path, *, allow_missing: bool) -> None: + _validate_parent_components(path) + try: + details = path.lstat() + except FileNotFoundError: + if allow_missing: + return + raise + if stat.S_ISLNK(details.st_mode): + raise ValueError("governance ledger path must not be a symbolic link") + if not stat.S_ISREG(details.st_mode): + raise ValueError("governance ledger path must be a regular file") + if os.name != "nt" and details.st_mode & _OWNER_ONLY_MASK: + raise PermissionError("governance ledger file must not be accessible by group or others") + + +def _read_path(path: Path) -> bytes: + _validate_file_path(path, allow_missing=False) + descriptor = _open_leaf_descriptor(path, os.O_RDONLY, create_parents=False) + try: + details = os.fstat(descriptor) + if not stat.S_ISREG(details.st_mode): + raise ValueError("governance ledger path must be a regular file") + return _read_descriptor(descriptor) + finally: + os.close(descriptor) + + +def _open_append_descriptor(path: Path) -> int: + _validate_file_path(path, allow_missing=True) + flags = os.O_RDWR | os.O_CREAT | os.O_APPEND + descriptor = _open_leaf_descriptor(path, flags, create_parents=True) + try: + details = os.fstat(descriptor) + if not stat.S_ISREG(details.st_mode): + raise ValueError("governance ledger path must be a regular file") + if os.name != "nt" and details.st_mode & _OWNER_ONLY_MASK: + raise PermissionError( + "governance ledger file must not be accessible by group or others" + ) + return descriptor + except Exception: + os.close(descriptor) + raise + + +def _open_new_descriptor(path: Path) -> int: + flags = os.O_RDWR | os.O_CREAT | os.O_EXCL + descriptor = _open_leaf_descriptor(path, flags, create_parents=True) + try: + if not stat.S_ISREG(os.fstat(descriptor).st_mode): + raise ValueError("governance ledger path must be a regular file") + return descriptor + except Exception: + os.close(descriptor) + raise + + +def _read_descriptor(descriptor: int) -> bytes: + os.lseek(descriptor, 0, os.SEEK_SET) + chunks: list[bytes] = [] + while True: + chunk = os.read(descriptor, 64 * 1024) + if not chunk: + return b"".join(chunks) + chunks.append(chunk) + + +def _write_all(descriptor: int, data: bytes) -> None: + offset = 0 + while offset < len(data): + offset += os.write(descriptor, data[offset:]) + + +def _write_owner_only_at(directory_descriptor: int, name: str, data: bytes) -> None: + _require_directory_fd_safety() + descriptor = os.open( + name, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, + dir_fd=directory_descriptor, + ) + try: + _write_all(descriptor, data) + os.fsync(descriptor) + finally: + os.close(descriptor) + + +def _open_leaf_descriptor(path: Path, flags: int, *, create_parents: bool) -> int: + """Open a regular ledger leaf relative to a nofollow directory descriptor.""" + + directory_descriptor = _open_parent_directory(path, create_parents=create_parents) + try: + return os.open( + path.name, + flags | os.O_NOFOLLOW, + 0o600, + dir_fd=directory_descriptor, + ) + finally: + os.close(directory_descriptor) + + +def _create_quarantine_directory(path: Path) -> int: + """Create and hold the quarantine directory through descriptor-relative writes.""" + + directory_descriptor = _open_parent_directory(path, create_parents=True) + try: + os.mkdir(path.name, mode=0o700, dir_fd=directory_descriptor) + return os.open( + path.name, + os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=directory_descriptor, + ) + finally: + os.close(directory_descriptor) + + +def _open_parent_directory(path: Path, *, create_parents: bool) -> int: + """Walk every parent by descriptor, rejecting swaps and symbolic links.""" + + _require_directory_fd_safety() + _validate_parent_components(path) + flags = os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW + descriptor = os.open(path.anchor, flags) + try: + for component in path.parts[1:-1]: + try: + child_descriptor = os.open(component, flags, dir_fd=descriptor) + except FileNotFoundError: + if not create_parents: + raise + os.mkdir(component, mode=0o700, dir_fd=descriptor) + child_descriptor = os.open(component, flags, dir_fd=descriptor) + os.close(descriptor) + descriptor = child_descriptor + return descriptor + except Exception: + os.close(descriptor) + raise + + +def _require_directory_fd_safety() -> None: + """Fail closed when the platform cannot safely anchor path traversal.""" + + required_flags = (getattr(os, "O_DIRECTORY", 0), getattr(os, "O_NOFOLLOW", 0)) + if ( + not all(required_flags) + or not _OPEN_SUPPORTS_DIRECTORY_FD + or not _MKDIR_SUPPORTS_DIRECTORY_FD + ): + raise OSError("secure directory descriptor operations are unavailable") + + +def _lock_exclusive(descriptor: int) -> None: + try: + import fcntl + except ImportError as exc: + raise OSError("OS-level file locking is unavailable") from exc + fcntl.flock(descriptor, fcntl.LOCK_EX) + + +def _unlock(descriptor: int) -> None: + try: + import fcntl + except ImportError: + return + fcntl.flock(descriptor, fcntl.LOCK_UN) + + +def _error_code(exc: BaseException) -> str: + message = str(exc) + known = { + "INVALID_RECORD", + "UNSUPPORTED_SCHEMA", + "SEQUENCE_GAP", + "PREVIOUS_HASH_MISMATCH", + "INVALID_TIMESTAMP", + "INVALID_PAYLOAD", + "PAYLOAD_HASH_MISMATCH", + "RECORD_HASH_MISMATCH", + } + return message if message in known else "IO_ERROR" diff --git a/src/marginal/integrations/codex/__init__.py b/src/marginal/integrations/codex/__init__.py index b7d7204..9a820ba 100644 --- a/src/marginal/integrations/codex/__init__.py +++ b/src/marginal/integrations/codex/__init__.py @@ -1,6 +1,13 @@ """Codex hook integration for privacy-safe tool governance.""" -from .events import PostToolUseEvent, PreToolUseEvent, SessionEvent, parse_hook_event +from .events import ( + PostToolUseEvent, + PreToolUseEvent, + SessionEvent, + UserPromptSubmitEvent, + parse_hook_event, +) +from .intent import UserIntent, is_control_plane_action, normalize_user_prompt from .normalization import normalize_pre_tool_use from .outcomes import classify_tool_outcome from .runtime import CodexIntegrationError, CodexSessionRuntime @@ -12,8 +19,12 @@ "PostToolUseEvent", "PreToolUseEvent", "SessionEvent", + "UserIntent", + "UserPromptSubmitEvent", "classify_tool_outcome", + "is_control_plane_action", "normalize_pre_tool_use", + "normalize_user_prompt", "parse_hook_event", "workspace_state_hash", ] diff --git a/src/marginal/integrations/codex/autopilot.py b/src/marginal/integrations/codex/autopilot.py new file mode 100644 index 0000000..b1669d4 --- /dev/null +++ b/src/marginal/integrations/codex/autopilot.py @@ -0,0 +1,301 @@ +"""Receipt-bound, fail-open quick Autopilot state for Codex.""" + +from __future__ import annotations + +import json +import os +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Any + +from marginal.controls import ActionOutcomeStatus + +from .evidence import EvidenceStore +from .intent import UserIntent + + +@dataclass(frozen=True, slots=True) +class QuickReceipt: + """The first-session L3 attestation, anchored to a verified v3 prefix.""" + + evidence_root: str + ledger_records: int + + +@dataclass(frozen=True, slots=True) +class AutopilotDecision: + allowed: bool + reason_code: str + + +@dataclass(frozen=True, slots=True) +class _Pending: + workload_key: str + eligible_family: bool + + +class AutopilotController: + """Persist only derived repeat state; uncertain inputs always pass through.""" + + def __init__( + self, + data_root: str | Path, + *, + repository_hash: str, + evidence: EvidenceStore, + identity_fingerprint: str = "", + user_consent: bool = False, + ) -> None: + if not isinstance(user_consent, bool): + raise TypeError("user_consent must be a bool") + self._root = Path(data_root).resolve() / "autopilot" + self._root.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + self._root.chmod(0o700) + self._path = self._root / f"{repository_hash}.json" + self._evidence = evidence + self._pending: dict[str, _Pending] = {} + self._integrity_valid = True + self._user_consent = user_consent + self._state = self._load(repository_hash, identity_fingerprint) + + @property + def consent_granted(self) -> bool: + return bool(self._user_consent or self._state["consent"]) + + @property + def enforcement_active(self) -> bool: + receipt = self.quick_receipt + return bool( + self._integrity_valid + and self._state["active"] + and receipt is not None + and self._evidence.verifies_governance_prefix( + root_hash=receipt.evidence_root, records=receipt.ledger_records + ) + ) + + @property + def quick_receipt(self) -> QuickReceipt | None: + value = self._state.get("quick_receipt") + if not isinstance(value, dict): + return None + root = value.get("evidence_root") + records = value.get("ledger_records") + if not isinstance(root, str) or isinstance(records, bool) or not isinstance(records, int): + return None + return QuickReceipt(root, records) + + def grant_consent(self) -> None: + """Record the one-time opt-in; no hook starts enforcement without it.""" + + self._state["consent"] = True + self._persist() + + def revoke(self, reason: str = "AUTOPILOT_REVOKED") -> None: + """Demote immediately while preserving the opt-in and audit counters.""" + + self._state["active"] = False + self._state["demotion_reason"] = reason + self._state["last_denial"] = None + self._persist() + + def pre_action( + self, + *, + action_id: str, + workload_key: str, + eligible_family: bool, + state_hash: str, + evidence_hash: str, + intent: UserIntent, + ) -> AutopilotDecision: + self._validate_inputs(action_id, workload_key, state_hash, evidence_hash, intent) + marker = { + "workload_key": workload_key, + "state_hash": state_hash, + "evidence_hash": evidence_hash, + } + last_denial = self._state.get("last_denial") + if last_denial is not None and last_denial != marker: + self._state["last_denial"] = None + self._persist() + if eligible_family and self._is_immediate_recovery(marker): + self._state["recoveries"] += 1 + self.revoke("RECOVERY") + self._pending[action_id] = _Pending(workload_key, True) + return AutopilotDecision(True, "RECOVERY") + if eligible_family and self._has_pending_workload(workload_key): + self._reserve(action_id, workload_key, True) + return AutopilotDecision(True, "PENDING_WORKLOAD") + if not eligible_family or intent.repeat_requested or intent.force_run: + self._reserve(action_id, workload_key, eligible_family) + return AutopilotDecision( + True, "USER_REQUESTED_REPEAT" if intent.repeat_requested else "PASS_THROUGH" + ) + if not self.consent_granted or not self._integrity_valid: + self._reserve(action_id, workload_key, True) + return AutopilotDecision(True, "INSUFFICIENT_TRUST") + history = self._state["histories"].get(workload_key) + exact_repeat = ( + isinstance(history, dict) + and history.get("state_hash") == state_hash + and history.get("evidence_hash") == evidence_hash + and history.get("successes") == 2 + ) + if not exact_repeat: + self._reserve(action_id, workload_key, True) + return AutopilotDecision(True, "PASS_THROUGH") + if not self._ensure_quick_receipt(): + self._reserve(action_id, workload_key, True) + return AutopilotDecision(True, "INSUFFICIENT_EVIDENCE") + self._state["active"] = True + self._state["avoided_actions"] += 1 + self._state["last_denial"] = marker + self._persist() + return AutopilotDecision(False, "NO_PROGRESS_ENFORCED") + + def settle_action( + self, + action_id: str, + *, + outcome: ActionOutcomeStatus, + state_hash: str, + evidence_hash: str, + ) -> None: + pending = self._pending.pop(action_id, None) + if pending is None: + return + if not pending.eligible_family: + return + if outcome is not ActionOutcomeStatus.SUCCESS or not state_hash or not evidence_hash: + self._state["histories"].pop(pending.workload_key, None) + self._state["quick_receipt"] = None + self._state["last_denial"] = None + self.revoke( + "OUTCOME_UNOBSERVABLE" + if outcome is ActionOutcomeStatus.UNKNOWN + else "OUTCOME_FAILURE" + ) + return + previous = self._state["histories"].get(pending.workload_key) + same = ( + isinstance(previous, dict) + and previous.get("state_hash") == state_hash + and previous.get("evidence_hash") == evidence_hash + ) + successes = min(2, int(previous.get("successes", 0)) + 1) if same else 1 + self._state["histories"][pending.workload_key] = { + "state_hash": state_hash, + "evidence_hash": evidence_hash, + "successes": successes, + } + try: + self._evidence.append( + { + "schema_version": 1, + "event": "autopilot_observation", + "action_hash": action_id, + "semantic_key": pending.workload_key, + "state_hash": state_hash, + "evidence_hash": evidence_hash, + "outcome": outcome.value, + } + ) + except (OSError, ValueError): + self._integrity_valid = False + self.revoke("EVIDENCE_INTEGRITY_FAILURE") + return + self._persist() + + def summary(self) -> dict[str, int]: + return { + "avoided_actions": int(self._state["avoided_actions"]), + "recoveries": int(self._state["recoveries"]), + "pending_actions": len(self._pending), + } + + def _ensure_quick_receipt(self) -> bool: + existing = self.quick_receipt + if existing is not None and self._evidence.verifies_governance_prefix( + root_hash=existing.evidence_root, records=existing.ledger_records + ): + return True + report = self._evidence.verified_governance_root() + if not report.valid or report.root_hash is None or report.records < 1: + self.revoke("EVIDENCE_INTEGRITY_FAILURE") + return False + self._state["quick_receipt"] = asdict(QuickReceipt(report.root_hash, report.records)) + self._persist() + return True + + def _is_immediate_recovery(self, marker: dict[str, str]) -> bool: + return self._state.get("last_denial") == marker and self.enforcement_active + + def _reserve(self, action_id: str, workload_key: str, eligible_family: bool) -> None: + if eligible_family: + self._pending[action_id] = _Pending(workload_key, True) + + def _has_pending_workload(self, workload_key: str) -> bool: + return any(pending.workload_key == workload_key for pending in self._pending.values()) + + @staticmethod + def _validate_inputs( + action_id: str, workload_key: str, state_hash: str, evidence_hash: str, intent: UserIntent + ) -> None: + if not all( + isinstance(value, str) and value for value in (action_id, workload_key, state_hash) + ): + raise ValueError("action identity and state must be non-empty strings") + if not isinstance(evidence_hash, str) or not isinstance(intent, UserIntent): + raise TypeError("evidence_hash must be a string and intent must be UserIntent") + + def _load(self, repository_hash: str, identity_fingerprint: str) -> dict[str, Any]: + default: dict[str, Any] = { + "schema_version": 1, + "repository_hash": repository_hash, + "identity_fingerprint": identity_fingerprint, + "consent": False, + "active": False, + "histories": {}, + "quick_receipt": None, + "last_denial": None, + "avoided_actions": 0, + "recoveries": 0, + "demotion_reason": "DEFERRED_CONSENT", + } + if not self._path.exists(): + return default + try: + loaded = json.loads(self._path.read_text(encoding="utf-8")) + if not isinstance(loaded, dict) or loaded.get("repository_hash") != repository_hash: + raise ValueError("invalid Autopilot state") + for key in default: + if key not in loaded: + raise ValueError("incomplete Autopilot state") + if identity_fingerprint and loaded["identity_fingerprint"] != identity_fingerprint: + self._integrity_valid = False + loaded["identity_fingerprint"] = identity_fingerprint + loaded["active"] = False + loaded["quick_receipt"] = None + loaded["last_denial"] = None + loaded["demotion_reason"] = "IDENTITY_DRIFT" + return loaded + except (OSError, ValueError, json.JSONDecodeError): + self._integrity_valid = False + return default + + def _persist(self) -> None: + temporary = self._path.with_suffix(".tmp") + descriptor = os.open(temporary, os.O_CREAT | os.O_TRUNC | os.O_WRONLY, 0o600) + try: + os.write( + descriptor, + (json.dumps(self._state, sort_keys=True, separators=(",", ":")) + "\n").encode(), + ) + os.fsync(descriptor) + finally: + os.close(descriptor) + os.replace(temporary, self._path) + if os.name == "posix": + self._path.chmod(0o600) diff --git a/src/marginal/integrations/codex/commands.py b/src/marginal/integrations/codex/commands.py index 671b730..7eac285 100644 --- a/src/marginal/integrations/codex/commands.py +++ b/src/marginal/integrations/codex/commands.py @@ -8,9 +8,10 @@ from pathlib import Path from typing import Any -from .evidence import EvidenceStore, summarize_evidence +from marginal.diagnostics import doctor_report, render_human, status_report + +from .evidence import EvidenceStore, summarize_verified_evidence from .identity import current_promotion_identity -from .installer import inspect_codex from .promotion import ( PromotionCriteria, activate_enforcement, @@ -18,7 +19,6 @@ evaluate_promotion, write_promotion_receipt, ) -from .service import read_mode from .transport import ConnectionInfo, request_session _MAX_SESSION_RECEIPTS = 64 @@ -121,45 +121,18 @@ def codex_command( evidence_store = EvidenceStore(root / "evidence" / identity.repository_hash) payload: dict[str, Any] if command == "status": - state = read_mode(root, repository_hash=identity.repository_hash) - records = evidence_store.read_all() - summary = summarize_evidence(records) - hooks_observed = any( - record.get("event") in {"session_start", "decision", "outcome", "session_end"} - for record in records - ) - coverage_ratio = ( - summary.covered_actions / summary.coverable_actions - if summary.coverable_actions - else 0.0 - ) - active_hook_sessions, stale_session_receipts = _active_hook_sessions( - root, - repository_hash=identity.repository_hash, - ) - hooks_active = active_hook_sessions > 0 - _emit( - { - **state, - "capability": "Tool Enforcement", - "repository_hash": identity.repository_hash, - "hook_state": ( - "active" if hooks_active else "observed" if hooks_observed else "not_observed" - ), - "hooks_observed": hooks_observed, - "hooks_active": hooks_active, - "active_hook_sessions": active_hook_sessions, - "stale_session_receipts": stale_session_receipts, - "evidence_records": len(records), - "covered_actions": summary.covered_actions, - "coverable_actions": summary.coverable_actions, - "coverage_ratio": coverage_ratio, - }, - as_json=as_json, - ) + payload = status_report(data_root=root, workspace=selected_workspace).to_dict() + if as_json: + _emit(payload, as_json=True) + else: + print(render_human(payload), end="") return 0 if command == "doctor": - _emit(inspect_codex().to_dict(), as_json=as_json) + payload = doctor_report(data_root=root, workspace=selected_workspace).to_dict() + if as_json: + _emit(payload, as_json=True) + else: + print(render_human(payload), end="") return 0 if command == "review": records = evidence_store.read_all() @@ -240,8 +213,15 @@ def codex_command( _emit(payload, as_json=as_json) return 0 if command == "promote": - summary = summarize_evidence(evidence_store.read_all()) - receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + summary, root_report = summarize_verified_evidence(evidence_store) + receipt = evaluate_promotion( + summary, + PromotionCriteria(), + identity=identity, + evidence_root=root_report.root_hash if root_report.valid else "", + ledger_records=root_report.records, + ledger_path=evidence_store.governance_ledger_path, + ) write_promotion_receipt(root, receipt) if not receipt.is_ready: payload = { @@ -252,7 +232,7 @@ def codex_command( } _emit(payload, as_json=as_json) return 2 - activate_enforcement(root, receipt) + activate_enforcement(root, receipt, ledger_path=evidence_store.governance_ledger_path) payload = { "schema_version": 1, "mode": "enforce", diff --git a/src/marginal/integrations/codex/events.py b/src/marginal/integrations/codex/events.py index 509a2f9..00e3ee1 100644 --- a/src/marginal/integrations/codex/events.py +++ b/src/marginal/integrations/codex/events.py @@ -48,7 +48,18 @@ class PostToolUseEvent: transcript_path: str | None = None -CodexHookEvent = SessionEvent | PreToolUseEvent | PostToolUseEvent +@dataclass(frozen=True, slots=True) +class UserPromptSubmitEvent: + session_id: str + cwd: str + hook_event_name: str + model: str + permission_mode: str + prompt: str + transcript_path: str | None = None + + +CodexHookEvent = SessionEvent | PreToolUseEvent | PostToolUseEvent | UserPromptSubmitEvent def _required_text(payload: Mapping[str, Any], name: str) -> str: @@ -113,6 +124,8 @@ def parse_hook_event(payload: Mapping[str, Any]) -> CodexHookEvent: **_tool_fields(payload), tool_response=payload["tool_response"], ) + if name == "UserPromptSubmit": + return UserPromptSubmitEvent(**common, prompt=_required_text(payload, "prompt")) raise ValueError(f"unsupported Codex hook event: {name}") diff --git a/src/marginal/integrations/codex/evidence.py b/src/marginal/integrations/codex/evidence.py index 5cd28ff..6f3400c 100644 --- a/src/marginal/integrations/codex/evidence.py +++ b/src/marginal/integrations/codex/evidence.py @@ -9,6 +9,8 @@ from pathlib import Path from typing import Any +from marginal.governance_ledger import GovernanceLedger, LedgerVerificationReport + from .promotion import CoverageSummary _ALLOWED_EVIDENCE_FIELDS = { @@ -36,6 +38,7 @@ "command", "credential", "prompt", + "prompt_hash", "source", "tool_input", "tool_response", @@ -69,6 +72,8 @@ def __init__(self, root: str | Path, *, max_record_bytes: int = 16_384) -> None: if os.name == "posix": self.root.chmod(0o700) self.path = self.root / "evidence.jsonl" + self.governance_ledger_path = self.root / "governance-v3.jsonl" + self._governance_ledger = GovernanceLedger(self.governance_ledger_path) self.checkpoint_path = self.root / "checkpoint.json" self.max_record_bytes = max_record_bytes @@ -93,6 +98,32 @@ def append(self, record: Mapping[str, Any]) -> None: os.close(descriptor) if os.name == "posix": self.path.chmod(0o600) + self._governance_ledger.append({"event": "codex_evidence", "evidence": dict(record)}) + + def verified_governance_root(self) -> LedgerVerificationReport: + """Return the verified v3 root that can anchor an enforcement receipt.""" + + return self._governance_ledger.verify() + + def verifies_governance_prefix(self, *, root_hash: str, records: int) -> bool: + report = self._governance_ledger.verify_prefix(records, expected_root=root_hash) + return report.valid and report.root_hash == root_hash + + def verified_records(self) -> tuple[list[dict[str, Any]], LedgerVerificationReport]: + """Read only the evidence records committed by a verified v3 chain.""" + + report, payloads = self._governance_ledger.read_verified_payloads() + if not report.valid: + return [], report + records: list[dict[str, Any]] = [] + for payload in payloads: + if payload.get("event") != "codex_evidence": + continue + evidence = payload.get("evidence") + if not isinstance(evidence, dict): + raise ValueError("Codex governance evidence payload is invalid") + records.append(dict(evidence)) + return records, report def read_all(self) -> list[dict[str, Any]]: if not self.path.exists(): @@ -220,3 +251,12 @@ def summarize_evidence(records: list[dict[str, Any]]) -> CoverageSummary: enforceable_outcomes_observable=bool(outcomes) and unknown_outcomes == 0, intervention_candidates=len(candidates), ) + + +def summarize_verified_evidence( + store: EvidenceStore, +) -> tuple[CoverageSummary, LedgerVerificationReport]: + """Summarize only records cryptographically committed to v3 governance evidence.""" + + records, report = store.verified_records() + return summarize_evidence(records), report diff --git a/src/marginal/integrations/codex/identity.py b/src/marginal/integrations/codex/identity.py index fdda22c..321244d 100644 --- a/src/marginal/integrations/codex/identity.py +++ b/src/marginal/integrations/codex/identity.py @@ -13,7 +13,7 @@ PLUGIN_VERSION = "0.3.3" ADAPTER_VERSION = "1" POLICY_HASH = hashlib.sha256(b"marginal:no-progress:v1:max-same-evidence=2").hexdigest() -DEFAULT_HOOK_HASH = "46b7a85a3a542957d055c615ab501f9fee284bb3193ac9ecfbe8951cce5a9942" +DEFAULT_HOOK_HASH = "6d010b4849d8748d59e9d32e49e0ace08fc9a3dbc7a759b70336490b745a6190" def repository_identity_hash(workspace: str | Path) -> str: diff --git a/src/marginal/integrations/codex/installer.py b/src/marginal/integrations/codex/installer.py index 7cd8cc7..b11b411 100644 --- a/src/marginal/integrations/codex/installer.py +++ b/src/marginal/integrations/codex/installer.py @@ -2,11 +2,13 @@ from __future__ import annotations +import json import os import re import shutil import subprocess from dataclasses import dataclass +from pathlib import Path from typing import Protocol @@ -80,6 +82,7 @@ class CodexInstallation: selector: str = "marginal@marginal" error_code: str = "" message: str = "" + autopilot_consent: bool = False def to_dict(self) -> dict[str, object]: return { @@ -88,6 +91,7 @@ def to_dict(self) -> dict[str, object]: "selector": self.selector, "error_code": self.error_code, "message": self.message, + "autopilot_consent": self.autopilot_consent, } @@ -127,7 +131,11 @@ def install( runner: CommandRunner | None = None, repository: str = "SignalLayerLabs/Marginal", ref: str = "main", + data_dir: str | Path | None = None, + autopilot_consent: bool = False, ) -> CodexInstallation: + if not isinstance(autopilot_consent, bool): + raise TypeError("autopilot_consent must be a bool") selected = runner or SubprocessRunner() report = inspect_codex(runner=selected) if report.capability_level != "tool_enforcement": @@ -164,7 +172,21 @@ def install( error_code="PLUGIN_ADD_FAILED", message=plugin.stderr.strip(), ) - return CodexInstallation(True, True, message="installed in Shadow Mode") + if autopilot_consent: + if data_dir is None: + return CodexInstallation( + True, + True, + error_code="AUTOPILOT_CONSENT_DATA_DIR_REQUIRED", + message="installed in Shadow Mode; Autopilot consent was not persisted", + ) + configure_autopilot_consent(data_dir, granted=True) + return CodexInstallation( + True, + True, + message="installed in Shadow Mode", + autopilot_consent=autopilot_consent, + ) def uninstall(*, runner: CommandRunner | None = None) -> CodexInstallation: @@ -178,3 +200,48 @@ def uninstall(*, runner: CommandRunner | None = None) -> CodexInstallation: message=result.stderr.strip(), ) return CodexInstallation(False, True, message="plugin removed; local evidence preserved") + + +def _user_config_path(data_dir: str | Path) -> Path: + return Path(data_dir).resolve() / "user-config.json" + + +def configure_autopilot_consent(data_dir: str | Path, *, granted: bool) -> None: + """Persist an explicit user-level Autopilot choice outside every repository. + + Repository configuration is deliberately not read here: it can constrain + local behavior but cannot grant a user's authority to enforce actions. + """ + + if not isinstance(granted, bool): + raise TypeError("granted must be a bool") + path = _user_config_path(data_dir) + path.parent.mkdir(parents=True, exist_ok=True, mode=0o700) + if os.name == "posix": + path.parent.chmod(0o700) + temporary = path.with_suffix(".tmp") + descriptor = os.open(temporary, os.O_CREAT | os.O_TRUNC | os.O_WRONLY, 0o600) + try: + os.write( + descriptor, + ( + json.dumps({"schema_version": 1, "autopilot_consent": granted}, sort_keys=True) + + "\n" + ).encode(), + ) + os.fsync(descriptor) + finally: + os.close(descriptor) + os.replace(temporary, path) + if os.name == "posix": + path.chmod(0o600) + + +def autopilot_consent_configured(data_dir: str | Path) -> bool: + """Return only a valid, explicit consent bit from the user-owned config.""" + + try: + value = json.loads(_user_config_path(data_dir).read_text(encoding="utf-8")) + except (OSError, ValueError, json.JSONDecodeError): + return False + return bool(isinstance(value, dict) and value.get("autopilot_consent") is True) diff --git a/src/marginal/integrations/codex/intent.py b/src/marginal/integrations/codex/intent.py new file mode 100644 index 0000000..705b96d --- /dev/null +++ b/src/marginal/integrations/codex/intent.py @@ -0,0 +1,175 @@ +"""Ephemeral user-control intent and strict MARGINAL control-plane recognition.""" + +from __future__ import annotations + +import re +import shlex +import unicodedata +from dataclasses import dataclass +from pathlib import Path + +from .events import PreToolUseEvent + +_CONTROL_SUBCOMMANDS = frozenset({"status", "doctor", "review", "promote", "demote"}) +_PYTHON_LAUNCHERS = frozenset({"python", "python3", "py"}) +_SHELL_SYNTAX = re.compile(r"[;&|<>`$\r\n]") +_NEGATED_CONTROL = re.compile( + r"\b(?:do\s+not|don't|dont|never)\s+(?:please\s+)?" + r"(?:repeat|redo|run|execute|proceed|force|pause|stop|suspend|resume|unpause|reactivate)\b|" + r"\b(?:non|mai)\s+(?:" + r"ripetere|rifare|eseguire|esegui|procedere|procedi|forzare|mettere\s+in\s+pausa|" + r"fermare|sospendere|" + r"riprendere|riattivare" + r")\b" +) +_ACTION_OPTIONS = { + "status": frozenset({"--workspace", "--json"}), + "doctor": frozenset({"--json"}), + "review": frozenset({"--workspace", "--candidate", "--verdict", "--json"}), + "promote": frozenset({"--workspace", "--json"}), + "demote": frozenset({"--workspace", "--json"}), +} +_VALUE_OPTIONS = frozenset({"--workspace", "--candidate", "--verdict"}) +_ACTION_HASH = re.compile(r"[0-9a-fA-F]{64}\Z") + + +@dataclass(frozen=True, slots=True) +class UserIntent: + """User-requested control retained only by the live authenticated session.""" + + repeat_requested: bool = False + force_run: bool = False + pause_marginal: bool = False + resume_marginal: bool = False + status_requested: bool = False + + +def _normalized_prompt(prompt: str) -> str: + if not isinstance(prompt, str): + raise TypeError("prompt must be a string") + return " ".join(unicodedata.normalize("NFKC", prompt).casefold().split()) + + +def _contains(prompt: str, expression: str) -> bool: + return re.search(expression, prompt) is not None + + +def normalize_user_prompt(prompt: str) -> UserIntent: + """Recognize only explicit English/Italian controls; uncertainty returns no intent.""" + + text = _normalized_prompt(prompt) + if not text or _NEGATED_CONTROL.search(text): + return UserIntent() + + repeat_requested = _contains( + text, + r"\b(?:repeat|rifai|ripeti)\b|\b(?:run|do|try|execute)\s+(?:it\s+)?(?:again|once more)\b|" + r"\b(?:esegui|fallo)\s+di\s+nuovo\b", + ) + force_run = _contains( + text, + r"\bforce\s+(?:the\s+)?(?:run|execution|action)\b|\brun\s+(?:it\s+)?anyway\b|" + r"\b(?:proceed|execute)\s+anyway\b|\bforza\s+l'?esecuzione\b|" + r"\b(?:esegui|procedi)\s+comunque\b", + ) + pause_marginal = "marginal" in text and _contains( + text, r"\b(?:pause|stop|suspend|pausa|ferma|sospendi)\b" + ) + resume_marginal = "marginal" in text and _contains( + text, r"\b(?:resume|unpause|reactivate|riprendi|riattiva)\b" + ) + status_requested = "marginal" in text and _contains(text, r"\b(?:status|state|stato)\b") + if pause_marginal and resume_marginal: + return UserIntent() + return UserIntent( + repeat_requested=repeat_requested, + force_run=force_run, + pause_marginal=pause_marginal, + resume_marginal=resume_marginal, + status_requested=status_requested, + ) + + +def _has_symlink(path: Path, root: Path) -> bool: + try: + relative = path.relative_to(root) + except ValueError: + return True + current = root + if current.is_symlink(): + return True + for component in relative.parts: + current = current / component + if current.is_symlink(): + return True + return False + + +def _valid_control_arguments(arguments: list[str]) -> bool: + """Validate the small wrapper contract without executing its parser.""" + + if not arguments or arguments[0].casefold() not in _CONTROL_SUBCOMMANDS: + return False + command = arguments[0].casefold() + allowed = _ACTION_OPTIONS[command] + values: dict[str, str] = {} + flags: set[str] = set() + index = 1 + while index < len(arguments): + option = arguments[index] + if option not in allowed or option in flags or option in values: + return False + if option in _VALUE_OPTIONS: + if index + 1 >= len(arguments) or arguments[index + 1].startswith("--"): + return False + values[option] = arguments[index + 1] + index += 2 + else: + flags.add(option) + index += 1 + + candidate = values.get("--candidate") + verdict = values.get("--verdict") + if (candidate is None) != (verdict is None): + return False + if candidate is not None and _ACTION_HASH.fullmatch(candidate) is None: + return False + return verdict is None or verdict in {"helpful", "waste"} + + +def is_control_plane_action(event: PreToolUseEvent, plugin_root: Path) -> bool: + """Accept only a direct invocation of the installed MARGINAL control script.""" + + if not isinstance(event, PreToolUseEvent) or event.tool_name.casefold() not in { + "bash", + "shell", + }: + return False + command = event.tool_input.get("command") + if not isinstance(command, str) or _SHELL_SYNTAX.search(command): + return False + try: + argv = shlex.split(command, posix=True) + except ValueError: + return False + if len(argv) < 3 or argv[0].casefold() not in _PYTHON_LAUNCHERS: + return False + script_index = 2 if argv[0].casefold() == "py" and argv[1] == "-3" else 1 + if len(argv) <= script_index + 1: + return False + script_argument = Path(argv[script_index]) + if not script_argument.is_absolute() or ".." in script_argument.parts: + return False + try: + root = Path(plugin_root) + if not root.is_absolute() or root.is_symlink(): + return False + root = root.resolve(strict=True) + expected = root / "scripts" / "marginal_control.py" + if not (root / ".codex-plugin" / "plugin.json").is_file() or _has_symlink(expected, root): + return False + if script_argument.resolve(strict=True) != expected.resolve(strict=True): + return False + except (OSError, RuntimeError): + return False + return _valid_control_arguments(argv[script_index + 1 :]) diff --git a/src/marginal/integrations/codex/normalization.py b/src/marginal/integrations/codex/normalization.py index e793e54..b7dceaf 100644 --- a/src/marginal/integrations/codex/normalization.py +++ b/src/marginal/integrations/codex/normalization.py @@ -5,12 +5,14 @@ import hashlib import json import re +from pathlib import Path from typing import Any from marginal.models import Cost from marginal.protocol import AgentAction, DeduplicationScope from .events import PreToolUseEvent +from .intent import is_control_plane_action as _is_control_plane_action _VERIFICATION_PATTERN = re.compile( r"(?:^|[\s/])(?:pytest|tox|nox|unittest|jest|vitest|mocha|rspec|" @@ -20,6 +22,12 @@ ) +def is_control_plane_action(event: PreToolUseEvent, plugin_root: Path) -> bool: + """Expose strict trusted-control recognition beside action normalization.""" + + return _is_control_plane_action(event, plugin_root) + + def _canonical_json(value: Any) -> str: try: return json.dumps( diff --git a/src/marginal/integrations/codex/promotion.py b/src/marginal/integrations/codex/promotion.py index 99cb3b5..f328224 100644 --- a/src/marginal/integrations/codex/promotion.py +++ b/src/marginal/integrations/codex/promotion.py @@ -10,6 +10,8 @@ from pathlib import Path from typing import Any +from marginal.governance_ledger import GovernanceLedger + @dataclass(frozen=True, slots=True) class PromotionIdentity: @@ -82,6 +84,8 @@ class PromotionReceipt: blocking_reasons: tuple[str, ...] is_ready: bool receipt_hash: str + evidence_root: str = "" + ledger_records: int = 0 def _hash_payload(self) -> dict[str, Any]: payload = self.to_dict() @@ -92,7 +96,12 @@ def verify_hash(self) -> bool: return self.receipt_hash == _hash(self._hash_payload()) def valid_for(self, identity: PromotionIdentity) -> bool: - return self.is_ready and self.verify_hash() and identity == self.identity + return ( + self.is_ready + and self.verify_hash() + and identity == self.identity + and _valid_root_range(self.evidence_root, self.ledger_records) + ) def to_dict(self) -> dict[str, Any]: return { @@ -105,6 +114,8 @@ def to_dict(self) -> dict[str, Any]: "blocking_reasons": list(self.blocking_reasons), "is_ready": self.is_ready, "receipt_hash": self.receipt_hash, + "evidence_root": self.evidence_root, + "ledger_records": self.ledger_records, } @classmethod @@ -121,6 +132,8 @@ def from_dict(cls, payload: dict[str, Any]) -> PromotionReceipt: blocking_reasons=tuple(payload["blocking_reasons"]), is_ready=bool(payload["is_ready"]), receipt_hash=str(payload["receipt_hash"]), + evidence_root=str(payload.get("evidence_root", "")), + ledger_records=int(payload.get("ledger_records", 0)), ) @@ -147,6 +160,9 @@ def evaluate_promotion( criteria: PromotionCriteria, *, identity: PromotionIdentity, + evidence_root: str | None = None, + ledger_records: int = 0, + ledger_path: str | Path | None = None, ) -> PromotionReceipt: """Create a self-verifying receipt for the conservative default evidence gate.""" @@ -177,6 +193,8 @@ def evaluate_promotion( reasons.append("OUTCOME_UNOBSERVABLE") if summary.unknown_enforceable_outcomes: reasons.append("UNKNOWN_ENFORCEABLE_OUTCOMES") + if not _verified_root_range(evidence_root, ledger_records, ledger_path): + reasons.append("EVIDENCE_ROOT_UNVERIFIED") provisional = PromotionReceipt( schema_version=1, @@ -188,6 +206,8 @@ def evaluate_promotion( blocking_reasons=tuple(reasons), is_ready=not reasons, receipt_hash="", + evidence_root=evidence_root if isinstance(evidence_root, str) else "", + ledger_records=ledger_records if isinstance(ledger_records, int) else 0, ) return replace(provisional, receipt_hash=_hash(provisional._hash_payload())) @@ -243,9 +263,44 @@ def read_promotion_receipt( return PromotionReceipt.from_dict(payload) -def activate_enforcement(data_root: str | Path, receipt: PromotionReceipt) -> Path: +def _valid_evidence_anchor(receipt: PromotionReceipt, ledger_path: str | Path | None) -> bool: + return _verified_root_range(receipt.evidence_root, receipt.ledger_records, ledger_path) + + +def _verified_root_range( + root: object, + records: object, + ledger_path: str | Path | None, +) -> bool: + if not _valid_root_range(root, records) or ledger_path is None: + return False + assert isinstance(root, str) + assert isinstance(records, int) and not isinstance(records, bool) + report = GovernanceLedger(ledger_path).verify_prefix(records, expected_root=root) + return report.valid and report.root_hash == root + + +def _valid_root_range(root: object, records: object) -> bool: + return bool( + isinstance(root, str) + and len(root) == 64 + and all(character in "0123456789abcdef" for character in root) + and not isinstance(records, bool) + and isinstance(records, int) + and records >= 1 + ) + + +def activate_enforcement( + data_root: str | Path, + receipt: PromotionReceipt, + *, + ledger_path: str | Path | None = None, +) -> Path: if not receipt.is_ready or not receipt.verify_hash(): raise ValueError("a ready, hash-valid receipt is required for enforcement") + if not _valid_evidence_anchor(receipt, ledger_path): + raise ValueError("a verified v3 evidence root is required for enforcement") stored = read_promotion_receipt(data_root, receipt.identity.repository_hash) if stored != receipt: raise ValueError("promotion receipt must be persisted before enforcement") @@ -258,6 +313,8 @@ def activate_enforcement(data_root: str | Path, receipt: PromotionReceipt) -> Pa "reason": "EARNED_ENFORCEMENT_PROMOTED", "receipt_hash": receipt.receipt_hash, "identity": asdict(receipt.identity), + "evidence_root": receipt.evidence_root, + "ledger_records": receipt.ledger_records, }, ) return path @@ -282,6 +339,7 @@ def enforcement_is_active( *, identity: PromotionIdentity, summary: CoverageSummary | None = None, + ledger_path: str | Path | None = None, ) -> bool: state_path = _state_path(data_root, identity.repository_hash) if not state_path.exists(): @@ -295,6 +353,14 @@ def enforcement_is_active( receipt is None or state.get("receipt_hash") != receipt.receipt_hash or not receipt.valid_for(identity) + or not _valid_evidence_anchor( + receipt, + ledger_path + or Path(data_root).resolve() + / "evidence" + / identity.repository_hash + / "governance-v3.jsonl", + ) ): demote_enforcement( data_root, @@ -312,7 +378,6 @@ def enforcement_is_active( coverage_ratio < receipt.criteria.minimum_coverage_ratio or summary.false_stops > receipt.summary.false_stops or summary.integration_failures > 0 - or summary.pending_actions > 0 or summary.unknown_enforceable_outcomes > 0 or summary.reviewed_candidates < summary.intervention_candidates or _p95(summary.decision_latencies_ms) > receipt.criteria.maximum_p95_latency_ms diff --git a/src/marginal/integrations/codex/runtime.py b/src/marginal/integrations/codex/runtime.py index a1770e5..5523270 100644 --- a/src/marginal/integrations/codex/runtime.py +++ b/src/marginal/integrations/codex/runtime.py @@ -11,9 +11,12 @@ from marginal.controls import ActionOutcomeStatus, NoProgressDetector, NoProgressSignal from marginal.models import Cost, Decision from marginal.protocol import AgentAction, AgentDecision +from marginal.reason_codes import ReasonCode from marginal.runtime import UniversalRuntime -from .events import PostToolUseEvent, PreToolUseEvent +from .autopilot import AutopilotController +from .events import PostToolUseEvent, PreToolUseEvent, UserPromptSubmitEvent +from .intent import UserIntent, is_control_plane_action, normalize_user_prompt from .normalization import normalize_pre_tool_use from .outcomes import classify_tool_outcome, completion_evidence_hash from .state import workspace_state_hash @@ -39,6 +42,8 @@ def __init__( workspace: str | Path, detector: NoProgressDetector | None = None, enforcement_enabled: Callable[[], bool] | None = None, + plugin_root: Path | None = None, + autopilot: AutopilotController | None = None, ) -> None: if not isinstance(runtime, UniversalRuntime): raise TypeError("runtime must be a UniversalRuntime") @@ -47,6 +52,8 @@ def __init__( workspace_state_hash(self.workspace) self.detector = detector or NoProgressDetector() self._enforcement_enabled = enforcement_enabled or (lambda: False) + self._plugin_root = Path(plugin_root) if plugin_root is not None else None + self._autopilot = autopilot self._pending: dict[str, _PendingAction] = {} self._evidence_by_semantic_key: dict[str, str] = {} self._last_signal: NoProgressSignal | None = None @@ -55,6 +62,7 @@ def __init__( self._failed = 0 self._unknown = 0 self._enforced_denials = 0 + self._user_intent = UserIntent() self._closed = False @property @@ -65,12 +73,39 @@ def last_no_progress_signal(self) -> NoProgressSignal | None: def last_action_evidence(self) -> dict[str, str] | None: return dict(self._last_action_evidence) if self._last_action_evidence else None + @property + def user_intent(self) -> UserIntent: + """Return prompt-derived state without retaining prompt material.""" + + return self._user_intent + + def user_prompt_submit(self, event: UserPromptSubmitEvent) -> None: + self._ensure_open() + self._validate_session(event.session_id) + self._user_intent = normalize_user_prompt(event.prompt) + def pre_tool_use(self, event: PreToolUseEvent) -> AgentDecision: self._ensure_open() self._validate_session(event.session_id) if event.tool_use_id in self._pending: raise CodexIntegrationError(f"tool identity is already pending: {event.tool_use_id}") state_hash = workspace_state_hash(self.workspace) + if self._plugin_root is not None and is_control_plane_action(event, self._plugin_root): + self._last_signal = None + self._last_action_evidence = self._safe_bypass_evidence(event, state_hash) + return AgentDecision.from_core( + event.tool_use_id, + Decision( + allowed=True, + reason="Trusted MARGINAL control-plane action bypasses workload governance", + recommended=False, + recommendation_reason="", + reason_code=ReasonCode.CONTROL_PLANE_BYPASS.value, + recommendation_reason_code=ReasonCode.CONTROL_PLANE_BYPASS.value, + mode="shadow", + confidence=1.0, + ), + ) action = normalize_pre_tool_use(event, state_hash=state_hash) semantic_key = str(action.metadata["semantic_key"]) evidence_hash = self._evidence_by_semantic_key.get(semantic_key, "") @@ -86,7 +121,33 @@ def pre_tool_use(self, event: PreToolUseEvent) -> AgentDecision: state_hash, evidence_hash, ) - if self._last_signal.enforcement_eligible and self._is_enforcement_enabled(): + if self._autopilot is not None and self._autopilot.consent_granted: + autopilot_decision = self._autopilot.pre_action( + action_id=event.tool_use_id, + workload_key=semantic_key, + eligible_family=self._is_autopilot_eligible(event, action), + state_hash=state_hash, + evidence_hash=evidence_hash, + intent=self._user_intent, + ) + if not autopilot_decision.allowed: + self._enforced_denials += 1 + return AgentDecision.from_core( + event.tool_use_id, + Decision( + allowed=False, + reason=( + "Repeated proven-success local action produced no new state or evidence" + ), + recommended=False, + recommendation_reason="", + reason_code=autopilot_decision.reason_code, + recommendation_reason_code=autopilot_decision.reason_code, + mode="enforce", + confidence=1.0, + ), + ) + elif self._last_signal.enforcement_eligible and self._is_enforcement_enabled(): self._enforced_denials += 1 return AgentDecision.from_core( event.tool_use_id, @@ -142,6 +203,13 @@ def post_tool_use(self, event: PostToolUseEvent) -> ActionOutcomeStatus: self._pending.pop(event.tool_use_id) self.detector.observe(semantic_key, post_state_hash, evidence_hash, outcome) + if self._autopilot is not None and self._autopilot.consent_granted: + self._autopilot.settle_action( + event.tool_use_id, + outcome=outcome, + state_hash=post_state_hash, + evidence_hash=evidence_hash, + ) if evidence_hash: self._evidence_by_semantic_key[semantic_key] = evidence_hash return outcome @@ -154,7 +222,7 @@ def action_evidence(self, action_id: str) -> dict[str, str] | None: return self._safe_action_evidence(pending.action) if pending is not None else None def summary(self) -> dict[str, int]: - return { + summary = { "successful_observations": self._successful, "failed_observations": self._failed, "unknown_observations": self._unknown, @@ -162,6 +230,9 @@ def summary(self) -> dict[str, int]: "pending_actions": len(self._pending), "enforced_denials": self._enforced_denials, } + if self._autopilot is not None: + summary.update(self._autopilot.summary()) + return summary def close(self) -> None: if self._closed: @@ -172,6 +243,14 @@ def close(self) -> None: reason="Codex session ended before the action outcome was observable", ) self._unknown += 1 + if self._autopilot is not None and self._autopilot.consent_granted: + pending = self._pending[action_id] + self._autopilot.settle_action( + action_id, + outcome=ActionOutcomeStatus.UNKNOWN, + state_hash=pending.action.state_hash, + evidence_hash="", + ) self._pending.pop(action_id) self._closed = True @@ -186,6 +265,23 @@ def _is_enforcement_enabled(self) -> bool: return False return enabled if isinstance(enabled, bool) else False + def _is_autopilot_eligible(self, event: PreToolUseEvent, action: AgentAction) -> bool: + """Accept only a direct constrained local-read adapter family for L3.""" + + del action + if event.tool_name.casefold() not in {"read", "read_file"}: + return False + if set(event.tool_input) not in ({"path"}, {"file_path"}): + return False + raw_path = event.tool_input.get("path", event.tool_input.get("file_path")) + if not isinstance(raw_path, str) or not raw_path or not Path(raw_path).is_absolute(): + return False + try: + Path(raw_path).resolve(strict=False).relative_to(self.workspace) + except ValueError: + return False + return True + @staticmethod def _safe_action_evidence(action: AgentAction) -> dict[str, str]: return { @@ -195,6 +291,15 @@ def _safe_action_evidence(action: AgentAction) -> dict[str, str]: "evidence_hash": str(action.metadata.get("evidence_hash", "")), } + @staticmethod + def _safe_bypass_evidence(event: PreToolUseEvent, state_hash: str) -> dict[str, str]: + return { + "action_hash": hashlib.sha256(event.tool_use_id.encode("utf-8")).hexdigest(), + "semantic_key": "", + "state_hash": state_hash, + "evidence_hash": "", + } + def _validate_session(self, session_id: str) -> None: if session_id != self.runtime.session_id: raise CodexIntegrationError("hook session identity does not match runtime identity") diff --git a/src/marginal/integrations/codex/service.py b/src/marginal/integrations/codex/service.py index 42c798f..daa7514 100644 --- a/src/marginal/integrations/codex/service.py +++ b/src/marginal/integrations/codex/service.py @@ -17,17 +17,21 @@ from marginal import BudgetLimits, Treasury from marginal.protocol import AgentCapabilities +from marginal.reason_codes import ReasonCode from marginal.runtime import UniversalRuntime +from .autopilot import AutopilotController from .events import ( PostToolUseEvent, PreToolUseEvent, SessionEvent, + UserPromptSubmitEvent, build_pre_tool_output, parse_hook_event, ) -from .evidence import EvidenceStore, summarize_evidence +from .evidence import EvidenceStore, summarize_verified_evidence from .identity import current_promotion_identity, repository_identity_hash +from .installer import autopilot_consent_configured from .promotion import PromotionIdentity, demote_enforcement, enforcement_is_active from .runtime import CodexSessionRuntime from .transport import ConnectionInfo, SessionServer, connection_filename, request_session @@ -74,6 +78,9 @@ def handle(operation: str, payload: dict[str, Any]) -> dict[str, Any] | None: threading.Timer(0.05, shutdown_event.set).start() return runtime.summary() event = parse_hook_event(payload) + if operation == "prompt" and isinstance(event, UserPromptSubmitEvent): + runtime.user_prompt_submit(event) + return None if operation == "pre" and isinstance(event, PreToolUseEvent): started = time.perf_counter_ns() decision = runtime.pre_tool_use(event) @@ -88,12 +95,15 @@ def handle(operation: str, payload: dict[str, Any]) -> dict[str, Any] | None: **action_evidence, "reason_code": decision.reason_code, "latency_ms": latency_ms, - "covered": True, - "coverable": True, + "covered": decision.reason_code != ReasonCode.CONTROL_PLANE_BYPASS.value, + "coverable": decision.reason_code != ReasonCode.CONTROL_PLANE_BYPASS.value, "recommended_stop": bool(signal and signal.should_recommend_stop), "reviewed": False, "false_stop": False, - "pending": decision.allowed, + "pending": ( + decision.allowed + and decision.reason_code != ReasonCode.CONTROL_PLANE_BYPASS.value + ), } ) return build_pre_tool_output( @@ -131,10 +141,29 @@ def handle(operation: str, payload: dict[str, Any]) -> dict[str, Any] | None: return handle +def _installed_plugin_root() -> Path | None: + """Derive the trusted installation root from the running immutable zipapp path.""" + + executable = Path(sys.argv[0]) + if executable.name != "marginal_runtime.pyz": + return None + try: + root = executable.resolve(strict=True).parent.parent + except OSError: + return None + return root if (root / ".codex-plugin" / "plugin.json").is_file() else None + + def _session_hash(session_id: str) -> str: return hashlib.sha256(session_id.encode("utf-8")).hexdigest() +def _identity_fingerprint(identity: PromotionIdentity) -> str: + return hashlib.sha256( + json.dumps(asdict(identity), sort_keys=True, separators=(",", ":")).encode("utf-8") + ).hexdigest() + + def _evidence_store(data_root: Path, repository_hash: str) -> EvidenceStore: return EvidenceStore(data_root / "evidence" / repository_hash) @@ -157,7 +186,6 @@ def _bootstrap_event_payload(event: SessionEvent) -> dict[str, Any]: "hook_event_name": event.hook_event_name, "model": event.model, "permission_mode": event.permission_mode, - "source": event.source, } @@ -262,7 +290,16 @@ def _serve_bootstrap(path: Path) -> int: enforcement_enabled=lambda: enforcement_is_active( data_root, identity=identity, - summary=summarize_evidence(evidence_store.read_all()), + summary=summarize_verified_evidence(evidence_store)[0], + ledger_path=evidence_store.governance_ledger_path, + ), + plugin_root=_installed_plugin_root(), + autopilot=AutopilotController( + data_root, + repository_hash=identity.repository_hash, + evidence=evidence_store, + identity_fingerprint=_identity_fingerprint(identity), + user_consent=autopilot_consent_configured(data_root), ), ) shutdown_event = threading.Event() @@ -322,7 +359,16 @@ def start_session_service( enforcement_enabled=lambda: enforcement_is_active( root, identity=identity, - summary=summarize_evidence(evidence_store.read_all()), + summary=summarize_verified_evidence(evidence_store)[0], + ledger_path=evidence_store.governance_ledger_path, + ), + plugin_root=_installed_plugin_root(), + autopilot=AutopilotController( + root, + repository_hash=identity.repository_hash, + evidence=evidence_store, + identity_fingerprint=_identity_fingerprint(identity), + user_consent=autopilot_consent_configured(root), ), ) server = SessionServer( @@ -407,6 +453,11 @@ def _fail_open_for_workspace(data_root: Path, cwd: str, reason: str) -> None: } ) store.start_new_window(reason_code=reason) + AutopilotController( + data_root, + repository_hash=repository_hash, + evidence=store, + ).revoke(reason) except OSError: pass with suppress(OSError): @@ -421,7 +472,7 @@ def run_hook(payload: dict[str, Any], *, data_root: str | Path) -> HookResult: """Execute one hook. Integration faults always fail open and demote enforcement.""" root = Path(data_root).resolve() - event: SessionEvent | PreToolUseEvent | PostToolUseEvent | None = None + event: SessionEvent | PreToolUseEvent | PostToolUseEvent | UserPromptSubmitEvent | None = None try: event = parse_hook_event(payload) if isinstance(event, SessionEvent): @@ -436,7 +487,13 @@ def run_hook(payload: dict[str, Any], *, data_root: str | Path) -> HookResult: _fail_open_for_workspace(root, event.cwd, "SERVICE_UNAVAILABLE") return HookResult(exit_code=0, warning_code="SERVICE_UNAVAILABLE") connection = ConnectionInfo.from_file(connection_path) - operation = "pre" if isinstance(event, PreToolUseEvent) else "post" + operation = ( + "pre" + if isinstance(event, PreToolUseEvent) + else "post" + if isinstance(event, PostToolUseEvent) + else "prompt" + ) response = request_session(connection, operation=operation, payload=payload) if response.get("ok") is not True: code = str(response.get("error_code", "SERVICE_ERROR")) diff --git a/src/marginal/ledger.py b/src/marginal/ledger.py index 002b903..30b2398 100644 --- a/src/marginal/ledger.py +++ b/src/marginal/ledger.py @@ -167,54 +167,54 @@ def read_decision_ledger(path: str | Path) -> list[dict[str, Any]]: """Load and structurally validate a MARGINAL v2 decision ledger.""" ledger_path = Path(path) + with _open_readonly_text(ledger_path) as stream: + return _read_decision_ledger_stream(stream) + + +def _read_decision_ledger_stream(stream: TextIO) -> list[dict[str, Any]]: + """Validate v2 ledger data from an already-secured text stream.""" + records: list[dict[str, Any]] = [] previous_sequence = 0 - with _open_readonly_text(ledger_path) as stream: - for line_number, raw in enumerate(stream, start=1): - if not raw.strip(): - continue + for line_number, raw in enumerate(stream, start=1): + if not raw.strip(): + continue + try: + record = json.loads(raw) + except json.JSONDecodeError as exc: + raise ValueError(f"invalid ledger JSON on line {line_number}") from exc + if not isinstance(record, dict): + raise ValueError(f"ledger record on line {line_number} must be an object") + if record.get("schema_version") != LEDGER_SCHEMA_VERSION: + raise ValueError(f"unsupported ledger schema on line {line_number}") + try: + privacy_profile = PrivacyProfile.parse(record.get("privacy_profile", "local_full")) + except (TypeError, ValueError) as exc: + raise ValueError(f"unsupported privacy profile on line {line_number}") from exc + if privacy_profile is PrivacyProfile.SAFE_TELEMETRY: try: - record = json.loads(raw) - except json.JSONDecodeError as exc: - raise ValueError(f"invalid ledger JSON on line {line_number}") from exc - if not isinstance(record, dict): - raise ValueError(f"ledger record on line {line_number} must be an object") - if record.get("schema_version") != LEDGER_SCHEMA_VERSION: - raise ValueError(f"unsupported ledger schema on line {line_number}") - try: - privacy_profile = PrivacyProfile.parse(record.get("privacy_profile", "local_full")) - except (TypeError, ValueError) as exc: - raise ValueError(f"unsupported privacy profile on line {line_number}") from exc - if privacy_profile is PrivacyProfile.SAFE_TELEMETRY: - try: - validate_safe_telemetry_record(record) - except ValueError as exc: - raise ValueError( - f"invalid safe telemetry on line {line_number}: {exc}" - ) from exc - for field_name in ("event_id", "timestamp", "run_id", "event"): - value = record.get(field_name) - if not isinstance(value, str) or not value.strip(): - raise ValueError( - f"ledger {field_name} missing or invalid on line {line_number}" - ) - for field_name in ("task_id", "trajectory_id", "engine", "model"): - value = record.get(field_name, "") - if not isinstance(value, str): - raise ValueError(f"ledger {field_name} must be a string on line {line_number}") - try: - datetime.fromisoformat(record["timestamp"]) + validate_safe_telemetry_record(record) except ValueError as exc: - raise ValueError(f"ledger timestamp invalid on line {line_number}") from exc - sequence = record.get("sequence") - if isinstance(sequence, bool) or not isinstance(sequence, int): - raise ValueError(f"invalid ledger sequence on line {line_number}") - if sequence <= previous_sequence: - raise ValueError( - f"ledger sequence is not strictly increasing on line {line_number}" - ) - previous_sequence = sequence - records.append(record) + raise ValueError(f"invalid safe telemetry on line {line_number}: {exc}") from exc + for field_name in ("event_id", "timestamp", "run_id", "event"): + value = record.get(field_name) + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"ledger {field_name} missing or invalid on line {line_number}") + for field_name in ("task_id", "trajectory_id", "engine", "model"): + value = record.get(field_name, "") + if not isinstance(value, str): + raise ValueError(f"ledger {field_name} must be a string on line {line_number}") + try: + datetime.fromisoformat(record["timestamp"]) + except ValueError as exc: + raise ValueError(f"ledger timestamp invalid on line {line_number}") from exc + sequence = record.get("sequence") + if isinstance(sequence, bool) or not isinstance(sequence, int): + raise ValueError(f"invalid ledger sequence on line {line_number}") + if sequence <= previous_sequence: + raise ValueError(f"ledger sequence is not strictly increasing on line {line_number}") + previous_sequence = sequence + records.append(record) if not records: raise ValueError("decision ledger is empty") return records diff --git a/src/marginal/policy.py b/src/marginal/policy.py index a628e15..ae30803 100644 --- a/src/marginal/policy.py +++ b/src/marginal/policy.py @@ -2,12 +2,11 @@ from __future__ import annotations -import hashlib -import json import math from dataclasses import asdict, dataclass from .budget import BudgetLedger +from .canonical import canonical_hash from .controls import DiminishingReturnDetector, DiminishingReturnSignal from .estimator import EstimatorIdentity, ValueEstimate, ValueEstimator from .models import Action, Decision @@ -93,13 +92,10 @@ def __init__( raise TypeError("version must be a string") if not version.strip(): raise ValueError("version must not be empty") - config_payload = json.dumps( - asdict(self.config), sort_keys=True, separators=(",", ":") - ).encode("utf-8") self.identity = PolicyIdentity( name=name, version=version, - config_hash=hashlib.sha256(config_payload).hexdigest(), + config_hash=canonical_hash(asdict(self.config)), ) self._executed_fingerprints: set[str] = set() diff --git a/src/marginal/reason_codes.py b/src/marginal/reason_codes.py new file mode 100644 index 0000000..f67621d --- /dev/null +++ b/src/marginal/reason_codes.py @@ -0,0 +1,22 @@ +"""Versioned, privacy-safe governance reason codes.""" + +from enum import Enum + +REASON_CODE_VERSION = "1.0" + + +class ReasonCode(str, Enum): + """Stable codes for evidence-based autonomy decisions.""" + + APPROVAL = "approval" + INSUFFICIENT_EVIDENCE = "insufficient_evidence" + INSUFFICIENT_TRUST = "insufficient_trust" + REPEATED_ACTION = "repeated_action" + NO_PROGRESS = "no_progress" + USER_REQUESTED_REPEAT = "user_requested_repeat" + CONTROL_PLANE_BYPASS = "control_plane_bypass" + POLICY_REVOKED = "policy_revoked" + DISTRIBUTION_SHIFT = "distribution_shift" + INTEGRITY_FAILURE = "integrity_failure" + RECOVERY = "recovery" + OUTCOME_UNKNOWN = "outcome_unknown" diff --git a/src/marginal/receipts.py b/src/marginal/receipts.py new file mode 100644 index 0000000..fd83670 --- /dev/null +++ b/src/marginal/receipts.py @@ -0,0 +1,322 @@ +"""Immutable, hash-bound governance decision receipts.""" + +from __future__ import annotations + +import hmac +import math +import re +from collections.abc import Mapping +from dataclasses import dataclass +from enum import Enum +from types import MappingProxyType +from typing import Any + +from .canonical import canonical_hash + +RECEIPT_SCHEMA_VERSION = "1.0" +PROGRESS_EVIDENCE_SCHEMA_VERSION = "1.0" + +_PRIVATE_PAYLOAD_TERMS = frozenset( + { + "auth", + "authorization", + "command", + "commands", + "credential", + "credentials", + "output", + "outputs", + "prompt", + "prompts", + "raw", + "response", + "responses", + "secret", + "secrets", + "source", + "sources", + "transcript", + "transcripts", + } +) +_SHA256_HEX_LENGTH = 64 +_LOWER_HEX_DIGITS = frozenset("0123456789abcdef") + + +class ProgressLevel(str, Enum): + """The strongest kind of evidence observed for a unit of work.""" + + ACTIVITY = "activity" + INFORMATION = "information" + PROGRESS = "progress" + VERIFIED_PROGRESS = "verified_progress" + + +def _validate_confidence(value: float, name: str = "confidence") -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number") + normalized = float(value) + if not math.isfinite(normalized) or not 0.0 <= normalized <= 1.0: + raise ValueError(f"{name} must be a finite value between 0 and 1") + return normalized + + +def _validate_non_negative_number(value: float, name: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number") + normalized = float(value) + if not math.isfinite(normalized) or normalized < 0: + raise ValueError(f"{name} must be a finite non-negative number") + return normalized + + +def _validate_non_negative_int(value: int, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an integer") + if value < 0: + raise ValueError(f"{name} must be non-negative") + return value + + +def _required_text(value: str, name: str) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a string") + if not value: + raise ValueError(f"{name} must not be empty") + return value + + +def _private_key(key: str) -> bool: + camel_separated = re.sub(r"(?<=[a-z0-9])(?=[A-Z])", "_", key) + terms = (term for term in re.split(r"[^a-z0-9]+", camel_separated.casefold()) if term) + return any(term in _PRIVATE_PAYLOAD_TERMS for term in terms) + + +def _valid_sha256_digest(value: str) -> bool: + return len(value) == _SHA256_HEX_LENGTH and all( + character in _LOWER_HEX_DIGITS for character in value + ) + + +def _freeze_json_value(value: Any, name: str) -> Any: + """Validate JSON-only evidence values without relying on ``repr`` coercion.""" + + if value is None or isinstance(value, (str, bool)): + return value + if isinstance(value, int) and not isinstance(value, bool): + return value + if isinstance(value, float): + if not math.isfinite(value): + raise ValueError(f"{name} must not contain non-finite numbers") + return value + if isinstance(value, list): + return tuple(_freeze_json_value(item, name) for item in value) + if isinstance(value, Mapping): + frozen: dict[str, Any] = {} + for key, item in value.items(): + if not isinstance(key, str): + raise TypeError(f"{name} mapping keys must be strings") + if _private_key(key): + raise ValueError(f"{name} must not contain raw private payload field {key!r}") + frozen[key] = _freeze_json_value(item, name) + return MappingProxyType(frozen) + raise TypeError(f"{name} must contain only canonical JSON values") + + +def _freeze_mapping(value: Mapping[str, Any], name: str) -> Mapping[str, Any]: + if not isinstance(value, Mapping): + raise TypeError(f"{name} must be a mapping") + frozen = _freeze_json_value(value, name) + assert isinstance(frozen, Mapping) + return frozen + + +def _json_value(value: Any) -> Any: + """Convert frozen attestation data back to ordinary JSON-compatible containers.""" + + if isinstance(value, Mapping): + return {key: _json_value(item) for key, item in value.items()} + if isinstance(value, tuple): + return [_json_value(item) for item in value] + return value + + +@dataclass(frozen=True, slots=True) +class ProgressEvidence: + """Derived evidence that keeps activity and verified progress distinct.""" + + schema_version: str + level: ProgressLevel + state_hash: str + evidence_hash: str + confidence: float + verifier: str | None + + def __post_init__(self) -> None: + if self.schema_version != PROGRESS_EVIDENCE_SCHEMA_VERSION: + raise ValueError("unsupported progress evidence schema version") + if not isinstance(self.level, ProgressLevel): + raise TypeError("level must be ProgressLevel") + _required_text(self.state_hash, "state_hash") + _required_text(self.evidence_hash, "evidence_hash") + object.__setattr__(self, "confidence", _validate_confidence(self.confidence)) + if self.verifier is not None: + _required_text(self.verifier, "verifier") + + def payload(self) -> dict[str, object]: + """Return the schema-shaped, privacy-safe progress payload.""" + + return { + "schema_version": self.schema_version, + "level": self.level.value, + "state_hash": self.state_hash, + "evidence_hash": self.evidence_hash, + "confidence": self.confidence, + "verifier": self.verifier, + } + + +@dataclass(frozen=True, slots=True) +class GovernanceCost: + """Measured governance overhead, with ``None`` reserved for unavailable measurements.""" + + wall_clock_ms: float + cpu_ms: float | None + memory_peak_bytes: int | None + storage_bytes: int + tokens: int + model_calls: int + additional_tool_calls: int + + def __post_init__(self) -> None: + object.__setattr__( + self, + "wall_clock_ms", + _validate_non_negative_number(self.wall_clock_ms, "wall_clock_ms"), + ) + if self.cpu_ms is not None: + object.__setattr__(self, "cpu_ms", _validate_non_negative_number(self.cpu_ms, "cpu_ms")) + if self.memory_peak_bytes is not None: + _validate_non_negative_int(self.memory_peak_bytes, "memory_peak_bytes") + for name in ("storage_bytes", "tokens", "model_calls", "additional_tool_calls"): + _validate_non_negative_int(getattr(self, name), name) + + def payload(self) -> dict[str, float | int | None]: + """Return explicit measurements suitable for a receipt payload.""" + + return { + "wall_clock_ms": self.wall_clock_ms, + "cpu_ms": self.cpu_ms, + "memory_peak_bytes": self.memory_peak_bytes, + "storage_bytes": self.storage_bytes, + "tokens": self.tokens, + "model_calls": self.model_calls, + "additional_tool_calls": self.additional_tool_calls, + } + + +@dataclass(frozen=True, slots=True) +class DecisionReceipt: + """A canonical, immutable attestation of one governance decision.""" + + schema_version: str + decision_id: str + timestamp: str + context: Mapping[str, str] + decision: str + reason_code: str + state_hash: str | None + evidence_hash: str | None + trajectory_hash: str | None + policy_hash: str + decision_hash: str + confidence: float + expected_utility: Mapping[str, Any] | None + estimated_cost: Mapping[str, Any] | None + enforcement_level: str + trust_snapshot: Mapping[str, Any] + governance_cost: GovernanceCost + + def __post_init__(self) -> None: + if self.schema_version != RECEIPT_SCHEMA_VERSION: + raise ValueError("unsupported decision receipt schema version") + for name in ( + "decision_id", + "timestamp", + "decision", + "reason_code", + "policy_hash", + "enforcement_level", + ): + _required_text(getattr(self, name), name) + if not isinstance(self.decision_hash, str): + raise TypeError("decision_hash must be a string") + for name in ("state_hash", "evidence_hash", "trajectory_hash"): + value = getattr(self, name) + if value is not None: + _required_text(value, name) + if not isinstance(self.context, Mapping): + raise TypeError("context must be a mapping") + for key, value in self.context.items(): + _required_text(key, "context key") + if _private_key(key): + raise ValueError(f"context must not contain raw private payload field {key!r}") + _required_text(value, f"context[{key!r}]") + object.__setattr__(self, "context", MappingProxyType(dict(self.context))) + object.__setattr__(self, "confidence", _validate_confidence(self.confidence)) + for name in ("expected_utility", "estimated_cost", "trust_snapshot"): + value = getattr(self, name) + if value is None: + if name == "trust_snapshot": + raise TypeError("trust_snapshot must be a mapping") + continue + object.__setattr__(self, name, _freeze_mapping(value, name)) + if not isinstance(self.governance_cost, GovernanceCost): + raise TypeError("governance_cost must be GovernanceCost") + + +def canonical_decision_payload(receipt: DecisionReceipt) -> dict[str, Any]: + """Return the exact receipt fields committed by ``decision_hash``.""" + + if not isinstance(receipt, DecisionReceipt): + raise TypeError("receipt must be DecisionReceipt") + return { + "schema_version": receipt.schema_version, + "decision_id": receipt.decision_id, + "timestamp": receipt.timestamp, + "context": _json_value(receipt.context), + "decision": receipt.decision, + "reason_code": receipt.reason_code, + "state_hash": receipt.state_hash, + "evidence_hash": receipt.evidence_hash, + "trajectory_hash": receipt.trajectory_hash, + "policy_hash": receipt.policy_hash, + "confidence": receipt.confidence, + "expected_utility": _json_value(receipt.expected_utility), + "estimated_cost": _json_value(receipt.estimated_cost), + "enforcement_level": receipt.enforcement_level, + "trust_snapshot": _json_value(receipt.trust_snapshot), + "governance_cost": receipt.governance_cost.payload(), + } + + +def decision_receipt_hash(receipt: DecisionReceipt) -> str: + """Hash the canonical payload without accepting arbitrary object representations.""" + + return canonical_hash(canonical_decision_payload(receipt)) + + +def receipt_payload(receipt: DecisionReceipt) -> dict[str, Any]: + """Return the complete, schema-shaped receipt including its binding hash.""" + + payload = canonical_decision_payload(receipt) + payload["decision_hash"] = receipt.decision_hash + return payload + + +def verify_decision_receipt(receipt: DecisionReceipt) -> bool: + """Return whether a receipt's current canonical payload matches its decision hash.""" + + if not isinstance(receipt, DecisionReceipt) or not _valid_sha256_digest(receipt.decision_hash): + return False + return hmac.compare_digest(receipt.decision_hash, decision_receipt_hash(receipt)) diff --git a/src/marginal/schemas/decision-receipt-v1.json b/src/marginal/schemas/decision-receipt-v1.json new file mode 100644 index 0000000..ebe948d --- /dev/null +++ b/src/marginal/schemas/decision-receipt-v1.json @@ -0,0 +1,69 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/decision-receipt-v1.json", + "title": "MARGINAL Decision Receipt v1", + "type": "object", + "required": [ + "schema_version", + "decision_id", + "timestamp", + "context", + "decision", + "reason_code", + "state_hash", + "evidence_hash", + "trajectory_hash", + "policy_hash", + "decision_hash", + "confidence", + "expected_utility", + "estimated_cost", + "enforcement_level", + "trust_snapshot", + "governance_cost" + ], + "properties": { + "schema_version": {"const": "1.0"}, + "decision_id": {"type": "string", "minLength": 1}, + "timestamp": {"type": "string", "format": "date-time"}, + "context": { + "type": "object", + "additionalProperties": {"type": "string", "minLength": 1} + }, + "decision": {"type": "string", "minLength": 1}, + "reason_code": {"type": "string", "minLength": 1}, + "state_hash": {"type": ["string", "null"]}, + "evidence_hash": {"type": ["string", "null"]}, + "trajectory_hash": {"type": ["string", "null"]}, + "policy_hash": {"type": "string", "minLength": 1}, + "decision_hash": {"type": "string", "minLength": 1}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + "expected_utility": {"type": ["object", "null"]}, + "estimated_cost": {"type": ["object", "null"]}, + "enforcement_level": {"type": "string", "minLength": 1}, + "trust_snapshot": {"type": "object"}, + "governance_cost": { + "type": "object", + "required": [ + "wall_clock_ms", + "cpu_ms", + "memory_peak_bytes", + "storage_bytes", + "tokens", + "model_calls", + "additional_tool_calls" + ], + "properties": { + "wall_clock_ms": {"type": "number", "minimum": 0}, + "cpu_ms": {"type": ["number", "null"], "minimum": 0}, + "memory_peak_bytes": {"type": ["integer", "null"], "minimum": 0}, + "storage_bytes": {"type": "integer", "minimum": 0}, + "tokens": {"type": "integer", "minimum": 0}, + "model_calls": {"type": "integer", "minimum": 0}, + "additional_tool_calls": {"type": "integer", "minimum": 0} + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/src/marginal/schemas/governance-ledger-v3.json b/src/marginal/schemas/governance-ledger-v3.json new file mode 100644 index 0000000..5edbe82 --- /dev/null +++ b/src/marginal/schemas/governance-ledger-v3.json @@ -0,0 +1,30 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/governance-ledger-v3.json", + "title": "MARGINAL Governance Ledger Record v3", + "type": "object", + "required": [ + "schema_version", + "sequence", + "timestamp", + "previous_hash", + "payload", + "payload_hash", + "record_hash" + ], + "properties": { + "schema_version": {"const": "3.0"}, + "sequence": {"type": "integer", "minimum": 1}, + "timestamp": {"type": "string", "format": "date-time"}, + "previous_hash": { + "oneOf": [ + {"type": "null"}, + {"type": "string", "pattern": "^[0-9a-f]{64}$"} + ] + }, + "payload": {"type": "object"}, + "payload_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "record_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + }, + "additionalProperties": false +} diff --git a/src/marginal/schemas/progress-evidence-v1.json b/src/marginal/schemas/progress-evidence-v1.json new file mode 100644 index 0000000..3effc9c --- /dev/null +++ b/src/marginal/schemas/progress-evidence-v1.json @@ -0,0 +1,25 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/progress-evidence-v1.json", + "title": "MARGINAL Progress Evidence v1", + "type": "object", + "required": [ + "schema_version", + "level", + "state_hash", + "evidence_hash", + "confidence", + "verifier" + ], + "properties": { + "schema_version": {"const": "1.0"}, + "level": { + "enum": ["activity", "information", "progress", "verified_progress"] + }, + "state_hash": {"type": "string", "minLength": 1}, + "evidence_hash": {"type": "string", "minLength": 1}, + "confidence": {"type": "number", "minimum": 0, "maximum": 1}, + "verifier": {"type": ["string", "null"], "minLength": 1} + }, + "additionalProperties": false +} diff --git a/src/marginal/schemas/trust-snapshot-v1.json b/src/marginal/schemas/trust-snapshot-v1.json new file mode 100644 index 0000000..e9d8e85 --- /dev/null +++ b/src/marginal/schemas/trust-snapshot-v1.json @@ -0,0 +1,33 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://github.com/SignalLayerLabs/Marginal/schemas/trust-snapshot-v1.json", + "title": "MARGINAL Trust Snapshot v1", + "type": "object", + "required": ["schema_version", "context", "components", "confidence_band", "eligible_authority", "current_authority", "authority", "blockers", "transition_receipt"], + "properties": { + "schema_version": {"const": "1.0"}, + "context": {"type": "object", "additionalProperties": {"type": "string", "minLength": 1}}, + "components": {"type": "object", "additionalProperties": {"type": ["number", "null"]}}, + "confidence_band": {"enum": ["low", "medium", "high"]}, + "eligible_authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "current_authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "authority": {"type": "integer", "minimum": 0, "maximum": 4}, + "blockers": {"type": "array", "items": {"type": "string", "minLength": 1}}, + "transition_receipt": { + "type": ["object", "null"], + "required": ["schema_version", "context", "previous", "current", "evidence_ledger_root", "ledger_records", "blockers", "receipt_hash"], + "properties": { + "schema_version": {"const": "1.0"}, + "context": {"type": "object", "additionalProperties": {"type": "string", "minLength": 1}}, + "previous": {"type": "integer", "minimum": 0, "maximum": 4}, + "current": {"type": "integer", "minimum": 0, "maximum": 4}, + "evidence_ledger_root": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "ledger_records": {"type": "integer", "minimum": 0}, + "blockers": {"type": "array", "items": {"type": "string", "minLength": 1}}, + "receipt_hash": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + }, + "additionalProperties": false + } + }, + "additionalProperties": false +} diff --git a/src/marginal/trust.py b/src/marginal/trust.py new file mode 100644 index 0000000..e0f2c4d --- /dev/null +++ b/src/marginal/trust.py @@ -0,0 +1,451 @@ +"""Contextual, explainable evidence evaluation for progressive authority.""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass, replace +from datetime import datetime, timedelta, timezone +from types import MappingProxyType + +from .authority import AuthorityLevel, AuthorityTransitionReceipt, transition_receipt_hash +from .governance_ledger import LedgerVerificationReport + +TRUST_SNAPSHOT_SCHEMA_VERSION = "1.0" +MIN_EVALUABLE_SAMPLES = 20 +MIN_COVERAGE = 0.95 +MAX_HARM_RATE = 0.05 +MAX_MEAN_REGRET = 0.10 +MAX_GOVERNANCE_TAX_RATIO = 0.10 +DEMOTION_MIN_EVALUABLE_SAMPLES = 10 +DEMOTION_MIN_COVERAGE = 0.90 +DEMOTION_MAX_HARM_RATE = 0.10 +DEMOTION_MAX_MEAN_REGRET = 0.20 +DEMOTION_MAX_GOVERNANCE_TAX_RATIO = 0.20 +INACTIVITY_WINDOW = timedelta(days=30) +CLOCK_SKEW_TOLERANCE = timedelta(minutes=5) +RECOVERY_CLEAN_EVALUATIONS = 1 + +_CRITICAL_SHIFT_REASONS = frozenset( + {"capability", "capability_drift", "integrity", "model", "policy"} +) +_SOFT_SHIFT_REASONS = frozenset({"large_repository", "repository", "task_class"}) +_HEX = frozenset("0123456789abcdef") + + +def _required_text(value: str, name: str) -> str: + if not isinstance(value, str): + raise TypeError(f"{name} must be a string") + if not value: + raise ValueError(f"{name} must not be empty") + return value + + +def _non_negative_int(value: int, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, int): + raise TypeError(f"{name} must be an integer") + if value < 0: + raise ValueError(f"{name} must be non-negative") + return value + + +def _ratio(value: float | None, name: str) -> float | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number or None") + result = float(value) + if result < 0.0 or result != result or result in (float("inf"), float("-inf")): + raise ValueError(f"{name} must be finite and non-negative") + return result + + +def _valid_root(value: str | None) -> bool: + return bool( + isinstance(value, str) + and len(value) == 64 + and all(character in _HEX for character in value) + ) + + +def _parse_timestamp(value: str | None) -> datetime | None: + if value is None: + return None + _required_text(value, "last_observed_at") + try: + parsed = datetime.fromisoformat(value) + except ValueError as exc: + raise ValueError("last_observed_at must be an ISO-8601 timestamp") from exc + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ValueError("last_observed_at must include an offset") + return parsed.astimezone(timezone.utc) + + +@dataclass(frozen=True, slots=True) +class TrustContext: + """The identity boundaries within which evidence is valid.""" + + repository: str + agent: str + model: str + task_class: str + policy_version: str + + def __post_init__(self) -> None: + for name in ("repository", "agent", "model", "task_class", "policy_version"): + _required_text(getattr(self, name), name) + + def payload(self) -> dict[str, str]: + return { + "repository": self.repository, + "agent": self.agent, + "model": self.model, + "task_class": self.task_class, + "policy_version": self.policy_version, + } + + +@dataclass(frozen=True, slots=True) +class TrustEvidence: + """Aggregated evidence with unavailable measurements explicitly represented by ``None``.""" + + observed: int + evaluable: int + covered: int + coverable: int + beneficial: int + neutral: int + harmful: int + indeterminate: int + governance_tax_ratio: float | None + mean_regret: float | None + integrity_valid: bool + last_observed_at: str | None + evidence_ledger_root: str | None = None + ledger_verification: LedgerVerificationReport | None = None + + def __post_init__(self) -> None: + for name in ( + "observed", + "evaluable", + "covered", + "coverable", + "beneficial", + "neutral", + "harmful", + "indeterminate", + ): + _non_negative_int(getattr(self, name), name) + if self.evaluable > self.observed: + raise ValueError("evaluable must not exceed observed") + if self.covered > self.coverable: + raise ValueError("covered must not exceed coverable") + outcomes = self.beneficial + self.neutral + self.harmful + self.indeterminate + if outcomes > self.evaluable: + raise ValueError("outcomes must not exceed evaluable") + object.__setattr__( + self, "governance_tax_ratio", _ratio(self.governance_tax_ratio, "governance_tax_ratio") + ) + object.__setattr__(self, "mean_regret", _ratio(self.mean_regret, "mean_regret")) + if not isinstance(self.integrity_valid, bool): + raise TypeError("integrity_valid must be a bool") + _parse_timestamp(self.last_observed_at) + if self.evidence_ledger_root is not None and not _valid_root(self.evidence_ledger_root): + raise ValueError("evidence_ledger_root must be a lowercase SHA-256 digest or None") + if self.ledger_verification is not None and not isinstance( + self.ledger_verification, LedgerVerificationReport + ): + raise TypeError("ledger_verification must be LedgerVerificationReport or None") + + +@dataclass(frozen=True, slots=True) +class TrustSnapshot: + """A schema-shaped, component-level explanation of one authority decision.""" + + schema_version: str + context: TrustContext + components: Mapping[str, float | int | None] + confidence_band: str + eligible_authority: AuthorityLevel + current_authority: AuthorityLevel + authority: AuthorityLevel + blockers: tuple[str, ...] + transition_receipt: AuthorityTransitionReceipt | None + + def __post_init__(self) -> None: + if self.schema_version != TRUST_SNAPSHOT_SCHEMA_VERSION: + raise ValueError("unsupported trust snapshot schema version") + if not isinstance(self.context, TrustContext): + raise TypeError("context must be TrustContext") + if not isinstance(self.components, Mapping): + raise TypeError("components must be a mapping") + normalized: dict[str, float | int | None] = {} + for key, value in self.components.items(): + _required_text(key, "component key") + if value is not None and ( + isinstance(value, bool) or not isinstance(value, (int, float)) + ): + raise TypeError("component values must be numeric or None") + normalized[key] = value + object.__setattr__(self, "components", MappingProxyType(normalized)) + if self.confidence_band not in {"low", "medium", "high"}: + raise ValueError("confidence_band must be low, medium, or high") + for name in ("eligible_authority", "current_authority", "authority"): + if not isinstance(getattr(self, name), AuthorityLevel): + raise TypeError(f"{name} must be AuthorityLevel") + if not isinstance(self.blockers, tuple) or not all( + isinstance(item, str) and item for item in self.blockers + ): + raise TypeError("blockers must be a tuple of non-empty strings") + if self.transition_receipt is not None and not isinstance( + self.transition_receipt, AuthorityTransitionReceipt + ): + raise TypeError("transition_receipt must be AuthorityTransitionReceipt or None") + + def payload(self) -> dict[str, object]: + return { + "schema_version": self.schema_version, + "context": self.context.payload(), + "components": dict(self.components), + "confidence_band": self.confidence_band, + "eligible_authority": int(self.eligible_authority), + "current_authority": int(self.current_authority), + "authority": int(self.authority), + "blockers": list(self.blockers), + "transition_receipt": ( + None + if self.transition_receipt is None + else self.transition_receipt.payload() + | {"receipt_hash": self.transition_receipt.receipt_hash} + ), + } + + +class TrustEngine: + """Evaluate evidence with conservative promotion and asymmetric demotion rules.""" + + def __init__(self, *, now: datetime | None = None) -> None: + if now is not None and (now.tzinfo is None or now.utcoffset() is None): + raise ValueError("now must include an offset") + self._now = now.astimezone(timezone.utc) if now is not None else None + self._recovery_remaining: dict[TrustContext, int] = {} + + def evaluate( + self, + context: TrustContext, + evidence: TrustEvidence, + current: AuthorityLevel, + *, + capabilities: int, + shift_reasons: tuple[str, ...] = (), + ) -> TrustSnapshot: + """Produce the next authority with visible evidence components and blockers.""" + + if not isinstance(context, TrustContext) or not isinstance(evidence, TrustEvidence): + raise TypeError("context and evidence must use their trust models") + if not isinstance(current, AuthorityLevel): + raise TypeError("current must be AuthorityLevel") + if isinstance(capabilities, bool) or not isinstance(capabilities, int): + raise TypeError("capabilities must be an integer authority ceiling") + if capabilities < int(AuthorityLevel.OBSERVE) or capabilities > int( + AuthorityLevel.COMPUTE_GOVERN + ): + raise ValueError("capabilities must be between OBSERVE and COMPUTE_GOVERN") + if not isinstance(shift_reasons, tuple) or not all( + isinstance(reason, str) and reason for reason in shift_reasons + ): + raise TypeError("shift_reasons must be a tuple of non-empty strings") + + ceiling = AuthorityLevel(capabilities) + now = self._now or datetime.now(timezone.utc) + components, blockers = self._components(evidence, now) + root_verified = self._root_is_verified(evidence) + if not root_verified: + blockers.append("unverified_evidence_ledger_root") + reasons = frozenset(shift_reasons) + critical = ( + not evidence.integrity_valid + or current > ceiling + or bool(reasons & _CRITICAL_SHIFT_REASONS) + ) + soft_shift = bool(reasons & _SOFT_SHIFT_REASONS) + if not evidence.integrity_valid: + blockers.append("integrity_failure") + if current > ceiling: + blockers.append("capability_drift") + if reasons & _CRITICAL_SHIFT_REASONS: + blockers.append("identity_shift") + if soft_shift: + blockers.append("distribution_shift") + + promotion_ok = not blockers + eligible = ceiling if promotion_ok else AuthorityLevel.OBSERVE + demotion_required = self._requires_soft_demotion(evidence, now, soft_shift) + if critical: + authority = AuthorityLevel.OBSERVE + elif demotion_required: + authority = AuthorityLevel(max(int(AuthorityLevel.OBSERVE), int(current) - 1)) + elif promotion_ok: + recovery_remaining = self._recovery_remaining.get(context, 0) + if recovery_remaining: + authority = current + blockers.append("recovery_hysteresis") + if recovery_remaining == 1: + del self._recovery_remaining[context] + else: + self._recovery_remaining[context] = recovery_remaining - 1 + else: + authority = AuthorityLevel(min(int(ceiling), int(current) + 1)) + if ( + authority == ceiling + and current == ceiling + and ceiling < AuthorityLevel.COMPUTE_GOVERN + ): + blockers.append("capability_ceiling") + else: + authority = current + + if critical: + self._recovery_remaining.pop(context, None) + elif authority < current: + self._recovery_remaining[context] = RECOVERY_CLEAN_EVALUATIONS + blockers = list(dict.fromkeys(blockers)) + receipt = self._receipt( + context, + current, + authority, + evidence.evidence_ledger_root, + evidence.ledger_verification, + blockers, + ) + confidence_band = ( + "high" + if promotion_ok and not soft_shift + else "medium" + if current >= authority + else "low" + ) + return TrustSnapshot( + schema_version=TRUST_SNAPSHOT_SCHEMA_VERSION, + context=context, + components=components, + confidence_band=confidence_band, + eligible_authority=eligible, + current_authority=current, + authority=authority, + blockers=tuple(blockers), + transition_receipt=receipt, + ) + + @staticmethod + def _components( + evidence: TrustEvidence, now: datetime + ) -> tuple[dict[str, float | int | None], list[str]]: + coverage = evidence.covered / evidence.coverable if evidence.coverable else None + harm_rate = evidence.harmful / evidence.evaluable if evidence.evaluable else None + outcomes = ( + evidence.beneficial + evidence.neutral + evidence.harmful + evidence.indeterminate + ) + observed_at = _parse_timestamp(evidence.last_observed_at) + age_hours = None if observed_at is None else (now - observed_at).total_seconds() / 3600 + components: dict[str, float | int | None] = { + "observed": evidence.observed, + "evaluable": evidence.evaluable, + "covered": evidence.covered, + "coverable": evidence.coverable, + "beneficial": evidence.beneficial, + "neutral": evidence.neutral, + "harmful": evidence.harmful, + "indeterminate": evidence.indeterminate, + "unclassified_outcomes": evidence.evaluable - outcomes, + "coverage": coverage, + "harm_rate": harm_rate, + "mean_regret": evidence.mean_regret, + "governance_tax_ratio": evidence.governance_tax_ratio, + "evidence_age_hours": age_hours, + } + blockers: list[str] = [] + if evidence.evaluable < MIN_EVALUABLE_SAMPLES: + blockers.append("minimum_evaluable_samples") + if outcomes != evidence.evaluable: + blockers.append("unclassified_outcomes") + if coverage is None or coverage < MIN_COVERAGE: + blockers.append("insufficient_coverage") + if harm_rate is None or harm_rate > MAX_HARM_RATE: + blockers.append("harm_rate_too_high") + if evidence.mean_regret is None or evidence.mean_regret > MAX_MEAN_REGRET: + blockers.append("mean_regret_too_high") + if ( + evidence.governance_tax_ratio is None + or evidence.governance_tax_ratio > MAX_GOVERNANCE_TAX_RATIO + ): + blockers.append("governance_tax_too_high") + if observed_at is None or now - observed_at > INACTIVITY_WINDOW: + blockers.append("inactivity") + if observed_at is not None and observed_at - now > CLOCK_SKEW_TOLERANCE: + blockers.append("future_observation") + return components, blockers + + @staticmethod + def _requires_soft_demotion(evidence: TrustEvidence, now: datetime, soft_shift: bool) -> bool: + coverage = evidence.covered / evidence.coverable if evidence.coverable else None + harm_rate = evidence.harmful / evidence.evaluable if evidence.evaluable else None + outcomes = ( + evidence.beneficial + evidence.neutral + evidence.harmful + evidence.indeterminate + ) + observed_at = _parse_timestamp(evidence.last_observed_at) + return bool( + soft_shift + or evidence.evaluable < DEMOTION_MIN_EVALUABLE_SAMPLES + or coverage is None + or coverage < DEMOTION_MIN_COVERAGE + or harm_rate is None + or harm_rate > DEMOTION_MAX_HARM_RATE + or evidence.mean_regret is None + or evidence.mean_regret > DEMOTION_MAX_MEAN_REGRET + or evidence.governance_tax_ratio is None + or evidence.governance_tax_ratio > DEMOTION_MAX_GOVERNANCE_TAX_RATIO + or outcomes != evidence.evaluable + or observed_at is None + or now - observed_at > INACTIVITY_WINDOW + or observed_at - now > CLOCK_SKEW_TOLERANCE + ) + + @staticmethod + def _root_is_verified(evidence: TrustEvidence) -> bool: + report = evidence.ledger_verification + return bool( + _valid_root(evidence.evidence_ledger_root) + and isinstance(report, LedgerVerificationReport) + and report.valid + and report.root_hash == evidence.evidence_ledger_root + ) + + @staticmethod + def _receipt( + context: TrustContext, + previous: AuthorityLevel, + current: AuthorityLevel, + root: str | None, + verification: LedgerVerificationReport | None, + blockers: list[str], + ) -> AuthorityTransitionReceipt | None: + if ( + previous == current + or not _valid_root(root) + or not isinstance(verification, LedgerVerificationReport) + or not verification.valid + or verification.root_hash != root + ): + return None + assert root is not None + unsigned = AuthorityTransitionReceipt( + schema_version="1.0", + context=context.payload(), + previous=previous, + current=current, + evidence_ledger_root=root, + ledger_verification=verification, + blockers=tuple(blockers), + receipt_hash="", + ) + return replace(unsigned, receipt_hash=transition_receipt_hash(unsigned)) diff --git a/src/marginal/utility.py b/src/marginal/utility.py new file mode 100644 index 0000000..b5c5e60 --- /dev/null +++ b/src/marginal/utility.py @@ -0,0 +1,168 @@ +"""Correctness-first structured utility and marginal-efficiency estimates.""" + +from __future__ import annotations + +import math +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Any + +from .receipts import GovernanceCost + + +def _finite_non_negative(value: float | int | None, name: str) -> float | None: + if value is None: + return None + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise TypeError(f"{name} must be a number or None") + normalized = float(value) + if not math.isfinite(normalized) or normalized < 0: + raise ValueError(f"{name} must be a finite non-negative number") + return normalized + + +def _confidence(value: float) -> float: + normalized = _finite_non_negative(value, "confidence") + assert normalized is not None + if normalized > 1.0: + raise ValueError("confidence must be between 0 and 1") + return normalized + + +def _compare_benefit(left: float | None, right: float | None) -> int: + if left is None: + return 0 if right is None else -1 + if right is None: + return 1 + return (left > right) - (left < right) + + +def _compare_cost(left: float | None, right: float | None) -> int: + if left is None: + return 0 if right is None else -1 + if right is None: + return 1 + return (right > left) - (right < left) + + +@dataclass(frozen=True, slots=True) +class UtilityVector: + """A lexicographic scorecard where correctness always dominates efficiency.""" + + verified_correctness: float | None + task_completion: float | None + safety_risk: float | None + latency_ms: float | None + tokens: int | None + monetary_cost: float | None + governance_overhead: float | None + + def __post_init__(self) -> None: + for name in ( + "verified_correctness", + "task_completion", + "safety_risk", + "latency_ms", + "monetary_cost", + "governance_overhead", + ): + object.__setattr__(self, name, _finite_non_negative(getattr(self, name), name)) + if self.tokens is not None: + if isinstance(self.tokens, bool) or not isinstance(self.tokens, int): + raise TypeError("tokens must be an integer or None") + if self.tokens < 0: + raise ValueError("tokens must be non-negative") + + def compare(self, other: UtilityVector) -> int: + """Compare utility vectors in the documented correctness-first order.""" + + if not isinstance(other, UtilityVector): + raise TypeError("other must be UtilityVector") + comparisons = ( + _compare_benefit(self.verified_correctness, other.verified_correctness), + _compare_benefit(self.task_completion, other.task_completion), + _compare_cost(self.safety_risk, other.safety_risk), + _compare_cost(self.latency_ms, other.latency_ms), + _compare_cost( + None if self.tokens is None else float(self.tokens), + None if other.tokens is None else float(other.tokens), + ), + _compare_cost(self.monetary_cost, other.monetary_cost), + _compare_cost(self.governance_overhead, other.governance_overhead), + ) + for comparison in comparisons: + if comparison: + return comparison + return 0 + + def payload(self) -> dict[str, float | int | None]: + """Return the scorecard without collapsing unavailable fields into zero.""" + + return { + "verified_correctness": self.verified_correctness, + "task_completion": self.task_completion, + "safety_risk": self.safety_risk, + "latency_ms": self.latency_ms, + "tokens": self.tokens, + "monetary_cost": self.monetary_cost, + "governance_overhead": self.governance_overhead, + } + + +@dataclass(frozen=True, slots=True) +class MarginalUtilityEstimate: + """An auditable utility estimate that withholds unjustified scalar efficiency claims.""" + + expected_utility: UtilityVector + estimated_cost: GovernanceCost + uncertainty: float + confidence: float + provenance: Mapping[str, str] + commensurable_cost: float | None = None + + def __post_init__(self) -> None: + if not isinstance(self.expected_utility, UtilityVector): + raise TypeError("expected_utility must be UtilityVector") + if not isinstance(self.estimated_cost, GovernanceCost): + raise TypeError("estimated_cost must be GovernanceCost") + uncertainty = _finite_non_negative(self.uncertainty, "uncertainty") + assert uncertainty is not None + object.__setattr__(self, "uncertainty", uncertainty) + object.__setattr__(self, "confidence", _confidence(self.confidence)) + if not isinstance(self.provenance, Mapping): + raise TypeError("provenance must be a mapping") + frozen_provenance: dict[str, str] = {} + for key, value in self.provenance.items(): + if not isinstance(key, str) or not key: + raise ValueError("provenance keys must be non-empty strings") + if not isinstance(value, str) or not value: + raise ValueError("provenance values must be non-empty strings") + frozen_provenance[key] = value + object.__setattr__(self, "provenance", MappingProxyType(frozen_provenance)) + commensurable_cost = _finite_non_negative(self.commensurable_cost, "commensurable_cost") + if commensurable_cost == 0.0: + raise ValueError("commensurable_cost must be positive when provided") + object.__setattr__(self, "commensurable_cost", commensurable_cost) + + def scalar_emu(self) -> float | None: + """Return a ratio only when verified utility and a comparable cost are both known.""" + + verified_utility = self.expected_utility.verified_correctness + if verified_utility is None or self.commensurable_cost is None: + return None + return verified_utility / self.commensurable_cost + + def scorecard(self) -> Mapping[str, Any]: + """Return the structured estimate, retaining all unavailable measurements as ``None``.""" + + return MappingProxyType( + { + "expected_utility": MappingProxyType(self.expected_utility.payload()), + "estimated_cost": MappingProxyType(self.estimated_cost.payload()), + "uncertainty": self.uncertainty, + "confidence": self.confidence, + "provenance": self.provenance, + "scalar_emu": self.scalar_emu(), + } + ) diff --git a/tests/integrations/codex/test_autopilot.py b/tests/integrations/codex/test_autopilot.py new file mode 100644 index 0000000..3cc3d36 --- /dev/null +++ b/tests/integrations/codex/test_autopilot.py @@ -0,0 +1,303 @@ +from __future__ import annotations + +from pathlib import Path + +from marginal.controls import ActionOutcomeStatus +from marginal.integrations.codex.autopilot import AutopilotController +from marginal.integrations.codex.evidence import EvidenceStore +from marginal.integrations.codex.intent import UserIntent + + +def _rooted_store(path: Path) -> EvidenceStore: + store = EvidenceStore(path) + store.append({"schema_version": 1, "event": "session_start", "session_hash": "session"}) + return store + + +def _controller(path: Path) -> tuple[AutopilotController, EvidenceStore]: + store = _rooted_store(path / "evidence") + controller = AutopilotController(path / "data", repository_hash="repository", evidence=store) + controller.grant_consent() + return controller, store + + +def test_consent_is_deferred_and_persists_without_token_estimates(tmp_path: Path) -> None: + store = _rooted_store(tmp_path / "evidence") + first = AutopilotController(tmp_path / "data", repository_hash="repository", evidence=store) + + assert first.consent_granted is False + assert first.summary() == {"avoided_actions": 0, "recoveries": 0, "pending_actions": 0} + + first.grant_consent() + restarted = AutopilotController(tmp_path / "data", repository_hash="repository", evidence=store) + + assert restarted.consent_granted is True + assert "tokens" not in restarted.summary() + + +def test_user_owned_install_consent_can_enable_autopilot_without_repository_config( + tmp_path: Path, +) -> None: + store = _rooted_store(tmp_path / "evidence") + + controller = AutopilotController( + tmp_path / "data", + repository_hash="repository", + evidence=store, + user_consent=True, + ) + + assert controller.consent_granted is True + + +def test_third_exact_safe_success_is_denied_only_with_a_verified_quick_receipt( + tmp_path: Path, +) -> None: + controller, store = _controller(tmp_path) + for action_id in ("one", "two"): + decision = controller.pre_action( + action_id=action_id, + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + assert decision.allowed is True + controller.settle_action( + action_id, + outcome=ActionOutcomeStatus.SUCCESS, + state_hash="state", + evidence_hash="evidence", + ) + store.append( + { + "schema_version": 1, + "event": "outcome", + "action_hash": action_id, + "outcome": "success", + } + ) + + denied = controller.pre_action( + action_id="three", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + + assert denied.allowed is False + assert denied.reason_code == "NO_PROGRESS_ENFORCED" + assert controller.summary()["avoided_actions"] == 1 + assert controller.quick_receipt is not None + assert controller.quick_receipt.evidence_root == store.verified_governance_root().root_hash + + +def test_repeat_intent_changed_evidence_and_uncovered_families_pass(tmp_path: Path) -> None: + controller, _store = _controller(tmp_path) + for action_id in ("one", "two"): + controller.pre_action( + action_id=action_id, + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.settle_action( + action_id, + outcome=ActionOutcomeStatus.SUCCESS, + state_hash="state", + evidence_hash="evidence", + ) + + assert controller.pre_action( + action_id="repeat", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(repeat_requested=True), + ).allowed + assert controller.pre_action( + action_id="changed", + workload_key="exact", + eligible_family=True, + state_hash="state-2", + evidence_hash="evidence", + intent=UserIntent(), + ).allowed + assert controller.pre_action( + action_id="shell", + workload_key="shell", + eligible_family=False, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ).allowed + + +def test_immediate_identical_retry_recovers_once_and_demotes(tmp_path: Path) -> None: + controller, _store = _controller(tmp_path) + for action_id in ("one", "two"): + controller.pre_action( + action_id=action_id, + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.settle_action( + action_id, + outcome=ActionOutcomeStatus.SUCCESS, + state_hash="state", + evidence_hash="evidence", + ) + assert not controller.pre_action( + action_id="three", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ).allowed + + recovery = controller.pre_action( + action_id="four", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + + assert recovery.allowed is True + assert recovery.reason_code == "RECOVERY" + assert controller.enforcement_active is False + assert controller.summary()["recoveries"] == 1 + + +def test_failures_and_unknowns_demote_but_unrelated_pending_workloads_do_not( + tmp_path: Path, +) -> None: + controller, _store = _controller(tmp_path) + controller.pre_action( + action_id="read", + workload_key="read", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.pre_action( + action_id="other", + workload_key="other", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.settle_action( + "read", outcome=ActionOutcomeStatus.SUCCESS, state_hash="state", evidence_hash="evidence" + ) + + assert controller.summary()["pending_actions"] == 1 + controller.settle_action( + "other", outcome=ActionOutcomeStatus.UNKNOWN, state_hash="state", evidence_hash="evidence" + ) + + assert controller.enforcement_active is False + + +def test_same_workload_pending_repeat_bypasses_a_denial_until_settled(tmp_path: Path) -> None: + controller, _store = _controller(tmp_path) + for action_id in ("one", "two"): + controller.pre_action( + action_id=action_id, + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.settle_action( + action_id, + outcome=ActionOutcomeStatus.SUCCESS, + state_hash="state", + evidence_hash="evidence", + ) + + pending = controller.pre_action( + action_id="requested", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(repeat_requested=True), + ) + concurrent = controller.pre_action( + action_id="concurrent", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + + assert pending.allowed is True + assert concurrent.allowed is True + assert concurrent.reason_code == "PENDING_WORKLOAD" + + +def test_failed_workload_clears_its_history_and_quick_receipt(tmp_path: Path) -> None: + controller, _store = _controller(tmp_path) + for action_id in ("one", "two"): + controller.pre_action( + action_id=action_id, + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ) + controller.settle_action( + action_id, + outcome=ActionOutcomeStatus.SUCCESS, + state_hash="state", + evidence_hash="evidence", + ) + assert not controller.pre_action( + action_id="denied", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ).allowed + controller.pre_action( + action_id="forced", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(force_run=True), + ) + controller.settle_action( + "forced", + outcome=ActionOutcomeStatus.FAILURE, + state_hash="state", + evidence_hash="evidence", + ) + + assert controller.quick_receipt is None + assert controller.pre_action( + action_id="again", + workload_key="exact", + eligible_family=True, + state_hash="state", + evidence_hash="evidence", + intent=UserIntent(), + ).allowed diff --git a/tests/integrations/codex/test_events.py b/tests/integrations/codex/test_events.py index 957304b..66aa6e0 100644 --- a/tests/integrations/codex/test_events.py +++ b/tests/integrations/codex/test_events.py @@ -6,6 +6,7 @@ PostToolUseEvent, PreToolUseEvent, SessionEvent, + UserPromptSubmitEvent, build_post_tool_output, build_pre_tool_output, parse_hook_event, @@ -69,9 +70,11 @@ def test_session_events_are_typed(name: str, extra: dict[str, str]) -> None: assert event.hook_event_name == name -def test_unknown_hook_event_is_rejected() -> None: - with pytest.raises(ValueError, match="unsupported Codex hook event"): - parse_hook_event(_common("UserPromptSubmit")) +def test_user_prompt_submit_is_typed_without_a_transcript_reference() -> None: + event = parse_hook_event({**_common("UserPromptSubmit"), "prompt": "run it again"}) + + assert isinstance(event, UserPromptSubmitEvent) + assert event.prompt == "run it again" def test_denial_uses_official_codex_shape() -> None: diff --git a/tests/integrations/codex/test_evidence.py b/tests/integrations/codex/test_evidence.py index f970145..f45bd9e 100644 --- a/tests/integrations/codex/test_evidence.py +++ b/tests/integrations/codex/test_evidence.py @@ -6,7 +6,11 @@ import pytest -from marginal.integrations.codex.evidence import EvidenceStore, summarize_evidence +from marginal.integrations.codex.evidence import ( + EvidenceStore, + summarize_evidence, + summarize_verified_evidence, +) def _record() -> dict[str, object]: @@ -29,7 +33,9 @@ def _record() -> dict[str, object]: } -@pytest.mark.parametrize("field", ["tool_input", "tool_response", "prompt", "command", "source"]) +@pytest.mark.parametrize( + "field", ["tool_input", "tool_response", "prompt", "prompt_hash", "command", "source"] +) def test_store_rejects_raw_payload_fields(tmp_path: Path, field: str) -> None: with pytest.raises(ValueError, match="forbidden evidence field"): EvidenceStore(tmp_path).append({**_record(), field: "secret"}) @@ -114,3 +120,30 @@ def test_new_window_preserves_audit_history_but_requires_fresh_evidence(tmp_path assert records[-1]["event"] == "window_start" assert summary.integration_failures == 0 assert summary.covered_actions == 0 + + +def test_evidence_records_are_anchored_to_a_verified_v3_prefix(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + store.append(_record()) + receipt_root = store.verified_governance_root() + store.append({**_record(), "action_hash": "second-action"}) + + assert receipt_root.valid is True + assert receipt_root.root_hash is not None + assert store.verifies_governance_prefix( + root_hash=receipt_root.root_hash, records=receipt_root.records + ) + + +def test_verified_summary_ignores_mutable_jsonl_tampering(tmp_path: Path) -> None: + store = EvidenceStore(tmp_path) + store.append(_record()) + store.path.write_text( + '{"schema_version":1,"event":"decision","covered":true,"coverable":true}\n', + encoding="utf-8", + ) + + summary, report = summarize_verified_evidence(store) + + assert report.valid is True + assert summary.covered_actions == 1 diff --git a/tests/integrations/codex/test_installer.py b/tests/integrations/codex/test_installer.py index 1cf06cd..e814eed 100644 --- a/tests/integrations/codex/test_installer.py +++ b/tests/integrations/codex/test_installer.py @@ -4,6 +4,7 @@ from marginal.integrations.codex.installer import ( CommandResult, + autopilot_consent_configured, inspect_codex, install, uninstall, @@ -92,3 +93,11 @@ def test_uninstall_uses_native_command() -> None: assert result.installed is False assert ["codex", "plugin", "remove", "marginal@marginal", "--json"] in runner.calls + + +def test_install_can_persist_explicit_user_autopilot_consent(tmp_path) -> None: + result = install(runner=RecordingRunner(), data_dir=tmp_path, autopilot_consent=True) + + assert result.installed is True + assert result.autopilot_consent is True + assert autopilot_consent_configured(tmp_path) is True diff --git a/tests/integrations/codex/test_intent.py b/tests/integrations/codex/test_intent.py new file mode 100644 index 0000000..9483a1d --- /dev/null +++ b/tests/integrations/codex/test_intent.py @@ -0,0 +1,141 @@ +from __future__ import annotations + +import os +from pathlib import Path + +import pytest + +from marginal.integrations.codex.events import PreToolUseEvent +from marginal.integrations.codex.intent import ( + UserIntent, + is_control_plane_action, + normalize_user_prompt, +) + + +@pytest.mark.parametrize( + ("prompt", "expected"), + [ + (" RUN\u3000IT AGAIN ", UserIntent(repeat_requested=True)), + ("Rifai l'azione", UserIntent(repeat_requested=True)), + ("esegui di nuovo", UserIntent(repeat_requested=True)), + ("Procedi comunque", UserIntent(force_run=True)), + ("Force the run", UserIntent(force_run=True)), + ("Esegui comunque", UserIntent(force_run=True)), + ("Metti in pausa MARGINAL", UserIntent(pause_marginal=True)), + ("Riattiva marginal", UserIntent(resume_marginal=True)), + ("Mostra lo stato di Marginal", UserIntent(status_requested=True)), + ], +) +def test_user_intent_normalizes_italian_english_and_unicode( + prompt: str, expected: UserIntent +) -> None: + assert normalize_user_prompt(prompt) == expected + + +@pytest.mark.parametrize( + "prompt", + [ + "Pause Marginal, then resume Marginal", + "Do not force run", + "Do not repeat the command", + "Don't pause Marginal", + "Don't proceed anyway", + "Do not execute anyway", + "Non ripetere il comando", + "Non sospendere Marginal", + "Non procedere comunque", + "Non eseguire comunque", + "Continue with the investigation", + ], +) +def test_user_intent_fails_open_for_ambiguous_or_negated_language(prompt: str) -> None: + assert normalize_user_prompt(prompt) == UserIntent() + + +def _event(command: str) -> PreToolUseEvent: + return PreToolUseEvent( + session_id="session-1", + cwd="/workspace", + hook_event_name="PreToolUse", + model="gpt-5.6-sol", + permission_mode="default", + turn_id="turn-1", + tool_name="Bash", + tool_use_id="call-1", + tool_input={"command": command}, + ) + + +def _trusted_plugin(root: Path) -> Path: + script = root / "scripts" / "marginal_control.py" + script.parent.mkdir(parents=True) + (root / ".codex-plugin").mkdir() + (root / ".codex-plugin" / "plugin.json").write_text("{}", encoding="utf-8") + script.write_text("#!/usr/bin/env python3\n", encoding="utf-8") + return script + + +def test_control_plane_accepts_only_exact_trusted_script_and_subcommand(tmp_path: Path) -> None: + trusted = tmp_path / "installed" / "marginal" + script = _trusted_plugin(trusted) + + assert is_control_plane_action( + _event(f"python3 {script} status --workspace /workspace --json"), trusted + ) + candidate = "a" * 64 + assert is_control_plane_action( + _event( + f'py -3 "{script}" review --workspace "/workspace with spaces" ' + f"--candidate {candidate} --verdict waste --json" + ), + trusted, + ) + assert not is_control_plane_action(_event(f"python3 {script} purge"), trusted) + + +@pytest.mark.parametrize( + "suffix", + [ + "status --data-dir /tmp/lookalike", + "status --workspace", + "status --unknown value", + "doctor --workspace /workspace", + "review --candidate abc", + "review --candidate abc --verdict unknown", + "promote positional-argument", + ], +) +def test_control_plane_rejects_arguments_outside_the_exact_command_contract( + tmp_path: Path, suffix: str +) -> None: + trusted = tmp_path / "installed" / "marginal" + script = _trusted_plugin(trusted) + + assert not is_control_plane_action(_event(f"python3 {script} {suffix}"), trusted) + + +def test_control_plane_rejects_lookalike_traversal_symlink_and_shell_injection( + tmp_path: Path, +) -> None: + trusted = tmp_path / "installed" / "marginal" + script = _trusted_plugin(trusted) + lookalike = _trusted_plugin(tmp_path / "repository" / "plugins" / "marginal") + + assert not is_control_plane_action(_event(f"python3 {lookalike} status"), trusted) + assert not is_control_plane_action( + _event(f"python3 {trusted}/scripts/../scripts/marginal_control.py status"), trusted + ) + assert not is_control_plane_action(_event(f"python3 {script} status; git status"), trusted) + assert not is_control_plane_action(_event(f"python3 {script} status && git status"), trusted) + assert not is_control_plane_action(_event(f"python3 {script} status $(id)"), trusted) + assert not is_control_plane_action(_event(f"python3 -I {script} status"), trusted) + assert not is_control_plane_action(_event(f"/usr/bin/env python3 {script} status"), trusted) + assert not is_control_plane_action(_event(f"/tmp/python3 {script} status"), trusted) + + symlink_target = tmp_path / "outside.py" + symlink_target.write_text("#!/usr/bin/env python3\n", encoding="utf-8") + script.unlink() + os.symlink(symlink_target, script) + + assert not is_control_plane_action(_event(f"python3 {script} status"), trusted) diff --git a/tests/integrations/codex/test_marketplace_smoke.py b/tests/integrations/codex/test_marketplace_smoke.py index c38aacb..04c8bdc 100644 --- a/tests/integrations/codex/test_marketplace_smoke.py +++ b/tests/integrations/codex/test_marketplace_smoke.py @@ -1,5 +1,6 @@ from __future__ import annotations +import json import shutil from pathlib import Path @@ -9,6 +10,12 @@ REPO = Path(__file__).resolve().parents[3] +def test_plugin_bundle_registers_user_prompt_submit_hook() -> None: + payload = json.loads((REPO / "plugins" / "marginal" / "hooks" / "hooks.json").read_text()) + + assert "UserPromptSubmit" in payload["hooks"] + + @pytest.mark.skipif(shutil.which("codex") is None, reason="Codex CLI is not installed") def test_marketplace_install_and_remove(tmp_path: Path) -> None: result = smoke_plugin( diff --git a/tests/integrations/codex/test_normalization.py b/tests/integrations/codex/test_normalization.py index 2c7ea88..2823bdf 100644 --- a/tests/integrations/codex/test_normalization.py +++ b/tests/integrations/codex/test_normalization.py @@ -5,7 +5,10 @@ import pytest from marginal.integrations.codex.events import PreToolUseEvent -from marginal.integrations.codex.normalization import normalize_pre_tool_use +from marginal.integrations.codex.normalization import ( + is_control_plane_action, + normalize_pre_tool_use, +) def _event(command: str, *, tool_name: str = "Bash") -> PreToolUseEvent: @@ -76,3 +79,14 @@ def test_non_json_tool_input_is_rejected() -> None: with pytest.raises(ValueError, match="canonical JSON"): normalize_pre_tool_use(event, state_hash="state") + + +def test_control_plane_actions_are_not_normalized_as_governable_work(tmp_path) -> None: + root = tmp_path / "installed" / "marginal" + script = root / "scripts" / "marginal_control.py" + script.parent.mkdir(parents=True) + (root / ".codex-plugin").mkdir() + (root / ".codex-plugin" / "plugin.json").write_text("{}", encoding="utf-8") + script.write_text("#!/usr/bin/env python3\n", encoding="utf-8") + + assert is_control_plane_action(_event(f"python3 {script} status"), root) diff --git a/tests/integrations/codex/test_promotion.py b/tests/integrations/codex/test_promotion.py index 563593c..e5eee1f 100644 --- a/tests/integrations/codex/test_promotion.py +++ b/tests/integrations/codex/test_promotion.py @@ -1,7 +1,11 @@ from __future__ import annotations import json +from pathlib import Path +import pytest + +from marginal.governance_ledger import GovernanceLedger from marginal.integrations.codex.promotion import ( CoverageSummary, PromotionCriteria, @@ -43,6 +47,15 @@ def _summary(**overrides: object) -> CoverageSummary: return CoverageSummary(**defaults) # type: ignore[arg-type] +def _anchor(path: Path) -> tuple[Path, str, int]: + ledger_path = path / "evidence-v3.jsonl" + ledger = GovernanceLedger(ledger_path) + ledger.append({"event": "evidence"}) + report = ledger.verify() + assert report.root_hash is not None + return ledger_path, report.root_hash, report.records + + def test_default_gate_requires_minimum_actions() -> None: receipt = evaluate_promotion( _summary(covered_actions=99, coverable_actions=100), @@ -54,9 +67,41 @@ def test_default_gate_requires_minimum_actions() -> None: assert "MINIMUM_ACTIONS" in receipt.blocking_reasons -def test_all_default_thresholds_produce_ready_receipt() -> None: +def test_unanchored_receipt_is_never_ready_or_activatable(tmp_path: Path) -> None: receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=_identity()) + assert receipt.is_ready is False + assert "EVIDENCE_ROOT_UNVERIFIED" in receipt.blocking_reasons + write_promotion_receipt(tmp_path, receipt) + with pytest.raises(ValueError, match="ready"): + activate_enforcement(tmp_path, receipt) + + +def test_ready_receipt_requires_a_verifiable_v3_prefix(tmp_path: Path) -> None: + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=_identity(), + evidence_root="a" * 64, + ledger_records=1, + ledger_path=tmp_path / "missing-v3.jsonl", + ) + + assert receipt.is_ready is False + assert "EVIDENCE_ROOT_UNVERIFIED" in receipt.blocking_reasons + + +def test_all_default_thresholds_produce_ready_receipt(tmp_path: Path) -> None: + ledger_path, root, records = _anchor(tmp_path) + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=_identity(), + evidence_root=root, + ledger_records=records, + ledger_path=ledger_path, + ) + assert receipt.is_ready is True assert receipt.blocking_reasons == () assert receipt.coverage_ratio == 1.0 @@ -103,25 +148,42 @@ def test_receipt_round_trip_is_hash_verifiable() -> None: def test_active_enforcement_requires_ready_matching_receipt(tmp_path) -> None: identity = _identity() - receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + ledger_path, root, records = _anchor(tmp_path) + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=identity, + evidence_root=root, + ledger_records=records, + ledger_path=ledger_path, + ) write_promotion_receipt(tmp_path, receipt) - activate_enforcement(tmp_path, receipt) + activate_enforcement(tmp_path, receipt, ledger_path=ledger_path) - assert enforcement_is_active(tmp_path, identity=identity) is True + assert enforcement_is_active(tmp_path, identity=identity, ledger_path=ledger_path) is True assert read_promotion_receipt(tmp_path, identity.repository_hash) == receipt def test_identity_drift_automatically_demotes_receipt(tmp_path) -> None: identity = _identity() - receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + ledger_path, root, records = _anchor(tmp_path) + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=identity, + evidence_root=root, + ledger_records=records, + ledger_path=ledger_path, + ) write_promotion_receipt(tmp_path, receipt) - activate_enforcement(tmp_path, receipt) + activate_enforcement(tmp_path, receipt, ledger_path=ledger_path) assert ( enforcement_is_active( tmp_path, identity=_identity(policy_hash="changed"), + ledger_path=ledger_path, ) is False ) @@ -132,18 +194,47 @@ def test_identity_drift_automatically_demotes_receipt(tmp_path) -> None: def test_evidence_drift_automatically_demotes_receipt(tmp_path) -> None: identity = _identity() - receipt = evaluate_promotion(_summary(), PromotionCriteria(), identity=identity) + ledger_path, root, records = _anchor(tmp_path) + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=identity, + evidence_root=root, + ledger_records=records, + ledger_path=ledger_path, + ) write_promotion_receipt(tmp_path, receipt) - activate_enforcement(tmp_path, receipt) + activate_enforcement(tmp_path, receipt, ledger_path=ledger_path) assert ( enforcement_is_active( tmp_path, identity=identity, summary=_summary(integration_failures=1), + ledger_path=ledger_path, ) is False ) state = json.loads((tmp_path / "repositories" / f"{identity.repository_hash}.json").read_text()) assert state["mode"] == "shadow" assert state["reason"] == "EVIDENCE_DRIFT" + + +def test_anchored_receipt_requires_its_verified_v3_prefix_for_activation(tmp_path: Path) -> None: + ledger_path = tmp_path / "evidence-v3.jsonl" + ledger = GovernanceLedger(ledger_path) + ledger.append({"event": "evidence"}) + report = ledger.verify() + receipt = evaluate_promotion( + _summary(), + PromotionCriteria(), + identity=_identity(), + evidence_root=report.root_hash, + ledger_records=report.records, + ledger_path=ledger_path, + ) + write_promotion_receipt(tmp_path, receipt) + + activate_enforcement(tmp_path, receipt, ledger_path=ledger_path) + + assert enforcement_is_active(tmp_path, identity=_identity(), ledger_path=ledger_path) diff --git a/tests/integrations/codex/test_runtime.py b/tests/integrations/codex/test_runtime.py index 64a2261..0711042 100644 --- a/tests/integrations/codex/test_runtime.py +++ b/tests/integrations/codex/test_runtime.py @@ -6,7 +6,12 @@ import pytest from marginal import BudgetLimits, Treasury -from marginal.integrations.codex.events import PostToolUseEvent, PreToolUseEvent +from marginal.integrations.codex.events import ( + PostToolUseEvent, + PreToolUseEvent, + UserPromptSubmitEvent, +) +from marginal.integrations.codex.normalization import normalize_pre_tool_use from marginal.integrations.codex.runtime import CodexIntegrationError, CodexSessionRuntime from marginal.protocol import AgentCapabilities from marginal.runtime import UniversalRuntime @@ -26,7 +31,12 @@ def _repository(path: Path) -> Path: return path -def _runtime(workspace: Path, *, enforcement_enabled: bool = False) -> CodexSessionRuntime: +def _runtime( + workspace: Path, + *, + enforcement_enabled: bool = False, + plugin_root: Path | None = None, +) -> CodexSessionRuntime: universal = UniversalRuntime( Treasury(BudgetLimits(max_tokens=100), mode="shadow"), engine="codex", @@ -38,6 +48,7 @@ def _runtime(workspace: Path, *, enforcement_enabled: bool = False) -> CodexSess universal, workspace=workspace, enforcement_enabled=lambda: enforcement_enabled, + plugin_root=plugin_root, ) @@ -160,3 +171,52 @@ def test_shadow_mode_never_applies_no_progress_denial(tmp_path: Path) -> None: assert runtime.last_no_progress_signal is not None assert runtime.last_no_progress_signal.enforcement_eligible is True assert runtime.summary()["enforced_denials"] == 0 + + +def test_user_prompt_intent_is_ephemeral_and_bound_to_the_hook_session(tmp_path: Path) -> None: + runtime = _runtime(_repository(tmp_path)) + event = UserPromptSubmitEvent( + session_id="session-1", + cwd="/workspace", + hook_event_name="UserPromptSubmit", + model="gpt-5.6-sol", + permission_mode="default", + prompt="Esegui di nuovo", + ) + + runtime.user_prompt_submit(event) + + assert runtime.user_intent.repeat_requested is True + assert "prompt" not in runtime.summary() + + +def test_trusted_control_plane_bypass_has_no_pending_workload_reservation(tmp_path: Path) -> None: + workspace = tmp_path / "workspace" + workspace.mkdir() + _repository(workspace) + root = tmp_path / "installed" / "marginal" + script = root / "scripts" / "marginal_control.py" + script.parent.mkdir(parents=True) + (root / ".codex-plugin").mkdir() + (root / ".codex-plugin" / "plugin.json").write_text("{}", encoding="utf-8") + script.write_text("#!/usr/bin/env python3\n", encoding="utf-8") + runtime = _runtime(workspace, enforcement_enabled=True, plugin_root=root) + + decision = runtime.pre_tool_use(_pre("control-1", command=f"python3 {script} status")) + + assert decision.allowed is True + assert decision.reason_code == "control_plane_bypass" + assert runtime.pending_action_ids() == () + assert runtime.summary()["enforced_denials"] == 0 + + +@pytest.mark.parametrize("command", ["pytest -q", "curl https://example.com"]) +def test_autopilot_does_not_classify_generic_shell_as_l3_eligible( + tmp_path: Path, command: str +) -> None: + workspace = _repository(tmp_path) + runtime = _runtime(workspace) + event = _pre("shell", command=command) + action = normalize_pre_tool_use(event, state_hash="state") + + assert runtime._is_autopilot_eligible(event, action) is False diff --git a/tests/integrations/codex/test_service.py b/tests/integrations/codex/test_service.py index 49be327..9d50412 100644 --- a/tests/integrations/codex/test_service.py +++ b/tests/integrations/codex/test_service.py @@ -7,7 +7,11 @@ import marginal.integrations.codex.service as service_module from marginal.integrations.codex.events import SessionEvent -from marginal.integrations.codex.evidence import EvidenceStore, summarize_evidence +from marginal.integrations.codex.evidence import ( + EvidenceStore, + summarize_evidence, + summarize_verified_evidence, +) from marginal.integrations.codex.identity import current_promotion_identity from marginal.integrations.codex.promotion import ( CoverageSummary, @@ -121,6 +125,7 @@ def test_bootstrap_redacts_transcript_and_hashes_session_filename(tmp_path: Path assert event.session_id not in bootstrap.name assert "transcript_path" not in payload assert "/private/raw-transcript.jsonl" not in json.dumps(payload) + assert "source" not in payload def test_missing_service_fails_open_and_demotes(tmp_path: Path) -> None: @@ -141,9 +146,19 @@ def test_missing_service_fails_open_and_demotes(tmp_path: Path) -> None: decision_latencies_ms=(1.0,), enforceable_outcomes_observable=True, ) - receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + store = EvidenceStore(data / "evidence" / identity.repository_hash) + store.append({"schema_version": 1, "event": "session_start", "session_hash": "seed"}) + root = store.verified_governance_root() + receipt = evaluate_promotion( + summary, + PromotionCriteria(), + identity=identity, + evidence_root=root.root_hash, + ledger_records=root.records, + ledger_path=store.governance_ledger_path, + ) write_promotion_receipt(data, receipt) - activate_enforcement(data, receipt) + activate_enforcement(data, receipt, ledger_path=store.governance_ledger_path) pre_payload = { "session_id": "missing", "cwd": str(workspace), @@ -213,6 +228,36 @@ def test_session_start_and_end_are_complete_hook_lifecycle(tmp_path: Path) -> No assert not (data / "sessions" / connection_filename("session-1")).exists() +def test_user_prompt_submit_reaches_only_the_authenticated_session_and_is_not_persisted( + tmp_path: Path, +) -> None: + workspace = tmp_path / "repo" + workspace.mkdir() + _repository(workspace) + data = tmp_path / "data" + start = json.loads(json.dumps(asdict(_start(workspace)))) + prompt = { + **start, + "hook_event_name": "UserPromptSubmit", + "prompt": "MARGINAL_PROMPT_SECRET: esegui di nuovo", + } + end = {**start, "hook_event_name": "SessionEnd", "source": None, "reason": "other"} + + assert run_hook(start, data_root=data).exit_code == 0 + try: + prompt_result = run_hook(prompt, data_root=data) + assert prompt_result.output is None + assert prompt_result.warning_code == "" + finally: + run_hook(end, data_root=data) + + persisted = "\n".join( + path.read_text(encoding="utf-8") for path in data.rglob("*") if path.is_file() + ) + assert "MARGINAL_PROMPT_SECRET" not in persisted + assert "prompt_hash" not in persisted + + def test_ready_repository_enforces_proven_no_progress_and_only_that(tmp_path: Path) -> None: workspace = tmp_path / "repo" workspace.mkdir() @@ -221,10 +266,17 @@ def test_ready_repository_enforces_proven_no_progress_and_only_that(tmp_path: Pa identity = current_promotion_identity(workspace) store = EvidenceStore(data / "evidence" / identity.repository_hash) _seed_ready_evidence(store) - summary = summarize_evidence(store.read_all()) - receipt = evaluate_promotion(summary, PromotionCriteria(), identity=identity) + summary, root = summarize_verified_evidence(store) + receipt = evaluate_promotion( + summary, + PromotionCriteria(), + identity=identity, + evidence_root=root.root_hash, + ledger_records=root.records, + ledger_path=store.governance_ledger_path, + ) write_promotion_receipt(data, receipt) - activate_enforcement(data, receipt) + activate_enforcement(data, receipt, ledger_path=store.governance_ledger_path) start_payload = json.loads(json.dumps(asdict(_start(workspace)))) assert run_hook(start_payload, data_root=data).exit_code == 0 try: diff --git a/tests/plugin/test_codex_plugin.py b/tests/plugin/test_codex_plugin.py index 35c965b..452f473 100644 --- a/tests/plugin/test_codex_plugin.py +++ b/tests/plugin/test_codex_plugin.py @@ -101,7 +101,13 @@ def test_directory_submission_archive_is_reproducible_and_complete(tmp_path: Pat def test_hooks_cover_exact_supported_lifecycle() -> None: hooks = json.loads((PLUGIN / "hooks" / "hooks.json").read_text(encoding="utf-8")) - assert set(hooks["hooks"]) == {"SessionStart", "PreToolUse", "PostToolUse", "SessionEnd"} + assert set(hooks["hooks"]) == { + "SessionStart", + "PreToolUse", + "PostToolUse", + "SessionEnd", + "UserPromptSubmit", + } for groups in hooks["hooks"].values(): command = groups[0]["hooks"][0] assert command["type"] == "command" diff --git a/tests/test_authority.py b/tests/test_authority.py new file mode 100644 index 0000000..2b38385 --- /dev/null +++ b/tests/test_authority.py @@ -0,0 +1,93 @@ +from __future__ import annotations + +from dataclasses import replace + +import pytest + +from marginal.authority import ( + AuthorityLevel, + AuthorityTransitionReceipt, + transition_receipt_hash, + verify_transition_receipt, +) +from marginal.governance_ledger import LedgerVerificationReport + +ROOT = "a" * 64 +VERIFIED_REPORT = LedgerVerificationReport(True, 3, ROOT, None, ()) + + +def test_authority_levels_are_ordered_from_observation_to_compute_governance() -> None: + """Catches assigning a stronger authority below a weaker intervention.""" + + assert list(AuthorityLevel) == [ + AuthorityLevel.OBSERVE, + AuthorityLevel.ADVISE, + AuthorityLevel.SOFT_INTERVENE, + AuthorityLevel.TOOL_GATE, + AuthorityLevel.COMPUTE_GOVERN, + ] + + +def test_transition_receipt_binds_the_evidence_ledger_root_and_detects_tampering() -> None: + """Catches a transition attestation that survives an edited authority or evidence root.""" + + unsigned = AuthorityTransitionReceipt( + schema_version="1.0", + context={"repository": "repo", "agent": "agent", "model": "model"}, + previous=AuthorityLevel.ADVISE, + current=AuthorityLevel.SOFT_INTERVENE, + evidence_ledger_root=ROOT, + ledger_verification=VERIFIED_REPORT, + blockers=(), + receipt_hash="", + ) + receipt = replace(unsigned, receipt_hash=transition_receipt_hash(unsigned)) + + assert verify_transition_receipt(receipt) + assert not verify_transition_receipt(replace(receipt, current=AuthorityLevel.TOOL_GATE)) + with pytest.raises(ValueError, match="ledger_verification"): + replace(receipt, evidence_ledger_root="b" * 64) + + +def test_transition_receipt_requires_a_valid_report_for_its_exact_ledger_root() -> None: + """Catches treating any digest-shaped string as verified governance evidence.""" + + with pytest.raises(ValueError, match="ledger_verification"): + AuthorityTransitionReceipt( + schema_version="1.0", + context={"repository": "repo"}, + previous=AuthorityLevel.OBSERVE, + current=AuthorityLevel.ADVISE, + evidence_ledger_root=ROOT, + ledger_verification=LedgerVerificationReport(False, 3, ROOT, 3, ("BAD",)), + blockers=(), + receipt_hash="", + ) + with pytest.raises(ValueError, match="evidence_ledger_root"): + AuthorityTransitionReceipt( + schema_version="1.0", + context={"repository": "repo"}, + previous=AuthorityLevel.OBSERVE, + current=AuthorityLevel.ADVISE, + evidence_ledger_root=ROOT, + ledger_verification=LedgerVerificationReport(True, 3, "b" * 64, None, ()), + blockers=(), + receipt_hash="", + ) + + +@pytest.mark.parametrize("root", ["", "not-a-root", "A" * 64, "a" * 63]) +def test_transition_receipt_rejects_an_invalid_evidence_ledger_root(root: str) -> None: + """Catches emitting a transition receipt that cannot anchor a verified evidence chain.""" + + with pytest.raises(ValueError, match="evidence_ledger_root"): + AuthorityTransitionReceipt( + schema_version="1.0", + context={"repository": "repo"}, + previous=AuthorityLevel.OBSERVE, + current=AuthorityLevel.ADVISE, + evidence_ledger_root=root, + ledger_verification=VERIFIED_REPORT, + blockers=(), + receipt_hash="", + ) diff --git a/tests/test_canonical.py b/tests/test_canonical.py new file mode 100644 index 0000000..6254490 --- /dev/null +++ b/tests/test_canonical.py @@ -0,0 +1,45 @@ +from __future__ import annotations + +import pytest + +from marginal.canonical import canonical_bytes, canonical_hash +from marginal.fingerprint import fingerprint_action +from marginal.models import Action +from marginal.policy import MarginalPolicy, PolicyConfig + + +def test_canonical_hash_is_stable_across_mapping_key_order() -> None: + """Catches a serializer regression that stops sorting mapping keys.""" + + first = {"z": [2, 1], "a": "café"} + second = {"a": "café", "z": [2, 1]} + + assert canonical_bytes(first) == b'{"a":"caf\xc3\xa9","z":[2,1]}' + assert canonical_hash(first) == canonical_hash(second) + + +@pytest.mark.parametrize("value", [float("nan"), {"value": float("nan")}, b"not-json"]) +def test_canonical_serialization_rejects_non_json_values(value: object) -> None: + """Catches accepting NaN or values without a JSON representation as attestation input.""" + + with pytest.raises((TypeError, ValueError)): + canonical_bytes(value) + + +def test_existing_fingerprint_and_policy_hashes_remain_compatible() -> None: + """Catches a shared-hash refactor changing frozen action or policy identities.""" + + action = Action( + name="café", + kind="tool", + metadata={"z": 1, "a": "é"}, + ) + + assert ( + fingerprint_action(action) + == "7ca348d0e0f9734011057570463aea0ef57b7d40e793e8059ccd21b730225b84" + ) + assert ( + MarginalPolicy(PolicyConfig(minimum_roi=1.2)).identity.config_hash + == "c4ace16a8fffaea091fbda0757630ff52f10867f5538adcd752f1e8f866f3388" + ) diff --git a/tests/test_cli_v2.py b/tests/test_cli_v2.py index 01224ca..56cbafe 100644 --- a/tests/test_cli_v2.py +++ b/tests/test_cli_v2.py @@ -173,3 +173,21 @@ def test_ledger_export_uses_owner_only_permissions(tmp_path: Path) -> None: if os.name != "nt": assert destination.stat().st_mode & 0o077 == 0 + + +def test_top_level_diagnostics_commands_share_json_reports(tmp_path: Path, capsys) -> None: + assert main(["status", "--data-dir", str(tmp_path), "--json"]) == 0 + status = json.loads(capsys.readouterr().out) + assert status["authority"]["current"] == "L0" + + assert main(["privacy", "inspect", "--json"]) == 0 + privacy = json.loads(capsys.readouterr().out) + assert "derived_enums" in privacy["persisted_categories"] + + assert main(["explain", "missing", "--data-dir", str(tmp_path), "--json"]) == 1 + explanation = json.loads(capsys.readouterr().out) + assert explanation == { + "decision_id": "missing", + "found": False, + "reason_code": "DECISION_NOT_FOUND", + } diff --git a/tests/test_diagnostics.py b/tests/test_diagnostics.py new file mode 100644 index 0000000..5b3008c --- /dev/null +++ b/tests/test_diagnostics.py @@ -0,0 +1,146 @@ +from __future__ import annotations + +from pathlib import Path + +from marginal.diagnostics import ( + decision_explanation, + doctor_report, + inspect_privacy, + status_report, +) +from marginal.integrations.codex.evidence import EvidenceStore +from marginal.integrations.codex.identity import current_promotion_identity + + +def test_status_exposes_only_observed_authority_trust_and_exact_blockers(tmp_path: Path) -> None: + workspace = tmp_path / "repository" + workspace.mkdir() + data_root = tmp_path / "data" + identity = current_promotion_identity(workspace, codex_version="test") + store = EvidenceStore(data_root / "evidence" / identity.repository_hash) + store.append( + { + "schema_version": 1, + "event": "decision", + "session_hash": "session", + "action_hash": "decision-1", + "semantic_key": "repeat", + "state_hash": "state", + "evidence_hash": "evidence", + "outcome": "success", + "reason_code": "APPROVED", + "latency_ms": 4.0, + "covered": True, + "coverable": True, + "recommended_stop": False, + "reviewed": False, + "false_stop": False, + "pending": False, + } + ) + + report = status_report(data_root=data_root, workspace=workspace) + payload = report.to_dict() + + assert payload["authority"]["current"] == "L0" + assert payload["authority"]["eligible"] == "L0" + assert payload["trust"]["components"]["coverage_ratio"] == 1.0 + assert "MINIMUM_ACTIONS" in payload["next_promotion_blockers"] + assert payload["ledger"]["valid"] is True + assert payload["counters"] == {"avoided_actions": 0, "recoveries": 0} + + +def test_decision_explanation_is_deterministic_and_uses_redacted_evidence(tmp_path: Path) -> None: + workspace = tmp_path / "repository" + workspace.mkdir() + data_root = tmp_path / "data" + identity = current_promotion_identity(workspace, codex_version="test") + EvidenceStore(data_root / "evidence" / identity.repository_hash).append( + { + "schema_version": 1, + "event": "decision", + "session_hash": "session", + "action_hash": "decision-1", + "semantic_key": "repeat", + "state_hash": "state", + "evidence_hash": "evidence", + "reason_code": "NO_PROGRESS_RECOMMENDED_UNKNOWN", + "latency_ms": 4.0, + "covered": True, + "coverable": True, + "recommended_stop": True, + "reviewed": False, + "false_stop": False, + "pending": True, + } + ) + + first = decision_explanation("decision-1", data_root=data_root, workspace=workspace).to_dict() + second = decision_explanation("decision-1", data_root=data_root, workspace=workspace).to_dict() + + assert first == second + assert first["found"] is True + assert first["reason_code"] == "NO_PROGRESS_RECOMMENDED_UNKNOWN" + assert "command" not in first + assert "prompt" not in first + + +def test_privacy_inspection_lists_persisted_categories_and_exclusions() -> None: + payload = inspect_privacy().to_dict() + + assert payload["persisted_categories"] == [ + "derived_enums", + "counts_and_metrics", + "pseudonymous_hashes", + "integrity_receipts", + ] + assert "prompt" in payload["never_persisted"] + assert "credentials" in payload["never_persisted"] + + +def test_status_reports_unvalidated_enforcement_as_configured_but_not_effective( + tmp_path: Path, +) -> None: + workspace = tmp_path / "repository" + workspace.mkdir() + data_root = tmp_path / "data" + identity = current_promotion_identity(workspace) + state_path = data_root / "repositories" / f"{identity.repository_hash}.json" + state_path.parent.mkdir(parents=True) + state_path.write_text( + '{"mode":"enforce","receipt_hash":"missing","schema_version":1}\n', + encoding="utf-8", + ) + + payload = status_report(data_root=data_root, workspace=workspace).to_dict() + + assert payload["authority"]["configured_mode"] == "enforce" + assert payload["authority"]["current"] == "L0" + assert payload["authority"]["effective"] == "L0" + assert payload["authority"]["effective_blockers"] == ["PROMOTION_RECEIPT_MISSING"] + + +def test_doctor_checks_schema_policy_provenance_and_all_governance_permissions( + tmp_path: Path, +) -> None: + workspace = tmp_path / "repository" + workspace.mkdir() + data_root = tmp_path / "data" + identity = current_promotion_identity(workspace) + consent = data_root / "user-config.json" + consent.parent.mkdir() + consent.write_text('{"autopilot_consent":true,"schema_version":1}\n', encoding="utf-8") + consent.chmod(0o600) + state = data_root / "repositories" / f"{identity.repository_hash}.json" + state.parent.mkdir() + state.write_text('{"mode":"shadow","schema_version":1}\n', encoding="utf-8") + state.chmod(0o600) + + payload = doctor_report(data_root=data_root, workspace=workspace).to_dict() + + assert payload["schemas"]["valid"] is True + assert payload["effective_policy"]["identity"]["repository_hash"] == identity.repository_hash + assert payload["effective_policy"]["provenance"]["present"] is True + assert payload["permissions"]["autopilot_consent"] == "owner_only" + assert payload["permissions"]["enforcement_state"] == "owner_only" + assert payload["permissions"]["enforcement_receipt"] == "not_created" diff --git a/tests/test_governance_ledger.py b/tests/test_governance_ledger.py new file mode 100644 index 0000000..56ef34f --- /dev/null +++ b/tests/test_governance_ledger.py @@ -0,0 +1,307 @@ +from __future__ import annotations + +import json +import os +from pathlib import Path + +import pytest + +from marginal import governance_ledger +from marginal.canonical import canonical_bytes, canonical_hash +from marginal.cli import main +from marginal.governance_ledger import GovernanceLedger + + +def test_append_writes_contiguous_hash_linked_canonical_records(tmp_path: Path) -> None: + """Removing sequence or linkage binding must invalidate this observable chain.""" + + path = tmp_path / "governance.jsonl" + ledger = GovernanceLedger(path) + + first_hash = ledger.append({"kind": "receipt", "value": 1}) + second_hash = ledger.append({"kind": "outcome", "value": 2}) + + records = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] + assert [record["sequence"] for record in records] == [1, 2] + assert records[0]["previous_hash"] is None + assert records[1]["previous_hash"] == first_hash + assert records[0]["record_hash"] == first_hash + assert records[1]["record_hash"] == second_hash + report = ledger.verify(expected_root=second_hash) + assert report.valid is True + assert report.records == 2 + assert report.root_hash == second_hash + + +def _write_records(path: Path, records: list[dict[str, object]]) -> None: + path.write_bytes(b"".join(canonical_bytes(record) + b"\n" for record in records)) + + +def _rehashed(record: dict[str, object]) -> dict[str, object]: + record["record_hash"] = canonical_hash( + {key: value for key, value in record.items() if key != "record_hash"} + ) + return record + + +@pytest.mark.parametrize( + ("mutation", "expected_code"), + [ + ( + lambda records: records[1].update({"payload": {"kind": "outcome", "value": 99}}), + "PAYLOAD_HASH_MISMATCH", + ), + ( + lambda records: _rehashed(records[1].update({"previous_hash": "0" * 64}) or records[1]), + "PREVIOUS_HASH_MISMATCH", + ), + ( + lambda records: records[1].update({"record_hash": "0" * 64}), + "RECORD_HASH_MISMATCH", + ), + ], +) +def test_verify_fails_closed_for_hash_chain_corruption( + tmp_path: Path, mutation, expected_code: str +) -> None: + """A changed payload, link, or envelope hash must never verify as evidence.""" + + path = tmp_path / "governance.jsonl" + ledger = GovernanceLedger(path) + ledger.append({"kind": "receipt", "value": 1}) + ledger.append({"kind": "outcome", "value": 2}) + records = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] + mutation(records) + _write_records(path, records) + + report = ledger.verify() + + assert report.valid is False + assert report.records == 1 + assert report.first_invalid_sequence == 2 + assert report.error_codes == (expected_code,) + + +def test_verify_detects_deleted_middle_record_and_expected_root_mismatch(tmp_path: Path) -> None: + """Dropping evidence or pointing at another root must fail closed.""" + + path = tmp_path / "governance.jsonl" + ledger = GovernanceLedger(path) + ledger.append({"kind": "first"}) + ledger.append({"kind": "second"}) + root = ledger.append({"kind": "third"}) + records = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] + _write_records(path, [records[0], records[2]]) + + missing = ledger.verify() + assert missing.valid is False + assert missing.first_invalid_sequence == 2 + assert missing.error_codes == ("SEQUENCE_GAP",) + + _write_records(path, records) + mismatch = ledger.verify(expected_root="0" * 64) + assert mismatch.valid is False + assert mismatch.root_hash == root + assert mismatch.error_codes == ("EXPECTED_ROOT_MISMATCH",) + + +def test_verify_rejects_incompatible_schema_and_noncanonical_lines(tmp_path: Path) -> None: + """Schema drift and noncanonical encodings are not attested records.""" + + path = tmp_path / "governance.jsonl" + ledger = GovernanceLedger(path) + ledger.append({"kind": "receipt"}) + record = json.loads(path.read_text(encoding="utf-8")) + record["schema_version"] = "4.0" + _write_records(path, [_rehashed(record)]) + assert ledger.verify().error_codes == ("UNSUPPORTED_SCHEMA",) + + path.write_text(json.dumps(record) + "\n", encoding="utf-8") + assert ledger.verify().error_codes == ("NONCANONICAL_LINE",) + + +def test_append_rejects_unsafe_or_non_json_payloads(tmp_path: Path) -> None: + """Attestations never stringify arbitrary objects or persist raw private fields.""" + + ledger = GovernanceLedger(tmp_path / "governance.jsonl") + + with pytest.raises(TypeError, match="canonical JSON"): + ledger.append({"evidence": object()}) + with pytest.raises(ValueError, match="private field"): + ledger.append({"raw_prompt": "do not persist"}) + with pytest.raises(ValueError, match="private field"): + ledger.append({"rawPrompt": "do not persist"}) + + +def test_governance_ledger_uses_owner_only_permissions_and_refuses_unsafe_paths( + tmp_path: Path, +) -> None: + """A public, linked, or non-regular target cannot become governance evidence.""" + + path = tmp_path / "governance.jsonl" + GovernanceLedger(path).append({"kind": "receipt"}) + if os.name != "nt": + assert path.stat().st_mode & 0o077 == 0 + link = tmp_path / "linked.jsonl" + try: + link.symlink_to(path) + except (OSError, NotImplementedError): + pytest.skip("symbolic links are unavailable") + with pytest.raises(ValueError, match="symbolic link"): + GovernanceLedger(link) + with pytest.raises(ValueError, match="regular file"): + GovernanceLedger(tmp_path) + + +def test_quarantine_preserves_source_and_copies_only_invalid_suffix(tmp_path: Path) -> None: + """Quarantine records corruption explicitly without deleting evidence.""" + + from marginal.governance_ledger import quarantine_invalid_records + + source = tmp_path / "governance.jsonl" + ledger = GovernanceLedger(source) + ledger.append({"kind": "first"}) + ledger.append({"kind": "second"}) + original = source.read_bytes() + records = [json.loads(line) for line in original.splitlines()] + records[1]["payload"] = {"kind": "tampered"} + _write_records(source, records) + corrupt_source = source.read_bytes() + + quarantine = quarantine_invalid_records(source, tmp_path / "quarantine") + + assert source.read_bytes() == corrupt_source + assert (quarantine / "invalid-records.jsonl").read_bytes() == canonical_bytes( + records[1] + ) + b"\n" + report = json.loads((quarantine / "report.json").read_text(encoding="utf-8")) + assert report["first_invalid_sequence"] == 2 + assert report["error_codes"] == ["PAYLOAD_HASH_MISMATCH"] + if os.name != "nt": + assert quarantine.stat().st_mode & 0o077 == 0 + + +def test_cli_verify_has_stable_json_success_and_integrity_error(tmp_path: Path, capsys) -> None: + """CLI consumers receive typed verification reports for valid and invalid chains.""" + + path = tmp_path / "governance.jsonl" + root = GovernanceLedger(path).append({"kind": "receipt"}) + + assert main(["verify", str(path), "--expected-root", root, "--json"]) == 0 + valid = json.loads(capsys.readouterr().out) + assert valid == { + "error_codes": [], + "first_invalid_sequence": None, + "records": 1, + "root_hash": root, + "valid": True, + } + + path.write_text('{"corrupted":true}\n', encoding="utf-8") + assert main(["verify", str(path), "--json"]) == 1 + invalid = json.loads(capsys.readouterr().out) + assert invalid["valid"] is False + assert invalid["error_codes"] == ["INVALID_RECORD"] + + +def test_append_rejects_traversal_and_symlinked_parent_before_creating_directories( + tmp_path: Path, +) -> None: + """A ledger path is an absolute, lexical path below no symlinked component.""" + + traversing = tmp_path / "untrusted" / ".." / "ledger.jsonl" + with pytest.raises(ValueError, match="traversal"): + GovernanceLedger(traversing) + assert not (tmp_path / "untrusted").exists() + + target = tmp_path / "target" + target.mkdir() + link = tmp_path / "link" + try: + link.symlink_to(target, target_is_directory=True) + except (OSError, NotImplementedError): + pytest.skip("symbolic links are unavailable") + with pytest.raises(ValueError, match="parent"): + GovernanceLedger(link / "ledger.jsonl").append({"kind": "receipt"}) + assert not (target / "ledger.jsonl").exists() + + +def test_verify_rejects_unsafe_path_without_creating_missing_parent(tmp_path: Path) -> None: + """Verification is read-only even when an attacker supplies a missing path.""" + + path = tmp_path / "missing-parent" / "ledger.jsonl" + + report = GovernanceLedger(path).verify() + + assert report.valid is False + assert report.error_codes == ("IO_ERROR",) + assert not path.parent.exists() + + +def test_verify_rejects_non_integer_sequence_and_naive_timestamp(tmp_path: Path) -> None: + """The chain enforces the schema's integer and RFC3339-style core envelope fields.""" + + path = tmp_path / "governance.jsonl" + GovernanceLedger(path).append({"kind": "receipt"}) + record = json.loads(path.read_text(encoding="utf-8")) + record["sequence"] = 1.0 + _write_records(path, [_rehashed(record)]) + assert GovernanceLedger(path).verify().error_codes == ("INVALID_RECORD",) + + record["sequence"] = 1 + record["timestamp"] = "2026-08-14T00:00:00" + _write_records(path, [_rehashed(record)]) + assert GovernanceLedger(path).verify().error_codes == ("INVALID_TIMESTAMP",) + + +def test_cli_verify_serializes_constructor_path_errors(tmp_path: Path, capsys) -> None: + """Unsafe verify input returns the same structured failure shape as bad evidence.""" + + assert main(["verify", str(tmp_path), "--json"]) == 1 + + payload = json.loads(capsys.readouterr().out) + assert payload == { + "error_codes": ["IO_ERROR"], + "first_invalid_sequence": None, + "records": 0, + "root_hash": None, + "valid": False, + } + + +def test_append_and_verify_open_the_leaf_relative_to_a_nofollow_directory_fd( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Parent validation cannot be invalidated between traversal and leaf open.""" + + path = tmp_path / "parents" / "governance.jsonl" + calls: list[tuple[object, int | None]] = [] + original_open = os.open + + def recording_open( + name: object, flags: int, mode: int = 0o777, *, dir_fd: int | None = None + ) -> int: + calls.append((name, dir_fd)) + if dir_fd is None: + return original_open(name, flags, mode) + return original_open(name, flags, mode, dir_fd=dir_fd) + + # Preserve real system behavior while recording the descriptor boundary contract. + monkeypatch.setattr(governance_ledger.os, "open", recording_open) + + ledger = GovernanceLedger(path) + ledger.append({"kind": "receipt"}) + ledger.verify() + + assert any(name == path.name and directory_fd is not None for name, directory_fd in calls) + + +def test_append_fails_closed_without_directory_fd_safety_primitives( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """No platform is advertised as race-safe when nofollow directory walks are unavailable.""" + + monkeypatch.setattr(governance_ledger.os, "O_DIRECTORY", 0) + + with pytest.raises(OSError, match="directory descriptor"): + GovernanceLedger(tmp_path / "governance.jsonl").append({"kind": "receipt"}) diff --git a/tests/test_ledger_migration_v3.py b/tests/test_ledger_migration_v3.py new file mode 100644 index 0000000..29e4801 --- /dev/null +++ b/tests/test_ledger_migration_v3.py @@ -0,0 +1,139 @@ +from __future__ import annotations + +import os +from pathlib import Path + +import pytest + +from marginal import governance_ledger +from marginal.cli import main +from marginal.governance_ledger import GovernanceLedger, migrate_v2_to_v3 +from marginal.ledger import DecisionLedgerContext, JsonlDecisionLedger + + +def test_migration_is_deterministic_preserves_v2_source_and_verifies_root(tmp_path: Path) -> None: + """Changing migration order, timestamps, or source data must change this chain.""" + + source = tmp_path / "decision-v2.jsonl" + legacy = JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")) + legacy.emit({"event": "custom", "evidence": {"kind": "first"}}) + legacy.emit({"event": "custom", "evidence": {"kind": "second"}}) + original = source.read_bytes() + + first_destination = tmp_path / "first-v3.jsonl" + second_destination = tmp_path / "second-v3.jsonl" + first = migrate_v2_to_v3(source, first_destination) + second = migrate_v2_to_v3(source, second_destination) + + assert source.read_bytes() == original + assert first.valid is True + assert first.records == 2 + assert first.root_hash is not None + assert second.root_hash == first.root_hash + assert second_destination.read_bytes() == first_destination.read_bytes() + assert GovernanceLedger(first_destination).verify(expected_root=first.root_hash).valid is True + + +def test_migration_rejects_unsafe_v2_records_without_creating_destination(tmp_path: Path) -> None: + """A legacy raw prompt is never copied into the governance-ledger v3.""" + + source = tmp_path / "unsafe-v2.jsonl" + JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")).emit( + {"event": "custom", "raw_prompt": "credential-bearing request"} + ) + destination = tmp_path / "unsafe-v3.jsonl" + + with pytest.raises(ValueError, match="private field"): + migrate_v2_to_v3(source, destination) + + assert not destination.exists() + + +def test_migration_never_overwrites_existing_destination(tmp_path: Path) -> None: + """A pre-existing v3 file remains authoritative during a migration attempt.""" + + source = tmp_path / "decision-v2.jsonl" + JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")).emit( + {"event": "custom"} + ) + destination = tmp_path / "existing-v3.jsonl" + destination.write_text("authoritative", encoding="utf-8") + + with pytest.raises(FileExistsError, match="already exists"): + migrate_v2_to_v3(source, destination) + + assert destination.read_text(encoding="utf-8") == "authoritative" + + +def test_cli_ledger_migrate_reports_verified_record_count_and_root(tmp_path: Path, capsys) -> None: + """The migration command reports the verified destination rather than a blind copy.""" + + source = tmp_path / "decision-v2.jsonl" + JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")).emit( + {"event": "custom"} + ) + destination = tmp_path / "decision-v3.jsonl" + + assert main(["ledger-migrate", str(source), str(destination)]) == 0 + + output = capsys.readouterr().out + report = GovernanceLedger(destination).verify() + assert report.valid is True + assert f"migrated 1 records to {destination}; root {report.root_hash}" in output + + +def test_migration_and_quarantine_reject_symlinked_destination_parent(tmp_path: Path) -> None: + """Derived v3 evidence never escapes through a symlinked output directory.""" + + source = tmp_path / "decision-v2.jsonl" + JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")).emit( + {"event": "custom"} + ) + target = tmp_path / "target" + target.mkdir() + link = tmp_path / "link" + try: + link.symlink_to(target, target_is_directory=True) + except (OSError, NotImplementedError): + pytest.skip("symbolic links are unavailable") + + with pytest.raises(ValueError, match="parent"): + migrate_v2_to_v3(source, link / "migrated.jsonl") + + corrupt = tmp_path / "corrupt.jsonl" + ledger = GovernanceLedger(corrupt) + ledger.append({"kind": "receipt"}) + corrupt.write_text('{"invalid":true}\n', encoding="utf-8") + from marginal.governance_ledger import quarantine_invalid_records + + with pytest.raises(ValueError, match="parent"): + quarantine_invalid_records(corrupt, link / "quarantine") + assert not (target / "migrated.jsonl").exists() + assert not (target / "quarantine").exists() + + +def test_migration_opens_the_v2_source_relative_to_a_secure_directory_fd( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The legacy reader cannot reopen a name after the source parent is validated.""" + + source = tmp_path / "decision-v2.jsonl" + JsonlDecisionLedger(source, context=DecisionLedgerContext(run_id="run")).emit( + {"event": "custom"} + ) + destination = tmp_path / "decision-v3.jsonl" + calls: list[tuple[object, int | None]] = [] + original_open = os.open + + def recording_open( + name: object, flags: int, mode: int = 0o777, *, dir_fd: int | None = None + ) -> int: + calls.append((name, dir_fd)) + if dir_fd is None: + return original_open(name, flags, mode) + return original_open(name, flags, mode, dir_fd=dir_fd) + + monkeypatch.setattr(governance_ledger.os, "open", recording_open) + + assert migrate_v2_to_v3(source, destination).valid is True + assert any(name == source.name and directory_fd is not None for name, directory_fd in calls) diff --git a/tests/test_packaged_schemas_v2.py b/tests/test_packaged_schemas_v2.py index 29284d3..b218e87 100644 --- a/tests/test_packaged_schemas_v2.py +++ b/tests/test_packaged_schemas_v2.py @@ -12,9 +12,13 @@ "agent-decision-v1.json", "agent-event-v1.json", "decision-ledger-v2.json", + "decision-receipt-v1.json", + "governance-ledger-v3.json", "outcome-v1.json", + "progress-evidence-v1.json", "safe-telemetry-v1.json", "token-usage-v2.json", + "trust-snapshot-v1.json", } diff --git a/tests/test_reason_codes.py b/tests/test_reason_codes.py new file mode 100644 index 0000000..9702f7c --- /dev/null +++ b/tests/test_reason_codes.py @@ -0,0 +1,23 @@ +from __future__ import annotations + +from marginal.reason_codes import REASON_CODE_VERSION, ReasonCode + + +def test_reason_code_registry_has_the_documented_versioned_governance_values() -> None: + """Catches an accidental rename, removal, or unversioned addition to governance codes.""" + + assert REASON_CODE_VERSION == "1.0" + assert {code.name: code.value for code in ReasonCode} == { + "APPROVAL": "approval", + "INSUFFICIENT_EVIDENCE": "insufficient_evidence", + "INSUFFICIENT_TRUST": "insufficient_trust", + "REPEATED_ACTION": "repeated_action", + "NO_PROGRESS": "no_progress", + "USER_REQUESTED_REPEAT": "user_requested_repeat", + "CONTROL_PLANE_BYPASS": "control_plane_bypass", + "POLICY_REVOKED": "policy_revoked", + "DISTRIBUTION_SHIFT": "distribution_shift", + "INTEGRITY_FAILURE": "integrity_failure", + "RECOVERY": "recovery", + "OUTCOME_UNKNOWN": "outcome_unknown", + } diff --git a/tests/test_receipts.py b/tests/test_receipts.py new file mode 100644 index 0000000..eaed20b --- /dev/null +++ b/tests/test_receipts.py @@ -0,0 +1,146 @@ +from __future__ import annotations + +import json +from dataclasses import replace +from pathlib import Path + +import pytest +from jsonschema import Draft202012Validator + +from marginal.receipts import ( + DecisionReceipt, + GovernanceCost, + ProgressEvidence, + ProgressLevel, + decision_receipt_hash, + receipt_payload, + verify_decision_receipt, +) + +ROOT = Path(__file__).resolve().parents[1] + + +def _cost() -> GovernanceCost: + return GovernanceCost( + wall_clock_ms=12.5, + cpu_ms=None, + memory_peak_bytes=None, + storage_bytes=0, + tokens=0, + model_calls=0, + additional_tool_calls=1, + ) + + +def _receipt(*, decision_hash: str = "") -> DecisionReceipt: + return DecisionReceipt( + schema_version="1.0", + decision_id="decision-1", + timestamp="2026-08-14T12:00:00+00:00", + context={"repository": "repo-identity", "agent": "unknown"}, + decision="deny", + reason_code="no_progress", + state_hash=None, + evidence_hash=None, + trajectory_hash="trajectory-digest", + policy_hash="policy-digest", + decision_hash=decision_hash, + confidence=0.8, + expected_utility=None, + estimated_cost=None, + enforcement_level="tool_gate", + trust_snapshot={"eligible_authority": "tool_gate", "observed": 3}, + governance_cost=_cost(), + ) + + +def test_receipt_hash_binds_the_canonical_payload_and_detects_tampering() -> None: + """Catches verifying a stale hash after an applied decision has been edited.""" + + unsigned = _receipt() + receipt = replace(unsigned, decision_hash=decision_receipt_hash(unsigned)) + + assert verify_decision_receipt(receipt) + assert not verify_decision_receipt(replace(receipt, decision="allow")) + + +def test_receipt_payload_keeps_unavailable_measurements_explicit_and_mappings_immutable() -> None: + """Catches omitting unavailable evidence or retaining a caller-mutable attestation map.""" + + unsigned = _receipt() + receipt = replace(unsigned, decision_hash=decision_receipt_hash(unsigned)) + payload = receipt_payload(receipt) + + assert payload["state_hash"] is None + assert payload["evidence_hash"] is None + assert payload["expected_utility"] is None + assert payload["governance_cost"]["cpu_ms"] is None + with pytest.raises(TypeError): + receipt.context["repository"] = "changed" # type: ignore[index] + + +@pytest.mark.parametrize("confidence", [-0.01, 1.01, float("nan")]) +def test_receipts_and_progress_evidence_reject_invalid_confidence(confidence: float) -> None: + """Catches accepting confidence outside the bounded, finite evidence scale.""" + + with pytest.raises(ValueError): + replace(_receipt(), confidence=confidence) + with pytest.raises(ValueError): + ProgressEvidence( + schema_version="1.0", + level=ProgressLevel.ACTIVITY, + state_hash="state-digest", + evidence_hash="evidence-digest", + confidence=confidence, + verifier=None, + ) + + +def test_receipt_rejects_values_that_cannot_be_canonically_attested() -> None: + """Catches silently coercing arbitrary object representations into a decision receipt.""" + + with pytest.raises(TypeError): + replace(_receipt(), expected_utility={"score": object()}) + + +@pytest.mark.parametrize( + "field", + ["prompt_hash", "promptHash", "raw_command", "raw_output"], +) +def test_receipt_rejects_qualified_private_payload_keys(field: str) -> None: + """Catches persisting private payloads behind qualified or camel-case field names.""" + + with pytest.raises(ValueError, match="raw private payload"): + replace(_receipt(), expected_utility={field: "must-not-persist"}) + + +@pytest.mark.parametrize( + "decision_hash", + ["é" * 64, "not-a-sha256-digest", "g" * 64, "a" * 63], +) +def test_malformed_decision_hash_verifies_false_without_raising(decision_hash: str) -> None: + """Catches untrusted malformed receipt hashes escaping verification as exceptions.""" + + assert not verify_decision_receipt(_receipt(decision_hash=decision_hash)) + + +def test_real_receipt_and_progress_examples_validate_against_published_schemas() -> None: + """Catches schema drift that rejects the values emitted by the public receipt models.""" + + unsigned = _receipt() + receipt = replace(unsigned, decision_hash=decision_receipt_hash(unsigned)) + progress = ProgressEvidence( + schema_version="1.0", + level=ProgressLevel.VERIFIED_PROGRESS, + state_hash="state-digest", + evidence_hash="evidence-digest", + confidence=1.0, + verifier="test-suite", + ) + + for name, example in ( + ("decision-receipt-v1.json", receipt_payload(receipt)), + ("progress-evidence-v1.json", progress.payload()), + ): + schema = json.loads((ROOT / "schemas" / name).read_text(encoding="utf-8")) + Draft202012Validator(schema).validate(example) diff --git a/tests/test_trust.py b/tests/test_trust.py new file mode 100644 index 0000000..98def4d --- /dev/null +++ b/tests/test_trust.py @@ -0,0 +1,244 @@ +from __future__ import annotations + +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path + +from jsonschema import Draft202012Validator + +from marginal.authority import AuthorityLevel +from marginal.governance_ledger import LedgerVerificationReport +from marginal.trust import TrustContext, TrustEngine, TrustEvidence + +ROOT = Path(__file__).resolve().parents[1] +LEDGER_ROOT = "a" * 64 +LEDGER_VERIFICATION = LedgerVerificationReport(True, 3, LEDGER_ROOT, None, ()) +NOW = datetime(2026, 8, 14, 12, tzinfo=timezone.utc) + + +def _context() -> TrustContext: + return TrustContext( + repository="repo-identity", + agent="agent-identity", + model="model-identity", + task_class="local-tool", + policy_version="policy-1", + ) + + +def _evidence(**changes: object) -> TrustEvidence: + values: dict[str, object] = { + "observed": 25, + "evaluable": 20, + "covered": 19, + "coverable": 20, + "beneficial": 18, + "neutral": 1, + "harmful": 1, + "indeterminate": 0, + "governance_tax_ratio": 0.05, + "mean_regret": 0.05, + "integrity_valid": True, + "last_observed_at": NOW.isoformat(), + "evidence_ledger_root": LEDGER_ROOT, + "ledger_verification": LEDGER_VERIFICATION, + } + values.update(changes) + return TrustEvidence(**values) # type: ignore[arg-type] + + +def _engine() -> TrustEngine: + return TrustEngine(now=NOW) + + +def test_minimum_evaluable_samples_is_a_promotion_blocker() -> None: + """Catches promotion when the evaluated evidence window is too small.""" + + snapshot = _engine().evaluate( + _context(), + _evidence(evaluable=19, beneficial=17, neutral=1, harmful=1), + AuthorityLevel.OBSERVE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.OBSERVE + assert "minimum_evaluable_samples" in snapshot.blockers + assert snapshot.components["evaluable"] == 19 + + +def test_unclassified_evaluable_outcomes_block_promotion() -> None: + """Catches promotion when an evaluable intervention has no outcome classification.""" + + snapshot = _engine().evaluate( + _context(), + _evidence(beneficial=17, neutral=1, harmful=1, indeterminate=0), + AuthorityLevel.OBSERVE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.OBSERVE + assert "unclassified_outcomes" in snapshot.blockers + assert snapshot.components["unclassified_outcomes"] == 1 + + +def test_coverage_harm_regret_and_tax_are_separate_transparent_blockers() -> None: + """Catches hiding independent safety failures inside an opaque trust score.""" + + snapshot = _engine().evaluate( + _context(), + _evidence( + covered=18, + beneficial=17, + harmful=2, + mean_regret=0.11, + governance_tax_ratio=0.11, + ), + AuthorityLevel.OBSERVE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.OBSERVE + assert set(snapshot.blockers) >= { + "insufficient_coverage", + "harm_rate_too_high", + "mean_regret_too_high", + "governance_tax_too_high", + } + assert snapshot.components["coverage"] == 0.9 + assert snapshot.components["harm_rate"] == 0.1 + + +def test_capability_ceiling_and_hysteresis_allow_only_one_promotion_step() -> None: + """Catches authority bypassing an adapter ceiling or jumping directly to tool denial.""" + + first = _engine().evaluate(_context(), _evidence(), AuthorityLevel.OBSERVE, capabilities=3) + second = _engine().evaluate(_context(), _evidence(), AuthorityLevel.ADVISE, capabilities=3) + capped = _engine().evaluate( + _context(), _evidence(), AuthorityLevel.SOFT_INTERVENE, capabilities=2 + ) + + assert first.authority == AuthorityLevel.ADVISE + assert second.authority == AuthorityLevel.SOFT_INTERVENE + assert capped.authority == AuthorityLevel.SOFT_INTERVENE + assert "capability_ceiling" in capped.blockers + + +def test_soft_evidence_decay_steps_down_exactly_one_level_without_flapping() -> None: + """Catches a non-critical quality drop resetting authority or oscillating it upward.""" + + snapshot = _engine().evaluate( + _context(), _evidence(governance_tax_ratio=0.21), AuthorityLevel.TOOL_GATE, capabilities=4 + ) + + assert snapshot.authority == AuthorityLevel.SOFT_INTERVENE + assert "governance_tax_too_high" in snapshot.blockers + assert snapshot.transition_receipt is not None + assert snapshot.transition_receipt.previous == AuthorityLevel.TOOL_GATE + + +def test_demotion_requires_a_clean_recovery_window_before_repromotion() -> None: + """Catches L3 to L2 decay immediately flapping back to L3 at the promotion boundary.""" + + engine = _engine() + demoted = engine.evaluate( + _context(), _evidence(governance_tax_ratio=0.21), AuthorityLevel.TOOL_GATE, capabilities=4 + ) + recovered = engine.evaluate(_context(), _evidence(), demoted.authority, capabilities=4) + + assert demoted.authority == AuthorityLevel.SOFT_INTERVENE + assert recovered.authority == AuthorityLevel.SOFT_INTERVENE + assert "recovery_hysteresis" in recovered.blockers + + +def test_promotion_and_demotion_quality_thresholds_are_asymmetric() -> None: + """Catches a marginal tax increase immediately revoking an already-earned authority level.""" + + snapshot = _engine().evaluate( + _context(), _evidence(governance_tax_ratio=0.11), AuthorityLevel.TOOL_GATE, capabilities=4 + ) + + assert snapshot.authority == AuthorityLevel.TOOL_GATE + assert "governance_tax_too_high" in snapshot.blockers + + +def test_unverified_ledger_root_cannot_support_a_promotion() -> None: + """Catches issuing a receipt from a digest that a ledger verifier did not validate.""" + + snapshot = _engine().evaluate( + _context(), + _evidence(ledger_verification=LedgerVerificationReport(False, 3, LEDGER_ROOT, 3, ("BAD",))), + AuthorityLevel.OBSERVE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.OBSERVE + assert "unverified_evidence_ledger_root" in snapshot.blockers + + +def test_integrity_and_capability_drift_reset_directly_to_observation() -> None: + """Catches retaining enforcement authority after critical evidence or adapter failure.""" + + integrity = _engine().evaluate( + _context(), _evidence(integrity_valid=False), AuthorityLevel.TOOL_GATE, capabilities=4 + ) + capability = _engine().evaluate( + _context(), _evidence(), AuthorityLevel.TOOL_GATE, capabilities=2 + ) + + assert integrity.authority == AuthorityLevel.OBSERVE + assert "integrity_failure" in integrity.blockers + assert capability.authority == AuthorityLevel.OBSERVE + assert "capability_drift" in capability.blockers + + +def test_identity_shifts_reset_and_large_repository_shift_decays_one_level() -> None: + """Catches reusing authority across a changed model, policy, or repository distribution.""" + + model = _engine().evaluate( + _context(), _evidence(), AuthorityLevel.TOOL_GATE, capabilities=4, shift_reasons=("model",) + ) + policy = _engine().evaluate( + _context(), _evidence(), AuthorityLevel.TOOL_GATE, capabilities=4, shift_reasons=("policy",) + ) + repository = _engine().evaluate( + _context(), + _evidence(), + AuthorityLevel.TOOL_GATE, + capabilities=4, + shift_reasons=("large_repository",), + ) + + assert model.authority == AuthorityLevel.OBSERVE + assert policy.authority == AuthorityLevel.OBSERVE + assert repository.authority == AuthorityLevel.SOFT_INTERVENE + assert "distribution_shift" in repository.blockers + + +def test_inactivity_is_soft_decay_and_snapshot_matches_the_published_schema() -> None: + """Catches stale evidence retaining a gate and schema drift in the diagnostic payload.""" + + snapshot = _engine().evaluate( + _context(), + _evidence(last_observed_at=(NOW - timedelta(days=31)).isoformat()), + AuthorityLevel.SOFT_INTERVENE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.ADVISE + assert "inactivity" in snapshot.blockers + schema = json.loads((ROOT / "schemas" / "trust-snapshot-v1.json").read_text(encoding="utf-8")) + Draft202012Validator(schema).validate(snapshot.payload()) + + +def test_future_observation_outside_clock_skew_tolerance_softly_demotes() -> None: + """Catches future-dated evidence retaining or increasing enforcement authority.""" + + snapshot = _engine().evaluate( + _context(), + _evidence(last_observed_at=(NOW + timedelta(minutes=6)).isoformat()), + AuthorityLevel.TOOL_GATE, + capabilities=4, + ) + + assert snapshot.authority == AuthorityLevel.SOFT_INTERVENE + assert "future_observation" in snapshot.blockers diff --git a/tests/test_utility.py b/tests/test_utility.py new file mode 100644 index 0000000..4f7f13d --- /dev/null +++ b/tests/test_utility.py @@ -0,0 +1,92 @@ +from __future__ import annotations + +from marginal.receipts import GovernanceCost +from marginal.utility import MarginalUtilityEstimate, UtilityVector + + +def _cost(*, tokens: int = 10) -> GovernanceCost: + return GovernanceCost( + wall_clock_ms=10.0, + cpu_ms=None, + memory_peak_bytes=None, + storage_bytes=0, + tokens=tokens, + model_calls=1, + additional_tool_calls=0, + ) + + +def test_utility_comparison_keeps_verified_correctness_ahead_of_lower_compute() -> None: + """Catches selecting a cheaper action when its correctness evidence is weaker.""" + + verified = UtilityVector( + verified_correctness=1.0, + task_completion=0.4, + safety_risk=0.3, + latency_ms=100.0, + tokens=100, + monetary_cost=1.0, + governance_overhead=10.0, + ) + cheaper_unknown = UtilityVector( + verified_correctness=None, + task_completion=1.0, + safety_risk=0.0, + latency_ms=1.0, + tokens=1, + monetary_cost=0.0, + governance_overhead=0.0, + ) + + assert verified.compare(cheaper_unknown) > 0 + assert cheaper_unknown.compare(verified) < 0 + + +def test_unknown_correctness_never_becomes_a_scalar_efficiency_claim() -> None: + """Catches reporting a token-saving ratio when verified utility is unavailable.""" + + estimate = MarginalUtilityEstimate( + expected_utility=UtilityVector( + verified_correctness=None, + task_completion=1.0, + safety_risk=0.0, + latency_ms=10.0, + tokens=10, + monetary_cost=0.1, + governance_overhead=1.0, + ), + estimated_cost=_cost(), + uncertainty=0.2, + confidence=0.8, + provenance={"evidence_root": "root-digest"}, + commensurable_cost=10.0, + ) + + scorecard = estimate.scorecard() + + assert estimate.scalar_emu() is None + assert scorecard["scalar_emu"] is None + assert scorecard["expected_utility"]["verified_correctness"] is None + + +def test_commensurable_verified_utility_reports_a_scalar_emu() -> None: + """Catches dropping a justified EMU ratio from a fully measured comparable estimate.""" + + estimate = MarginalUtilityEstimate( + expected_utility=UtilityVector( + verified_correctness=0.5, + task_completion=0.6, + safety_risk=0.2, + latency_ms=10.0, + tokens=10, + monetary_cost=0.1, + governance_overhead=1.0, + ), + estimated_cost=_cost(), + uncertainty=0.2, + confidence=0.8, + provenance={"evidence_root": "root-digest"}, + commensurable_cost=2.0, + ) + + assert estimate.scalar_emu() == 0.25