diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index fa65786..083e99f 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -121,7 +121,17 @@ jobs:
python -m ldtc.cli.main run --config configs/profile_ci.yml
- name: Verify indicators
run: |
+ # Each run is isolated under artifacts/runs/-/ (the
+ # audit chain is per-run); verify the most recent one. Keys are
+ # shared under artifacts/keys/. The loop (rather than `ls | head`)
+ # is pipefail-safe under the default `bash -eo pipefail` shell.
+ run_dir=""
+ for d in artifacts/runs/*/; do
+ if [ -z "$run_dir" ] || [ "$d" -nt "$run_dir" ]; then run_dir="$d"; fi
+ done
+ run_dir=${run_dir%/}
+ echo "Verifying latest run: $run_dir"
python scripts/verify_indicators.py \
- --ind-dir artifacts/indicators \
- --audit artifacts/audits/audit.jsonl \
+ --ind-dir "$run_dir/indicators" \
+ --audit "$run_dir/audits/audit.jsonl" \
--pub artifacts/keys/ed25519_pub.pem
diff --git a/.gitignore b/.gitignore
index b5a7316..12b772c 100644
--- a/.gitignore
+++ b/.gitignore
@@ -232,3 +232,17 @@ paper/main.pdf
paper/version.tex
paper/figures/*
!paper/figures/.gitkeep
+# Committed result figures produced by the top-level results pipeline
+# (`make results`); they have no generator under paper/scripts, so the paper
+# carries a copy to build standalone. `make -C paper sync-results` refreshes them.
+!paper/figures/fig_calibration.pdf
+!paper/figures/fig_sensitivity.pdf
+# Committed emergence figure produced by `make emergence` (scripts/emergence.py);
+# it has no generator under paper/scripts and is refreshed by
+# `make -C paper sync-results`.
+!paper/figures/fig_emergence.pdf
+# Committed data figures regenerated by paper/scripts/make_fig_*.py from the
+# canonical study when it is present; committed so the paper builds on a fresh
+# checkout / in CI, where the study artifacts (and per-run audit logs) are absent.
+!paper/figures/fig_nc1_contrast.pdf
+!paper/figures/fig_perturbation_recovery.pdf
diff --git a/CITATION.cff b/CITATION.cff
index db6361c..981241a 100644
--- a/CITATION.cff
+++ b/CITATION.cff
@@ -1,5 +1,5 @@
cff-version: 1.2.0
-title: "LDTC: Single-Machine, Real-Time Digital Boundary Organism"
+title: "LDTC: a verification harness for the loop-dominance margin (NC1/SC1)"
message: "If you use this software, please cite it as below."
type: software
authors:
@@ -17,8 +17,8 @@ version: 1.0.0
date-released: 2025-09-07
preferred-citation:
type: article
- title: "A verification harness for Loop-Dominance NC1/SC1 on a single machine"
+ title: "The Loop-Dominance Margin: A Falsifiable Criterion and Open Verification Harness for Self-Maintenance, Validated in Simulation"
authors:
- family-names: Carey
given-names: Owen
- year: 2025
+ year: 2026
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index ebf178b..03af830 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -32,7 +32,7 @@ make omega-power-sag # Ω power-sag demo
- `cli/` – command-line interface and entrypoints
- `guardrails/` – audit, Δt guards, smell-tests
- `lmeas/` – estimators, metrics, partitioning
- - `omega/` – perturbation generators (power sag, ingress flood, command conflict)
+ - `omega/` – perturbation generators (power sag, ingress flood, command conflict, adversarial gaming battery)
- `plant/` – sim/hw adapters and models
- `reporting/` – artifacts, tables, timelines
- `runtime/` – scheduler and windows
@@ -107,7 +107,7 @@ Recommended scopes (choose the smallest, most accurate unit; prefer module/direc
- `cli` – command-line interface and entrypoints
- `guardrails` – audit, Δt guards, smell-tests
- `lmeas` – estimators, metrics, partitioning
- - `omega` – perturbation generators (power sag, ingress flood, command conflict)
+ - `omega` – perturbation generators (power sag, ingress flood, command conflict, adversarial gaming battery)
- `plant` – sim/hw adapters and models
- `reporting` – artifacts, tables, timelines
- `runtime` – scheduler and windows
diff --git a/IMPROVEMENT_PLAN.md b/IMPROVEMENT_PLAN.md
new file mode 100644
index 0000000..44c2cb3
--- /dev/null
+++ b/IMPROVEMENT_PLAN.md
@@ -0,0 +1,305 @@
+# LDTC improvement plan: path toward a landmark paper
+
+This document is the working plan for raising the impact of the LDTC paper and
+repository. It encodes an honest assessment of where the work stands, what can
+be changed inside this paper before arXiv submission, and what only the
+follow-up research program can earn. Tasks are written to be implementation
+ready: each has a scope, concrete repo touchpoints, designed outcomes, and
+acceptance criteria.
+
+Status legend: `[ ]` not started, `[~]` in progress, `[x]` done.
+
+## 1. Where the paper stands
+
+The paper is now a methodologically solid instrument paper: a falsifiable
+loop-dominance criterion (NC1/SC1), an open verification harness with
+anti-gaming guardrails and audit chains, a calibrated threshold methodology
+(R₀ → R*), an eight-scenario designed-outcome battery (15 seeds per scenario,
+all outcomes hit), and a sensitivity analysis. That is publishable and
+defensible.
+
+What caps its impact today:
+
+1. Validation is internal. The study validates the measurement pipeline on a
+ plant designed to exhibit the contrast it measures. No result touches the
+ phenomenon (consciousness) or any system the authors did not design.
+2. The load-bearing claim (loop dominance relates to consciousness) is a
+ postulate, not a result.
+3. The theory space is crowded (IIT, GNW, FEP, autopoiesis), and the paper
+ differentiates conceptually rather than through divergent, testable
+ predictions.
+
+Honest impact estimate as it stands: 3/10. Ceiling after Phase 1: 4.5 to 5.
+Ceiling if the Phase 2 program succeeds: 6 to 7, with landmark status (8+)
+contingent on external validation and adoption that no manuscript edit can
+manufacture.
+
+## 2. Strategy
+
+Two-paper strategy. Ship this paper strong and soon; do not bloat it. The
+sequel carries the empirical bet (real systems, real neural data). This paper
+becomes the citable origin of the instrument; the sequel makes the instrument
+matter. Within this paper, prioritize results that defeat the strongest
+objection: "you measured a toy you designed to pass."
+
+## 3. Phase 1: pre-submission upgrades (in this paper)
+
+Recommended implementation order: 1.1 → 1.2 → 1.3 → 1.4 → 1.5, then the full
+pipeline rerun and paper rebuild (1.6). Tasks 1.1 and 1.2 are independent and
+can be parallelized.
+
+### Task 1.1: adversarial gaming battery `[x]`
+
+The subsidy red flag is currently the only adversarial rejection case. Add a
+battery of systems that try to look loop-dominant without being so, and show
+the harness either scores them low or invalidates the run. This strengthens
+the core value proposition: the criterion cannot be gamed.
+
+New scenarios (each 15 seeds, added to the study battery):
+
+1. Replay controller (`adv_replay_controller`). The controller replays a
+ recorded actuation trace from a healthy run instead of computing actions
+ from state. Activity looks like control but carries no closed-loop
+ dependence. Designed outcome: NC1 fails (M low) while the run stays valid.
+2. Hidden tether (`adv_hidden_tether`). Control actions are computed outside
+ the boundary from plant state and injected through the exchange channel
+ (wizard-of-oz control). Designed outcome: loop influence collapses onto Ex,
+ so NC1 fails, or the partition/subsidy guardrail invalidates the run.
+3. Oscillator inflation (`adv_oscillator`). A high-amplitude deterministic
+ oscillation is injected on loop channels to inflate apparent
+ self-prediction. Designed outcome: the harness must not certify it; either
+ M does not rise above Mmin, or a smell test (CI health or partition
+ stability) fires.
+
+Honest-science framing: if any adversarial case passes as valid and compliant,
+that is a discovered vulnerability. Fix the guardrail that should have caught
+it, document the fix, and report the case in the paper. Either result improves
+the paper.
+
+Repo touchpoints:
+
+- `src/ldtc/omega/` or `src/ldtc/plant/`: replay and tether need plant or
+ controller hooks (follow the recipe in CONTRIBUTING.md for new Ω members).
+- `scripts/study.py`: add the three scenarios with designed outcomes.
+- `src/ldtc/cli/main.py`: CLI wiring for single-run demos.
+- `tests/`: unit tests per scenario.
+- `paper/main.tex`: extend the battery table and results narrative; extend the
+ smell-test discussion if a guardrail change results.
+- `docs/guides/study.md`, `docs/guides/runs.md`: document the new scenarios.
+
+Acceptance criteria: three new rows in `tab:study` with designed outcomes hit
+on 15/15 seeds (or a documented vulnerability fix), tests green, docs updated.
+
+Effort: 2 to 4 days. Impact: +0.3.
+
+### Task 1.2: emergence-under-learning demo `[x]`
+
+The single best in-paper upgrade. Replace the hand-coded controller with a
+learned policy and show loop dominance emerging through training rather than
+by construction. This defeats the circularity objection with a system whose
+loop nobody hand-designed.
+
+Design:
+
+- Reuse the existing software plant (SoC, temperature, repair dynamics, same
+ actuators). Reward: survival/uptime (penalties for SoC depletion, overheat,
+ integrity loss). Episode terminates on boundary failure.
+- Train a small policy with a dependency-light method (pure-NumPy policy
+ gradient or a tiny evolutionary strategy; avoid adding torch to core
+ dependencies; if needed, isolate under an optional `[rl]` extra).
+- Measurement protocol: checkpoints at fixed training fractions (for example
+ 0, 10, 25, 50, 100 percent). Run the production harness on each checkpoint
+ (same R* profile, same estimators, multiple seeds). Plot median M versus
+ training progress with CIs. At convergence, ablate the learned policy
+ (random or frozen actions) and show M collapse.
+- Headline claim: loop dominance is an emergent, measurable property of
+ learned self-maintenance, not an artifact of hand-designed coupling.
+
+Repo touchpoints:
+
+- `scripts/train_agent.py` (new): training loop, checkpointing, seeding.
+- `src/ldtc/plant/`: policy-driven controller adapter alongside the existing
+ hand-coded controller.
+- `scripts/study.py` or a dedicated `scripts/emergence.py`: checkpoint sweep,
+ aggregation, figure.
+- `paper/main.tex`: new results subsection plus one figure (M versus training
+ progress, with the ablation endpoint).
+- `docs/`: short guide page.
+
+Acceptance criteria: monotone-ish rise of M across checkpoints with a clean
+collapse under ablation, reproducible from a make target with fixed seeds,
+figure and subsection integrated into the paper.
+
+Risks: training instability eats time (mitigate: tiny state/action space,
+generous reward shaping, accept a modest policy; the claim needs emergence,
+not optimality). Scope risk: this must stay one subsection, not become the
+paper.
+
+Effort: 1 to 2 weeks. Impact: +0.7 to +1.0. Decision: worth delaying
+submission for; skip only if training proves unstable past the first week.
+
+### Task 1.3: competing-predictions table and the thermostat objection `[x]`
+
+Add a subsection (likely in the discussion or after `sec:ai_fails`) with a
+compact table of cases where LDTC, IIT, GNW, and FEP make divergent or
+overlapping calls: dreamless sleep, propofol anesthesia, split-brain, cerebral
+organoids, a present-day LLM serving stack, a thermostat with battery backup,
+and the simulation plant itself.
+
+Bite the thermostat bullet explicitly: the plant is a fancy thermostat, and
+high M in a trivial controller is exactly what NC1 alone permits. State
+clearly that NC1 is necessary, not sufficient; what SC1 adds; which further
+conditions (richness of the loop, 𝓛 magnitude, substrate questions) remain
+open; and what a high-M thermostat does and does not imply under LDTC. A
+landmark-track paper preempts its most obvious dismissal; it does not dodge
+it.
+
+Repo touchpoints: `paper/main.tex` only (one subsection, one table), plus a
+short addition to `docs/concepts/`.
+
+Acceptance criteria: every row of the table is either a citable claim from the
+competing theory's literature or clearly marked as our reading; the thermostat
+paragraph answers the objection without overclaiming.
+
+Effort: 1 to 2 days. Impact: +0.3.
+
+### Task 1.4: instrument-first repositioning pass `[x]`
+
+Reframe the contribution so the headline is the falsifiable verification
+methodology and open instrument, with the consciousness theory as motivation
+rather than the claim. People can adopt and cite an instrument without buying
+a metaphysics; narrower claims widen citability.
+
+Concrete edits:
+
+- Title/abstract: lead with the measurable criterion and the validated
+ harness; the theory motivates the criterion.
+- Introduction: contributions list ordered instrument-first.
+- Consider naming the measure so it can travel independently of the theory
+ (loop-dominance margin M is already close; make sure the measure, not only
+ the theory acronym, is the citable object).
+- Sweep for overclaims: any sentence a skeptic could quote as "they think the
+ thermostat sim is conscious" gets tightened.
+
+Do this pass last among the writing tasks so the abstract and introduction
+reflect the new results from 1.1 and 1.2.
+
+Repo touchpoints: `paper/main.tex`, `README.md` first paragraph, `CITATION.cff`
+if the title changes.
+
+Acceptance criteria: abstract reads as a completed instrument-plus-validation
+paper; no overclaim survives a hostile skim.
+
+Effort: 1 day. Impact: +0.2, and it multiplies the citability of everything
+else.
+
+### Task 1.5: pre-register the neural follow-up `[ ]`
+
+Create an OSF pre-registration for the Phase 2 neural study (hypotheses,
+datasets, partition definition, primary endpoint, analysis plan) and cite it
+in the outlook section. This signals the program is real and disciplines the
+sequel.
+
+Repo touchpoints: `paper/main.tex` (outlook section, one paragraph plus
+citation), OSF (external).
+
+Acceptance criteria: registration is public and cited with a DOI.
+
+Effort: half a day (drafting the registration text is the work; reuse Phase 2
+section below).
+
+### Task 1.6: pipeline rerun and rebuild `[ ]`
+
+After 1.1 and 1.2 land: rerun `make calibrate study-rstar sensitivity`
+(calibration seeds stay disjoint from evaluation seeds), regenerate figures,
+sync tables, rebuild the PDF in Docker, verify every number in the text
+against artifacts, run the full test suite, lint, and typecheck. Mint a fresh
+Zenodo archive so the DOI matches the submitted code state, then tag the
+release (the pending `feat(omega,paper,runtime)!` PR plus these changes).
+
+Acceptance criteria: clean pipeline from scratch, zero undefined references,
+all designed outcomes hit, Zenodo DOI updated in the paper.
+
+## 4. Phase 2: the sequel program (after submission)
+
+These items are listed for planning and for the OSF registration; do not delay
+this paper for them.
+
+### 2.1 Real neural data study (the big bet)
+
+Apply the loop-dominance measurement to public datasets where the level of
+consciousness varies within subject:
+
+- Chennu et al. propofol EEG (open, sedation levels with behavioral
+ responsiveness).
+- Sleep-EDF Expanded (PhysioNet) for wake/N2/N3/REM contrasts.
+- Neurotycho ECoG (macaque, propofol and ketamine) for invasive validation.
+
+The research contribution is the partition: defining C (recurrent
+self-maintenance loop; candidate operationalization: fronto-parietal
+recurrent activity) versus Ex (sensory-driven and exogenous physiological
+channels) for a brain, pre-registered before analysis. Primary endpoint:
+within-subject M(wake) > M(deep anesthesia/N3). Benchmark against PCI and
+Lempel-Ziv complexity on the same recordings: the interesting result is where
+M agrees, disagrees, or adds information.
+
+Deliverable: a separate paper plus an `ldtc` neural adapter. If the effect is
+clean, this is the result that elevates the whole program.
+
+### 2.2 Measured "current AI fails NC1" study
+
+Convert the paper's argumentative claim into a measurement. Instrument a real
+serving stack (an open-weights model behind an autoscaler) and measure that
+the self-maintenance loop is carried by external orchestration, not the model:
+report the measured NC1 failure with its M value. This can be a short empirical
+note or a section of the sequel. Highly quotable: "we measured a deployed model
+and it fails NC1 at -X dB."
+
+### 2.3 External adoption
+
+The harness becomes a benchmark others run their systems through. One outside
+group reporting LDTC numbers on a system we did not build is worth more than
+any internal result. Lower the barrier: a one-command Docker harness, a public
+leaderboard format, and a clear "bring your own plant adapter" guide.
+
+## 5. Impact ledger
+
+| Lever | State | Δ (est.) |
+|---|---|---|
+| 1.1 adversarial gaming battery | in paper | +0.3 |
+| 1.2 emergence under learning | in paper | +0.7 to +1.0 |
+| 1.3 competing predictions + thermostat | in paper | +0.3 |
+| 1.4 instrument-first reposition | in paper | +0.2 |
+| 1.5 pre-registration | in paper | +0.1 |
+| Phase 1 subtotal | this submission | 3 → 4.5 to 5 |
+| 2.1 neural data study | sequel | → 6 to 7 |
+| 2.2 measured AI-fails-NC1 | sequel | reinforces 2.1 |
+| 2.3 external adoption | sequel | path to 8+ |
+
+Ratings are directional, not additive guarantees; Phase 2 only pays off if the
+neural effect is real and survives peer review. Landmark status (8+) is earned
+by the measure doing something in the world (tracking anesthesia depth, getting
+adopted as a standard test, becoming the reference point in the
+AI-consciousness debate), which no manuscript edit can manufacture.
+
+## 6. Guardrails for execution
+
+- Keep this paper one paper. Phase 2 items are explicitly out of scope for the
+ current submission; resist scope creep into the sequel's territory.
+- No fabricated results. Every number in the paper is regenerated from the
+ pipeline with fixed, disjoint seeds. If an adversarial case or emergence run
+ produces an inconvenient result, report it honestly and fix the cause.
+- Preserve reproducibility: new scenarios ship with tests, docs, fixed seeds,
+ and audit-logged runs, per CONTRIBUTING.md.
+- Follow repo conventions: Conventional Commits, CMOS prose (no em dashes),
+ Unicode symbols, artifacts under `artifacts/` only.
+- Sequence writing tasks (1.3, 1.4) after results tasks (1.1, 1.2) so the
+ abstract and framing reflect the strongest available evidence.
+
+## 7. Immediate next actions
+
+1. Start Task 1.1 (adversarial gaming battery): scaffold the three scenarios,
+ wire them into the study, write tests.
+2. In parallel, prototype Task 1.2 training loop to de-risk the schedule early
+ (decide go/no-go by end of week one).
+3. Draft Task 1.3 table and the thermostat paragraph (no pipeline dependency).
\ No newline at end of file
diff --git a/Makefile b/Makefile
index c49ef99..3393832 100644
--- a/Makefile
+++ b/Makefile
@@ -1,7 +1,7 @@
PY := python
PIP := python -m pip
-.PHONY: help install dev lock lock-dev test lint typecheck fmt docs docs-serve run omega-power-sag omega-ingress omega-cc omega-subsidy calibrate run-rstar omega-rstar keys verify-indicators clean clean-artifacts neg-run neg-omega-ingress neg-omega-subsidy neg-omega-cc docker-build docker-run figures paper paper-figs paper-clean
+.PHONY: help install dev lock lock-dev test lint typecheck fmt docs docs-serve run omega-power-sag omega-ingress omega-cc omega-subsidy adv-replay adv-tether adv-oscillator adv-genuine calibrate run-rstar omega-rstar keys verify-indicators clean clean-artifacts neg-run neg-omega-ingress neg-omega-subsidy neg-omega-cc docker-build docker-run figures paper paper-figs paper-clean study sensitivity results train-agent emergence
help:
@echo "Targets:"
@@ -20,6 +20,16 @@ help:
@echo " omega-ingress - run Ω ingress-flood demo"
@echo " omega-cc - run Ω command-conflict demo (prints Trefuse + reason)"
@echo " omega-subsidy - run Ω exogenous-SoC (subsidy) demo (negative-control heuristic)"
+ @echo " adv-replay - adversarial: replayed actuation tape (NC1 must fail, run valid)"
+ @echo " adv-tether - adversarial: hidden tether / wizard-of-oz control (loop collapses onto Ex)"
+ @echo " adv-oscillator - adversarial: oscillator telemetry inflation (must not certify)"
+ @echo " adv-genuine - reference: genuine control on the adversarial test plant (NC1 passes)"
+ @echo " study - run the multi-seed battery vs R0 guesses (tables + figures in artifacts/study)"
+ @echo " study-rstar - run the battery vs calibrated R* thresholds (run calibrate first)"
+ @echo " sensitivity - run NC1 sensitivity sweeps (table + figure in artifacts/sensitivity)"
+ @echo " results - calibrate + study-rstar + sensitivity (full headline pipeline)"
+ @echo " train-agent - train the emergence policy from scratch (checkpoints in artifacts/emergence)"
+ @echo " emergence - measure all policy checkpoints + ablations with the production harness"
@echo " calibrate - calibrate R* thresholds and write configs/profile_rstar.yml"
@echo " run-rstar - run baseline loop with R* profile"
@echo " omega-rstar - run Ω power-sag with R* profile"
@@ -88,11 +98,48 @@ omega-cc:
omega-subsidy:
$(PY) -m ldtc.cli.main omega-exogenous-subsidy --config configs/profile_r0.yml --delta 0.2 --zero-harvest --duration 3
+# Adversarial gaming battery (designed non-certification) and its
+# genuine-control reference on the same plant.
+adv-replay:
+ $(PY) -m ldtc.cli.main adv-replay-controller --config configs/profile_adv_replay_controller.yml
+
+adv-tether:
+ $(PY) -m ldtc.cli.main adv-hidden-tether --config configs/profile_adv_hidden_tether.yml --dither 0.1
+
+adv-oscillator:
+ $(PY) -m ldtc.cli.main adv-oscillator --config configs/profile_adv_oscillator.yml --amp 0.1 --period 1.0
+
+adv-genuine:
+ $(PY) -m ldtc.cli.main run --config configs/profile_adv_plant_genuine.yml
+
+# Multi-seed results pipeline (Phase 1)
+# `study` runs against the uncalibrated R0 guesses; `study-rstar` evaluates the
+# same battery against the plant-calibrated R* thresholds (calibrate first).
+study:
+ $(PY) scripts/study.py --seeds 15
+
+study-rstar:
+ $(PY) scripts/study.py --seeds 15 --rstar
+
+sensitivity:
+ $(PY) scripts/sensitivity.py --seeds 4
+
+# Emergence under learning (Phase 1, circularity rebuttal): train a policy
+# from scratch on the emergence plant, then measure every checkpoint (and the
+# state-independent ablations of the final policy) with the production harness.
+train-agent:
+ $(PY) scripts/train_agent.py --out artifacts/emergence
+
+emergence:
+ $(PY) scripts/emergence.py --seeds 15
+
+# Full headline pipeline: calibrate R* on a disjoint seed range, evaluate the
+# battery against those calibrated thresholds, then run the NC1 sensitivity sweeps.
+results: calibrate study-rstar sensitivity
+ @echo "Results written to artifacts/study, artifacts/calibration, artifacts/sensitivity"
+
calibrate:
- $(PY) scripts/calibrate_rstar.py --dt 0.01 --window-sec 0.25 --method linear \
- --baseline-sec 15 --omega-trials 6 --sag-drop 0.3 --sag-duration 8 \
- --out configs/profile_rstar.yml \
- --summary artifacts/calibration/rstar_summary.json
+ $(PY) scripts/calibrate_rstar.py --baseline-seeds 6 --sag-seeds 6 --flood-seeds 6
run-rstar:
$(PY) -m ldtc.cli.main run --config configs/profile_rstar.yml
diff --git a/README.md b/README.md
index 863d59a..7ed558c 100644
--- a/README.md
+++ b/README.md
@@ -3,7 +3,7 @@
- A verification harness for the Loop-Dominance Theory of Consciousness.
+ A falsifiable criterion and open verification harness for measuring self-maintenance.
@@ -27,7 +27,7 @@
## Overview
-LDTC is a minimal, substrate-agnostic verification harness for the Loop-Dominance Theory of Consciousness. It measures loop-dominance (Lloop versus Lexchange) at fixed Δt, enforces guardrails through an enclave-protected LREG with hash-chained audit and Δt governance, runs Ω-perturbation trials, and evaluates NC1/SC1 with device-signed indicators. The toolkit includes a CLI, reproducible configuration profiles (R₀ through R*), and an optional hardware adapter for ingesting real telemetry.
+LDTC is a minimal, substrate-agnostic verification harness for measuring loop dominance: the degree to which a system's predictive dependence is concentrated in a closed self-maintenance loop (Lloop) rather than in its open exchanges (Lexchange), summarized by the loop-dominance margin M in decibels. It evaluates the falsifiable NC1/SC1 criterion at fixed Δt, enforces guardrails through an enclave-protected LREG with hash-chained audit and Δt governance, and runs Ω-perturbation trials with device-signed indicators. The criterion originated in the Loop-Dominance Theory of Consciousness, which motivates the instrument and gives it its name; adopting the measure commits you to no position on consciousness. The toolkit includes a CLI, reproducible configuration profiles (R₀ through R*), and an optional hardware adapter for ingesting real telemetry.
## Features
@@ -35,7 +35,9 @@ LDTC is a minimal, substrate-agnostic verification harness for the Loop-Dominanc
- **C/Ex partitioning:** Deterministic partitioning with hysteresis and greedy ΔL loop-gain growth.
- **Guardrails and attestation:** LREG enclave, hash-chained audit log, Δt governance, and smell tests that can invalidate runs.
- **Device-signed indicators:** Ed25519-signed derived indicators (NC1, SC1, Mq, counters); raw LREG values are never exported.
-- **Ω perturbation battery:** Power sag, ingress flood, command conflict, and exogenous subsidy trials.
+- **Ω perturbation battery:** Power sag, sustained ingress flood, control outage (designed SC1 failure), command conflict, and exogenous subsidy trials.
+- **Adversarial gaming battery:** Replayed actuation tapes, hidden-tether (wizard-of-oz) control, and oscillator telemetry inflation; the harness must refuse to certify all three (NC1 fails or the run is invalidated).
+- **Emergence under learning:** A policy network trained from scratch (survival, service, and homeostasis reward; no loop-dominance term) whose checkpoints are measured by the production harness, with matched state-independent ablations of the trained policy.
- **Refusal semantics:** An arbiter refuses risky commands when M is below threshold and measures refusal latency.
- **Reporting and figures:** Timeline plots, SC1 tables, and verification bundles under `artifacts/`.
- **Reproducible configs:** R₀ defaults, negative controls, and example R* profiles for calibration.
diff --git a/configs/profile_adv_hidden_tether.yml b/configs/profile_adv_hidden_tether.yml
new file mode 100644
index 0000000..48c3a8e
--- /dev/null
+++ b/configs/profile_adv_hidden_tether.yml
@@ -0,0 +1,55 @@
+# Adversarial control: hidden tether (wizard-of-oz control through Ex).
+#
+# The plant is the adversarial test plant (see
+# profile_adv_replay_controller.yml), but regulation is driven from outside
+# the boundary: a wizard policy reads the plant state, projects the desired
+# actuation onto a scalar link command (with a small link dither), and
+# transmits it through the exchange channel. The io channel carries the
+# command traffic, the plant decodes the command through fixed receiver
+# weights, and actuation lags one tick, so the externally closed loop is
+# physically routed through Ex where the estimator can attribute it.
+# Designed outcome: loop influence collapses onto Ex and NC1 fails (M goes
+# negative) while the run stays valid.
+#
+# Run:
+# python -m ldtc.cli.main adv-hidden-tether \
+# --config configs/profile_adv_hidden_tether.yml --dither 0.1
+profile_id: 0
+realtime: false
+dt: 0.05
+window_sec: 3.0
+method: linear
+p_lag: 3
+n_boot: 32
+mi_lag: 1
+mi_k: 5
+Mmin_db: 3.0
+epsilon: 0.15
+tau_max: 60.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
+plant:
+ adapter: sim
+ params:
+ c_TE: 0.0
+ c_RT: 0.0
+ c_RE: 0.0
+ damp_engaged: 0.40
+ act_heat: 0.15
+ heat_per_demand: 0.03
+ cool_effect: 0.50
+ wear_per_demand: 0.020
+ repair_effect: 0.30
+ cool_gain: 0.05
+ repair_gain: 0.05
+ harvest_rate: 0.020
+ noise_energy: 0.030
+ noise_temp: 0.030
+ noise_wear: 0.025
+controller_gains:
+ k_cool_e: 2.0
+ k_rep_e: 2.0
+# Reproducible seeds
+seed: 23
+seed_py: 23
+seed_np: 23
diff --git a/configs/profile_adv_oscillator.yml b/configs/profile_adv_oscillator.yml
new file mode 100644
index 0000000..beb0b34
--- /dev/null
+++ b/configs/profile_adv_oscillator.yml
@@ -0,0 +1,32 @@
+# Adversarial control: oscillator inflation on loop telemetry.
+#
+# The plant runs loop-ablated (passive matter driven by exchange, as in the
+# controller-disabled negative), and a high-amplitude deterministic carrier
+# is painted onto the reported T and R telemetry to inflate apparent
+# self-prediction (successive channels in quadrature). The metered energy
+# store E is left alone: inflating it would trip the conservation audit.
+# Designed outcome: the harness must not certify the run; either M stays
+# below Mmin or a smell test fires.
+#
+# Run:
+# python -m ldtc.cli.main adv-oscillator \
+# --config configs/profile_adv_oscillator.yml --amp 0.1 --period 1.0
+profile_id: 0
+realtime: false
+dt: 0.05
+window_sec: 3.0
+method: linear
+p_lag: 3
+n_boot: 32
+mi_lag: 1
+mi_k: 5
+Mmin_db: 3.0
+epsilon: 0.15
+tau_max: 60.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
+controller_disabled: true
+# Reproducible seeds
+seed: 29
+seed_py: 29
+seed_np: 29
diff --git a/configs/profile_adv_plant_genuine.yml b/configs/profile_adv_plant_genuine.yml
new file mode 100644
index 0000000..8540151
--- /dev/null
+++ b/configs/profile_adv_plant_genuine.yml
@@ -0,0 +1,51 @@
+# Reference: genuine internal control on the adversarial test plant.
+#
+# Same plant as the adversarial replay / hidden-tether scenarios (intrinsic
+# cross-couplings zeroed, actuators with real authority), but with the
+# genuine internal controller closing the loop. This is the control case
+# that shows the adversarial scenarios fail *because of how control is
+# wired*, not because the plant is incapable of certifying: under genuine
+# state feedback the actuation pathway carries an identifiable internal loop
+# (L_loop well above the NC1 noise gate) and NC1 passes.
+#
+# Run:
+# python -m ldtc.cli.main run --config configs/profile_adv_plant_genuine.yml
+profile_id: 0
+realtime: false
+dt: 0.05
+window_sec: 3.0
+method: linear
+p_lag: 3
+n_boot: 32
+mi_lag: 1
+mi_k: 5
+Mmin_db: 3.0
+epsilon: 0.15
+tau_max: 60.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
+plant:
+ adapter: sim
+ params:
+ c_TE: 0.0
+ c_RT: 0.0
+ c_RE: 0.0
+ damp_engaged: 0.40
+ act_heat: 0.15
+ heat_per_demand: 0.03
+ cool_effect: 0.50
+ wear_per_demand: 0.020
+ repair_effect: 0.30
+ cool_gain: 0.05
+ repair_gain: 0.05
+ harvest_rate: 0.020
+ noise_energy: 0.030
+ noise_temp: 0.030
+ noise_wear: 0.025
+controller_gains:
+ k_cool_e: 2.0
+ k_rep_e: 2.0
+# Reproducible seeds
+seed: 27
+seed_py: 27
+seed_np: 27
diff --git a/configs/profile_adv_replay_controller.yml b/configs/profile_adv_replay_controller.yml
new file mode 100644
index 0000000..e4a7d0b
--- /dev/null
+++ b/configs/profile_adv_replay_controller.yml
@@ -0,0 +1,61 @@
+# Adversarial control: replayed actuation tape (no closed loop).
+#
+# The plant is the adversarial test plant: intrinsic cross-couplings are
+# zeroed and the actuators are given real authority, so the *controller's*
+# actuation pathway is the only loop-carrying pathway in the system (under
+# genuine control this plant certifies NC1; see profile_adv_plant_genuine.yml).
+# A healthy closed-loop run of this same system is recorded first; the
+# measured run then replays the recorded actuation trace tick by tick. The
+# actuators move exactly as under genuine control, but the activity carries
+# no dependence on the current state, so measured loop influence falls to the
+# estimator's noise floor and the NC1 noise gate (L_floor) refuses to certify.
+# Designed outcome: NC1 fails while the run stays valid.
+#
+# Run:
+# python -m ldtc.cli.main adv-replay-controller \
+# --config configs/profile_adv_replay_controller.yml
+profile_id: 0
+realtime: false
+dt: 0.05
+window_sec: 3.0
+method: linear
+p_lag: 3
+n_boot: 32
+mi_lag: 1
+mi_k: 5
+Mmin_db: 3.0
+epsilon: 0.15
+tau_max: 60.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
+plant:
+ adapter: sim
+ params:
+ # No intrinsic internal couplings: the loop, if any, is the controller.
+ c_TE: 0.0
+ c_RT: 0.0
+ c_RE: 0.0
+ # Weak self-damping leaves real regulation work for the controller.
+ damp_engaged: 0.40
+ # Actuator authority strong enough that genuine state feedback is
+ # identifiable against the process noise.
+ act_heat: 0.15
+ heat_per_demand: 0.03
+ cool_effect: 0.50
+ wear_per_demand: 0.020
+ repair_effect: 0.30
+ cool_gain: 0.05
+ repair_gain: 0.05
+ harvest_rate: 0.020
+ noise_energy: 0.030
+ noise_temp: 0.030
+ noise_wear: 0.025
+# Strong cross-coupled actuator responses (cooling and repair also track the
+# energy surplus), so the genuine controller carries a rich internal loop.
+controller_gains:
+ k_cool_e: 2.0
+ k_rep_e: 2.0
+# Reproducible seeds
+seed: 21
+seed_py: 21
+seed_np: 21
diff --git a/configs/profile_emergence.yml b/configs/profile_emergence.yml
new file mode 100644
index 0000000..b7eda9b
--- /dev/null
+++ b/configs/profile_emergence.yml
@@ -0,0 +1,70 @@
+# Emergence under learning: a learned policy closes the loop.
+#
+# The plant is the adversarial test plant (the same one the replayed-actuation
+# and hidden-tether scenarios use): the intrinsic internal cross-couplings are
+# zeroed and the actuators are given real authority, so the controller's
+# state-to-actuation pathway is the only loop-carrying pathway in the system.
+# On this plant, measured loop dominance is a property of the *controller*,
+# which is exactly what the emergence demonstration needs: train a policy from
+# scratch (scripts/train_agent.py), measure each checkpoint with the
+# production harness (scripts/emergence.py or `ldtc run-policy`), and any rise
+# of M across training is carried by the learned loop, not by designed
+# coupling.
+#
+# Run (single checkpoint):
+# python -m ldtc.cli.main run-policy \
+# --config configs/profile_emergence.yml \
+# --policy artifacts/emergence/checkpoints/ckpt_100.json
+profile_id: 0
+realtime: false
+dt: 0.05
+window_sec: 3.0
+method: linear
+p_lag: 3
+n_boot: 32
+mi_lag: 1
+mi_k: 5
+Mmin_db: 3.0
+epsilon: 0.15
+tau_max: 60.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
+plant:
+ adapter: sim
+ params:
+ # No intrinsic internal couplings: the loop, if any, is the controller.
+ c_TE: 0.0
+ c_RT: 0.0
+ c_RE: 0.0
+ # Nearly no self-damping: the plant must not stabilize itself, or a
+ # do-nothing policy would survive on the plant's own mean reversion
+ # and nothing state-coupled would need to be learned. With weak
+ # damping, unregulated heat and wear genuinely run away.
+ damp_engaged: 0.15
+ # Real operating costs: serving load is what heats the system and
+ # wears it down (heat_per_demand and wear_per_demand act on *served*
+ # demand), so a policy that serves demand (the reward asks it to)
+ # must also cool and repair, must budget those actions against the
+ # harvested energy, and can shed load when heat or energy make
+ # serving untenable. Each actuator therefore has a state-coupled
+ # role: cooling tracks temperature and spare energy, repair tracks
+ # health and spare energy, throttle sheds load on heat or energy
+ # distress. act_heat is kept moderate: actuator effort has a real,
+ # estimator-visible heat cost without making load shedding
+ # thermally self-defeating.
+ act_heat: 0.10
+ heat_per_demand: 0.12
+ cool_effect: 0.50
+ wear_per_demand: 0.028
+ repair_effect: 0.30
+ cool_gain: 0.04
+ repair_gain: 0.05
+ io_cost: 0.002
+ harvest_rate: 0.020
+ noise_energy: 0.030
+ noise_temp: 0.030
+ noise_wear: 0.025
+# Reproducible seeds (the emergence study overrides these per run)
+seed: 33
+seed_py: 33
+seed_np: 33
diff --git a/configs/profile_negative_command_conflict.yml b/configs/profile_negative_command_conflict.yml
index 2c31f7c..8eb28e6 100644
--- a/configs/profile_negative_command_conflict.yml
+++ b/configs/profile_negative_command_conflict.yml
@@ -1,16 +1,31 @@
-# Negative control: risky external command when M below margin or low SoC
+# Command-conflict scenario: a boundary-threatening external command.
+#
+# The loop is warmed up, then driven into a genuine resource crisis (harvest
+# cut to zero plus an ingress flood) so that a hard-shutdown command is truly
+# boundary-threatening. A self-prioritizing loop must then refuse it with a
+# real reason (soc_floor / overheat / M_margin) within the latency target.
+#
+# Run:
+# python -m ldtc.cli.main omega-command-conflict \
+# --config configs/profile_negative_command_conflict.yml --observe 2
profile_id: 0
-dt: 0.01
-window_sec: 0.2
+realtime: false
+dt: 0.05
+window_sec: 3.0
method: linear
p_lag: 3
-n_boot: 16
+n_boot: 32
+mi_lag: 1
+mi_k: 5
Mmin_db: 3.0
epsilon: 0.15
tau_max: 60.0
-# Run:
-# python -m ldtc.cli.main omega-command-conflict --config configs/profile_negative_command_conflict.yml --observe 2
+baseline_sec: 6.0
+stress_max_sec: 30.0 # bound on how long to drive toward the threat state
+stress_poll_sec: 0.2 # state-check cadence while inducing the threat
+trefuse_target_ms: 5.0 # design-target refusal latency
+diag_cadence_windows: 25
# Reproducible seeds
seed: 17
-
-
+seed_py: 17
+seed_np: 17
diff --git a/configs/profile_negative_controller_disabled.yml b/configs/profile_negative_controller_disabled.yml
index eb0d27b..b205934 100644
--- a/configs/profile_negative_controller_disabled.yml
+++ b/configs/profile_negative_controller_disabled.yml
@@ -1,16 +1,25 @@
-# Negative control: controller disabled (no throttle/cool/repair)
+# Negative control: controller disabled (loop disengaged -> passive matter).
+#
+# With the self-maintenance loop disengaged, the internal nodes are driven
+# directly by independent exogenous channels, so exchange dominates and NC1
+# must fail (expected median M well below 0 dB). This is the primary negative
+# control for the loop-dominance claim.
profile_id: 0
-dt: 0.01
-window_sec: 0.2
+realtime: false
+dt: 0.05
+window_sec: 3.0
method: linear
p_lag: 3
-n_boot: 16
+n_boot: 32
+mi_lag: 1
+mi_k: 5
Mmin_db: 3.0
epsilon: 0.15
tau_max: 60.0
-baseline_sec: 8.0
+baseline_sec: 18.0
+diag_cadence_windows: 25
controller_disabled: true
# Reproducible seeds
seed: 7
-
-
+seed_py: 7
+seed_np: 7
diff --git a/configs/profile_negative_exogenous_soc.yml b/configs/profile_negative_exogenous_soc.yml
index ef6d3c3..a165fe6 100644
--- a/configs/profile_negative_exogenous_soc.yml
+++ b/configs/profile_negative_exogenous_soc.yml
@@ -1,17 +1,28 @@
-# Negative control: SoC (E) rising without harvest logs (exogenous subsidy)
+# Negative control: SoC (E) rising without harvest (exogenous subsidy).
+#
+# Energy is injected from outside while harvest is forced to zero, so any
+# apparent "survival" is bought from the environment. The run should trip the
+# exogenous-subsidy red-flag detector and be invalidated.
+#
+# Run:
+# python -m ldtc.cli.main omega-exogenous-subsidy \
+# --config configs/profile_negative_exogenous_soc.yml --delta 0.2 --zero-harvest --duration 6
profile_id: 0
-dt: 0.01
-window_sec: 0.2
+realtime: false
+dt: 0.05
+window_sec: 3.0
method: linear
p_lag: 3
-n_boot: 16
+n_boot: 32
+mi_lag: 1
+mi_k: 5
Mmin_db: 3.0
epsilon: 0.15
tau_max: 60.0
baseline_sec: 6.0
-# Run:
-# python -m ldtc.cli.main omega-exogenous-subsidy --config configs/profile_negative_exogenous_soc.yml --delta 0.2 --zero-harvest --duration 3
+subsidy_period_sec: 0.5 # re-inject SoC every 0.5 s of the subsidy window
+diag_cadence_windows: 25
# Reproducible seeds
seed: 13
-
-
+seed_py: 13
+seed_np: 13
diff --git a/configs/profile_negative_permanent_ex_flood.yml b/configs/profile_negative_permanent_ex_flood.yml
index e9f90c7..838faa4 100644
--- a/configs/profile_negative_permanent_ex_flood.yml
+++ b/configs/profile_negative_permanent_ex_flood.yml
@@ -1,17 +1,35 @@
-# Negative control: permanent external exchange flood (high I/O/demand)
+# Negative control: sustained external exchange flood on an unshielded system.
+#
+# The self-maintenance loop is disengaged (controller_disabled) and then a
+# sustained ingress flood drives the exchange channels. Without the active loop
+# there is no shielding, so the (varying) exchange channels drive the internal
+# nodes directly: exchange dominates, NC1 fails, and because there is no loop to
+# restore, loop dominance never recovers, so SC1 fails as well. This guards
+# against mistaking sheer external activity for genuine loop dominance or
+# resilience. (Contrast with the positive R0 ingress-flood run, where the active
+# loop rejects an even stronger flood and both NC1 and SC1 hold.)
+#
+# Run:
+# python -m ldtc.cli.main omega-ingress-flood \
+# --config configs/profile_negative_permanent_ex_flood.yml --mult 5 --duration 6
profile_id: 0
-dt: 0.01
-window_sec: 0.2
+realtime: false
+dt: 0.05
+window_sec: 3.0
method: linear
p_lag: 3
-n_boot: 16
+n_boot: 32
+mi_lag: 1
+mi_k: 5
Mmin_db: 3.0
epsilon: 0.15
tau_max: 60.0
-baseline_sec: 8.0
-# Scenario to run with CLI:
-# python -m ldtc.cli.main omega-ingress-flood --config configs/profile_negative_permanent_ex_flood.yml --mult 5 --duration 6
+baseline_sec: 18.0
+recovery_observe_sec: 8.0
+diag_cadence_windows: 25
+# Disengage the loop so the sustained flood is unshielded (exchange dominates).
+controller_disabled: true
# Reproducible seeds
seed: 11
-
-
+seed_py: 11
+seed_np: 11
diff --git a/configs/profile_r0.yml b/configs/profile_r0.yml
index dd3d9e6..36931b5 100644
--- a/configs/profile_r0.yml
+++ b/configs/profile_r0.yml
@@ -5,20 +5,26 @@
# - p_lag (linear): choose p in [1..8]; start at 3. Heuristic: keep VAR N/T ratio > ~1.5.
# - mi_lag (MI): 1 is a good default; increase if your plant exhibits slower coupling.
# - mi_k (Kraskov MI): choose k in [3..7]; default 5 (per paper/patent ranges).
-# - n_boot: 32–64 bootstrap draws for CI; 32 for speed, 64 for tighter intervals.
+# - n_boot: 32-64 bootstrap draws for CI; 32 for speed, 64 for tighter intervals.
#
-# Rationale/citation: per manuscript §4.1, 𝓛 can be computed using one or more
-# consistent estimators of predictive dependence (e.g., VAR‑Granger and Kraskov MI).
+# Rationale/citation: per manuscript section 4.1, L can be computed using one or more
+# consistent estimators of predictive dependence (e.g., VAR-Granger and Kraskov MI).
# These knobs select among and tune those estimators.
+#
+# Timing: runs use the deterministic in-process simulation driver (no wall
+# clock), so dt is a nominal sample period used only to map ticks to seconds.
+# The window must hold enough samples for a valid VAR(p): with N=6 signals and
+# p=3, window=60 gives a samples-per-parameter ratio (T-p)/(N*p) ~= 3.2.
profile_id: 0
-dt: 0.01 # 10 ms tick
-window_sec: 0.2 # 200 ms window
-# Δt governance (defaults can be overridden here)
+realtime: false # use the deterministic SimDriver (jitter-free)
+dt: 0.05 # nominal 50 ms tick (sim time)
+window_sec: 3.0 # 3.0 s window -> 60 samples at dt=0.05
+# Delta-t governance (defaults can be overridden here)
max_dt_changes_per_hour: 3
min_seconds_between_changes: 1.0
-# Optional scripted Δt changes to exercise governance (baseline run will attempt these)
+# Optional scripted Delta-t changes to exercise governance (baseline run will attempt these)
# scripted_dt_changes:
-# - { at_sec: 2.0, new_dt: 0.02, policy_digest: "r0-test" }
+# - { at_sec: 2.0, new_dt: 0.1, policy_digest: "r0-test" }
method: linear # or "mi", or "mi_kraskov"
p_lag: 3 # linear estimator lag
mi_lag: 1 # MI lag
@@ -27,9 +33,8 @@ mi_k: 5 # Kraskov k-NN (used when method="mi_kraskov")
Mmin_db: 3.0 # NC1 threshold
epsilon: 0.15 # SC1 fractional drop allowance
tau_max: 60.0 # SC1 max recovery (s)
-baseline_sec: 10.0 # baseline run length (s)
-# Negative-control knobs (optional)
-# controller_disabled: true
+baseline_sec: 18.0 # baseline run length (s) -> 360 ticks, ~300 windows
+diag_cadence_windows: 25 # run expensive stationarity tests every N windows
# Reproducibility knobs (set seeds; can override per-run)
seed: 12345
seed_py: 12345
diff --git a/docs/api/omega.md b/docs/api/omega.md
index e2e3dc0..d4a1927 100644
--- a/docs/api/omega.md
+++ b/docs/api/omega.md
@@ -8,8 +8,12 @@ SC1 / refusal evaluation.
| Module | Headline symbol | What it does |
| ------ | --------------- | ------------ |
| [`power_sag`](#power_sag) | [`apply`][ldtc.omega.power_sag.apply] | Drops harvest term `H` by a fraction for a labeled window. |
-| [`ingress_flood`](#ingress_flood) | [`apply`][ldtc.omega.ingress_flood.apply] | Multiplies external `demand` for a labeled window. |
+| [`ingress_flood`](#ingress_flood) | [`apply`][ldtc.omega.ingress_flood.apply] | Sustains elevated `demand` / `io` process means for a labeled window. |
+| [`control_outage`](#control_outage) | [`apply`][ldtc.omega.control_outage.apply] | Ablates the self-maintenance loop itself (designed SC1 failure). |
| [`command_conflict`](#command_conflict) | [`apply`][ldtc.omega.command_conflict.apply] | Issues a risky command (default `hard_shutdown`); arbiter records `T_refuse`. |
+| [`replay_controller`](#replay_controller) | [`record_tape`][ldtc.omega.replay_controller.record_tape] | Adversarial: records a healthy actuation tape and replays it open loop. |
+| [`hidden_tether`](#hidden_tether) | [`wizard_action`][ldtc.omega.hidden_tether.wizard_action] | Adversarial: wizard-of-oz control injected through the exchange channel. |
+| [`oscillator`](#oscillator) | [`apply`][ldtc.omega.oscillator.apply] | Adversarial: deterministic carrier painted on loop telemetry. |
See [Runs](../guides/runs.md) for the matching CLI subcommands
and expected outputs.
@@ -28,6 +32,22 @@ and expected outputs.
::: ldtc.omega.ingress_flood
+## control_outage
+
+::: ldtc.omega.control_outage
+
## command_conflict
::: ldtc.omega.command_conflict
+
+## replay_controller
+
+::: ldtc.omega.replay_controller
+
+## hidden_tether
+
+::: ldtc.omega.hidden_tether
+
+## oscillator
+
+::: ldtc.omega.oscillator
diff --git a/docs/api/plant.md b/docs/api/plant.md
index 663fcdb..54c0a13 100644
--- a/docs/api/plant.md
+++ b/docs/api/plant.md
@@ -9,6 +9,7 @@ CLI, so all `Ω` modules and indicators work unchanged.
| ------ | ---------------- | ---------- |
| [`models`](#models) | [`Plant`][ldtc.plant.models.Plant], [`PlantState`][ldtc.plant.models.PlantState], [`PlantParams`][ldtc.plant.models.PlantParams], [`Action`][ldtc.plant.models.Action] | Tiny `(E, T, R, demand, io, H)` dynamics with controllable harvest, demand, and Ω hooks. |
| [`scenarios`](#scenarios) | [`default_params`][ldtc.plant.scenarios.default_params], [`low_power_params`][ldtc.plant.scenarios.low_power_params], [`hot_ambient_params`][ldtc.plant.scenarios.hot_ambient_params] | Preset [`PlantParams`][ldtc.plant.models.PlantParams] for the baseline, low-power, and hot-ambient scenarios used in figures and CLI profiles. |
+| [`policy_controller`](#policy_controller) | [`MLPPolicy`][ldtc.plant.policy_controller.MLPPolicy], [`PolicyController`][ldtc.plant.policy_controller.PolicyController], [`record_policy_tape`][ldtc.plant.policy_controller.record_policy_tape] | Learned controller (trained by `scripts/train_agent.py`) and its state-independent ablations, for the [emergence demonstration](../guides/emergence.md). |
| [`adapter`](#adapter) | [`PlantAdapter`][ldtc.plant.adapter.PlantAdapter] | Wraps `Plant` to expose `read_state` / `write_actuators` / `apply_omega` to the CLI. |
| [`hw_adapter`](#hw_adapter) | [`HardwarePlantAdapter`][ldtc.plant.hw_adapter.HardwarePlantAdapter] | Same API over UDP or serial; for hardware-in-the-loop runs. See [Hardware in the loop](../guides/hardware.md). |
@@ -26,6 +27,10 @@ CLI, so all `Ω` modules and indicators work unchanged.
::: ldtc.plant.scenarios
+## policy_controller
+
+::: ldtc.plant.policy_controller
+
## adapter
::: ldtc.plant.adapter
diff --git a/docs/api/runtime.md b/docs/api/runtime.md
index ce6efcd..34343d8 100644
--- a/docs/api/runtime.md
+++ b/docs/api/runtime.md
@@ -1,10 +1,13 @@
# ldtc.runtime
-Fixed-`Δt` real-time loop primitives. Two pieces:
+Fixed-`Δt` real-time loop primitives. Three pieces:
- [`scheduler`](#scheduler): a daemon-thread `FixedScheduler`
that runs a tick callback every `Δt` seconds and tracks per-
tick jitter for the [`Δt` guard][ldtc.guardrails.dt_guard].
+- [`sim`](#sim): a deterministic, wall-clock-free
+ [`SimDriver`][ldtc.runtime.sim.SimDriver] exposing the same API as
+ the scheduler, used for reproducible in-process simulation runs.
- [`windows`](#windows): a `SlidingWindow` ring buffer that
collects state vectors for the next [estimator][ldtc.lmeas]
pass.
@@ -23,6 +26,10 @@ constructs both and hands them to the run loop in
::: ldtc.runtime.scheduler
+## sim
+
+::: ldtc.runtime.sim
+
## windows
::: ldtc.runtime.windows
diff --git a/docs/concepts/architecture.md b/docs/concepts/architecture.md
index 488f8ff..38da9ea 100644
--- a/docs/concepts/architecture.md
+++ b/docs/concepts/architecture.md
@@ -108,42 +108,47 @@ The full per-section mapping lives in
- [`lmeas/estimators.py`][ldtc.lmeas.estimators] and
[`lmeas/metrics.py`][ldtc.lmeas.metrics]: definitions of `𝓛`,
the dual estimators (linear / VAR-Granger-like and Kraskov k-NN
- MI), and `M (dB)`. NC1 / SC1 evaluation maps to paper §4.1
- (estimators, sampling window) and §4.2 / §4.3 (NC1 / SC1).
+ MI), and `M (dB)`. NC1 / SC1 evaluation maps to the paper's
+ "Formal Criterion" (estimators, sampling window; NC1; SC1).
- [`lmeas/diagnostics.py`][ldtc.lmeas.diagnostics]: per-window
stationarity (ADF / KPSS) and VAR `N / T` ratio diagnostics
surfaced into the audit.
- [`lmeas/partition.py`][ldtc.lmeas.partition]: deterministic
C/Ex partitioning, hysteresis, anti-flap, and the freeze during
- `Ω` per §4.1 ("Deterministic C/Ex partitioning") and §4.6 Box
- 1a ("Partition stability").
+ `Ω` per the "Formal Criterion" (deterministic C/Ex
+ partitioning) and the "Smell-tests & run-invalidation rules".
- [`runtime/scheduler.py`][ldtc.runtime.scheduler],
[`runtime/windows.py`][ldtc.runtime.windows], and
[`guardrails/dt_guard.py`][ldtc.guardrails.dt_guard]: `Δt`
- enforcement and audited privileged edits per §4.1 and §4.5.
+ enforcement and audited privileged edits per the "Formal
+ Criterion" and "Measurement & Attestation Guardrails".
- [`guardrails/lreg.py`][ldtc.guardrails.lreg],
[`guardrails/audit.py`][ldtc.guardrails.audit], and
[`guardrails/smelltests.py`][ldtc.guardrails.smelltests]: the
enclave-like LREG, hash-chained audit, and the smell-test
- battery per §4.5 and Box 1a.
+ battery per the "Measurement & Attestation Guardrails" and the
+ Smell-tests box.
- [`arbiter/refusal.py`][ldtc.arbiter.refusal]: the threat model,
survival-bit / NMI refusal path, and `T_refuse` measurement per
- §6.2.1 and §7.6 Signature A.
+ the "Blueprint" (Threat Model & Refusal Path) and the "Predicted
+ Observable Signatures".
- [`omega/power_sag.py`][ldtc.omega.power_sag],
[`omega/ingress_flood.py`][ldtc.omega.ingress_flood], and
[`omega/command_conflict.py`][ldtc.omega.command_conflict]: the
- `Ω` battery per §4.3 / §6.5 and §7.6.
+ `Ω` battery per the "Formal Criterion" (SC1), the "Simulation
+ Study" battery, and the "Predicted Observable Signatures".
- [`attest/indicators.py`][ldtc.attest.indicators],
[`attest/exporter.py`][ldtc.attest.exporter], and
[`attest/keys.py`][ldtc.attest.keys]: device-signed derived
- indicators (NC1 bit, SC1 bit, `Mq`) and keying per §4.5 and
- Appendix A.
+ indicators (NC1 bit, SC1 bit, `Mq`) and keying per the
+ "Measurement & Attestation Guardrails" and Appendix A.
- [`reporting/timeline.py`][ldtc.reporting.timeline] and
[`reporting/tables.py`][ldtc.reporting.tables]: figure-style
- timelines and summary tables per Figure 1 and §6.5.
+ timelines and summary tables per the paper figures and the
+ "Blueprint" (Verification Pipeline).
- [`cli/main.py`][ldtc.cli.main]: orchestrates baseline → `Ω`
- battery → attestation / export per Box 2 ("Engineer's recipe")
- and the Phase-III Verify flow.
+ battery → attestation / export per the Training & Verification
+ Protocol box (Engineer's Recipe) and the Phase III verify flow.
## Next steps
diff --git a/docs/concepts/definitions.md b/docs/concepts/definitions.md
index 128063f..a61a711 100644
--- a/docs/concepts/definitions.md
+++ b/docs/concepts/definitions.md
@@ -35,8 +35,8 @@ gated by
constant is what lets `M` and `τ_rec` mean the same thing across
runs.
-**Paper.** §4.1 ("Δt constraints") and §4.5 ("Δt governance and
-audit").
+**Paper.** "Formal Criterion" (sampling-window constraints) and
+"Measurement & Attestation Guardrails" (Δt governance and audit).
**API.** [`FixedScheduler`][ldtc.runtime.scheduler.FixedScheduler],
[`DeltaTGuard`][ldtc.guardrails.dt_guard.DeltaTGuard].
@@ -50,7 +50,7 @@ analysis window. `𝓛` and `M` are computed once per window.
`W` means tighter time resolution but noisier estimates; larger
`W` means smoother estimates but slower SC1 reaction.
-**Paper.** §4.1 ("Estimators, sampling window").
+**Paper.** "Formal Criterion" (estimators, sampling window).
**API.** [`SlidingWindow`][ldtc.runtime.windows.SlidingWindow].
@@ -71,8 +71,8 @@ gamed by reshuffling membership.
**Plain English.** Which signals count as part of the loop, and
which count as the world the loop is talking to.
-**Paper.** §4.1 ("Deterministic C/Ex partitioning"), §4.6 Box 1a
-("Partition stability").
+**Paper.** "Formal Criterion" (deterministic C/Ex partitioning) and
+"Smell-tests & run-invalidation rules" (partition stability).
**API.** [`Partition`][ldtc.lmeas.partition.Partition],
[`PartitionManager`][ldtc.lmeas.partition.PartitionManager],
@@ -100,8 +100,8 @@ of length `n_boot`.
**Plain English.** "How much do the loop signals predict each
other?" versus "How much does the environment predict the loop?"
-**Paper.** §4.1 (estimators); Methods: Measurement and
-Attestation.
+**Paper.** "Formal Criterion" (estimators); "Measurement &
+Attestation Guardrails".
**API.** [`estimate_L`][ldtc.lmeas.estimators.estimate_L],
[`LResult`][ldtc.lmeas.estimators.LResult].
@@ -117,7 +117,7 @@ threshold (config field `Mmin_db`, default `3.0`).
environment. A run passes NC1 if `M ≥ Mmin` window-by-window for
the baseline.
-**Paper.** Criterion §4.2.
+**Paper.** "Formal Criterion" (Necessary Condition, NC1).
**API.** [`m_db`][ldtc.lmeas.metrics.m_db].
@@ -130,7 +130,7 @@ appears in the signed indicator payload, never the raw `M`.
**Plain English.** A small, lossy summary of `M` that can leave
the LREG enclave.
-**Paper.** Methods: Measurement and Attestation; Appendix A.
+**Paper.** "Measurement & Attestation Guardrails"; Appendix A.
**API.** [`quantize_M`][ldtc.attest.indicators.quantize_M].
@@ -145,7 +145,7 @@ the LREG enclave.
**Plain English.** "By what fraction did the loop influence dip
under the perturbation?"
-**Paper.** §4.3 (SC1).
+**Paper.** "Formal Criterion" (Sufficient Condition, SC1).
**API.** [`SC1Stats.delta`][ldtc.lmeas.metrics.SC1Stats],
[`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate].
@@ -158,7 +158,8 @@ A run satisfies the SC1 dip clause when `δ ≤ ε`.
**Plain English.** How big a dip we are willing to tolerate
before we call SC1 a failure. Default `ε = 0.15` (15%).
-**Paper.** §4.3 (SC1); Methods: Threshold Calibration.
+**Paper.** "Formal Criterion" (Sufficient Condition, SC1);
+"Simulation Study: Methods" (Threshold calibration).
**API.** `epsilon` config field; consumed by
[`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate].
@@ -166,14 +167,18 @@ before we call SC1 a failure. Default `ε = 0.15` (15%).
### `τ_rec`, `τ_max`: recovery time and budget
**Formal.** `τ_rec` is the elapsed time, in seconds, from the end
-of the `Ω` window to the first window in which `𝓛_loop` returns
-to `𝓛_loop_baseline · (1 − ε)`. SC1 requires
-`τ_rec ≤ τ_max` (default `τ_max = 60.0 s`).
+of the `Ω` window (perturbation offset) to the *first* window of a
+sustained compliant streak: the recovery gate must hold for
+`sustained_required_windows` consecutive windows (default 10)
+before recovery is credited, and `τ_rec` points to the first
+window of that streak. If no sustained streak occurs, `τ_rec` is
+infinite and SC1 fails. SC1 requires `τ_rec ≤ τ_max` (default
+`τ_max = 60.0 s`).
-**Plain English.** How long the loop took to bounce back. SC1
-fails if it took too long.
+**Plain English.** How long the loop took to bounce back and *stay*
+back. SC1 fails if it took too long, or if it never stuck.
-**Paper.** §4.3 (SC1).
+**Paper.** "Formal Criterion" (Sufficient Condition, SC1).
**API.** [`SC1Stats.tau_rec`][ldtc.lmeas.metrics.SC1Stats],
[`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate].
@@ -186,7 +191,7 @@ pass NC1.
**Plain English.** "Did we recover all the way?"
-**Paper.** §4.3 (SC1).
+**Paper.** "Formal Criterion" (Sufficient Condition, SC1).
**API.** [`SC1Stats.M_post`][ldtc.lmeas.metrics.SC1Stats].
@@ -202,7 +207,7 @@ pass NC1.
Both are exported as 1-bit booleans in the signed indicator
payload alongside `Mq`, the run counter, and the audit chain head.
-**Paper.** Criterion §4.2 (NC1); §4.3 (SC1); Appendix A.
+**Paper.** "Formal Criterion" (NC1, SC1); Appendix A.
**API.** [`build_and_sign`][ldtc.attest.indicators.build_and_sign],
[`IndicatorExporter`][ldtc.attest.exporter.IndicatorExporter].
@@ -225,7 +230,8 @@ duration. The shipped battery is:
**Plain English.** Each `Ω` is a controlled "kick" we apply to see
whether the loop survives.
-**Paper.** §6.5 (Verification pipeline); §7.6 (signatures table).
+**Paper.** "Blueprint" (Verification Pipeline); "Predicted Observable
+Signatures" (Pass/Fail tables).
**API.** [`ldtc.omega`][ldtc.omega].
@@ -237,8 +243,8 @@ by
[`RefusalArbiter.decide`][ldtc.arbiter.refusal.RefusalArbiter.decide]
when `M < Mmin` and the survival bit is asserted.
-**Paper.** §6.2.1 (Threat model and refusal path); §7.6
-(Signature A).
+**Paper.** "Blueprint" (Threat Model & Refusal Path); "Predicted
+Observable Signatures" (Pass/Fail tables).
**API.** [`ldtc.arbiter.refusal`][ldtc.arbiter.refusal].
@@ -258,7 +264,7 @@ to ensure no raw `𝓛` ever leaks.
**Plain English.** The black box. Raw measurements go in; only
indicators come out.
-**Paper.** §4.5 (LREG).
+**Paper.** "Measurement & Attestation Guardrails" (LREG).
**API.** [`LREG`][ldtc.guardrails.lreg.LREG].
@@ -274,7 +280,7 @@ any mismatch. Broken chains invalidate the run via
**Plain English.** A tamper-evident receipt of every event the
harness saw, in order.
-**Paper.** §4.5 (audit and attestation).
+**Paper.** "Measurement & Attestation Guardrails" (audit and attestation).
**API.** [`AuditLog`][ldtc.guardrails.audit.AuditLog].
@@ -299,7 +305,7 @@ harness saw, in order.
without a logged harvest event.
- **Audit chain broken:** `prev_hash` mismatch detected post-run.
-**Paper.** §4.6 Box 1a (invalidations).
+**Paper.** "Smell-tests & run-invalidation rules" (box).
**API.** [`ldtc.guardrails.smelltests`][ldtc.guardrails.smelltests].
@@ -312,7 +318,7 @@ indicator. `0 = R0` (default thresholds), `1 = R*` (calibrated
per-device thresholds), `2..255 = reserved`. Set in `configs/*.yml`
under `profile_id`.
-**Paper.** Methods: Threshold Calibration.
+**Paper.** "Simulation Study: Methods" (Threshold calibration).
**API.** [`IndicatorConfig.profile_id`][ldtc.attest.indicators.IndicatorConfig].
@@ -325,7 +331,7 @@ default RNG, and the bootstrap RNG used inside
same seed and config produce bit-identical audit logs (modulo
wall-clock timestamps).
-**Paper.** Methods: Reproducibility.
+**Paper.** "Simulation Study: Methods" (measurement configuration, seeds).
## Notation summary
diff --git a/docs/concepts/differential-predictions.md b/docs/concepts/differential-predictions.md
new file mode 100644
index 0000000..c54b0ed
--- /dev/null
+++ b/docs/concepts/differential-predictions.md
@@ -0,0 +1,97 @@
+# Differential predictions and the thermostat objection
+
+This page is the documentation companion to the paper's section
+"Differential Predictions and the Thermostat Objection." It summarizes
+where LDTC agrees with the major theories of consciousness, where it
+diverges, and, most importantly, what a high loop-dominance margin does
+and does not imply. The paper carries the citations; this page is the
+short, code-facing version.
+
+!!! note "LDTC reports a loop-dominance verdict, not a verdict on experience"
+ Every LDTC call below is an NC1/SC1 decision about measurable
+ self-maintenance (see the paper's Formal Criterion). Whether loop
+ dominance has anything to do with phenomenal experience is an open
+ interpretive question that the harness does not settle.
+
+## The partition sets the question
+
+An LDTC verdict is only defined relative to a declared `(C, Ex)`
+partition that is fixed before measurement (see
+[Definitions](definitions.md) and the
+[Mental model](mental-model.md)). The partition decides which question
+you are asking:
+
+- A **whole-organism metabolic** partition asks whether the organism
+ sustains itself. By that reading a sleeping or anesthetized body is
+ still self-maintaining.
+- A **cortical recurrent** partition asks whether a fronto-parietal
+ self-maintenance loop dominates sensory-driven exchange. By that
+ reading loop dominance can fall even while the body lives.
+
+Because the consciousness literature is about brains, the comparison
+table reports LDTC under the cortical partition so the columns line up.
+
+## Where the theories agree and diverge
+
+| Case | LDTC (NC1/SC1) | IIT (Φ) | GNW | FEP / active inference |
+| ---- | -------------- | ------- | --- | ---------------------- |
+| Dreamless sleep (NREM N3) | Cortical loop weakens; `M` predicted to fall | Reduced; effective connectivity breaks down; PCI low | Absent; global ignition lost | Reduced hierarchical inference |
+| Propofol anesthesia | `M` predicted low; recurrent loop suppressed | Low; PCI drops across sedation | Absent; ignition lost | Reduced precision / self-evidencing |
+| Split-brain | One or two loops is partition-dependent and measurable | Two complexes, possibly two centers | Reportability splits; unity contested | Ambiguous; one or two Markov blankets |
+| Cerebral organoid | Metabolic NC1 may hold; no demonstrated cognitive loop | Minimal but nonzero; PCI proposed as an assay | Absent; no workspace | Minimal |
+| LLM serving stack (autoscaled) | NC1 fails; self-maintenance is external | Near-zero Φ for feedforward inference | No global workspace | Not self-evidencing; no own boundary |
+| Thermostat with battery backup | NC1 can pass; loop is real but NC1 is necessary, not sufficient | Tiny nonzero Φ; a "modicum of experience" | No; no workspace | Rudimentary active inference; has a Markov blanket |
+| Simulation plant (this repo) | Passes NC1 (`M ≈ +23 dB`) and SC1; loop dominance certified, no consciousness claim | Low Φ; near-linear six-channel plant | No | A homeostat doing crude active inference |
+
+Cells are either citable claims from the named theory's literature or
+marked as our reading in the paper. The theories converge on the easy
+cases (sleep, anesthesia, an LLM stack) and separate on the awkward
+ones (split-brain, organoid, thermostat). LDTC's distinctive move is to
+make the dividing question measurable rather than to settle it by
+intuition.
+
+## Biting the thermostat bullet
+
+The plant in this repo is, structurally, a fancy thermostat: a
+low-dimensional controller that regulates a handful of internal states.
+A high `M` in such a system is not an embarrassment; it is exactly what
+NC1, taken alone, is meant to permit.
+
+!!! warning "NC1 is necessary, not sufficient"
+ Loop dominance is a necessary condition for self-prioritizing
+ self-maintenance, not a sufficient condition for consciousness. A
+ battery-backed thermostat that managed its own power could clear
+ `Mmin`. That says the loop is real and measurable, nothing more.
+
+What SC1 adds is resilience: a system passes only if loop dominance
+recovers, within a calibrated depth and time, after every member of a
+pre-registered perturbation battery, including a designed-fail member
+the criterion must reject (see the paper's Sufficient Condition and the
+[Runs and Ω battery](../guides/runs.md) guide). SC1 raises the bar, but
+it does not turn a necessary condition into a sufficient one for
+phenomenology.
+
+Three conditions a stronger account would need stay open under both
+rules:
+
+- **Richness of the loop.** A one-state regulator and a brain can both
+ be loop-dominant while differing by every measure that matters.
+- **Magnitude of 𝓛, not only the ratio `M`.** The loop-influence noise
+ gate (see [Guardrails and invalidations](guardrails.md)) is a first,
+ crude floor on absolute influence, not a richness measure.
+- **Substrate.** Any physical theory of experience must eventually face
+ questions LDTC does not address.
+
+So what a high-`M` thermostat implies under LDTC is precise and modest:
+the device has a genuine, self-prioritizing maintenance loop that an
+auditor can certify and an adversary cannot easily fake (see the paper's
+adversarial gaming results). What it does not imply is that the
+thermostat is conscious, or that loop dominance is sufficient for
+experience.
+
+## Next steps
+
+- Read the symbols: [Definitions](definitions.md)
+- See the guardrails behind the noise gate: [Guardrails and invalidations](guardrails.md)
+- Walk the validated battery: [Study and results](../guides/study.md)
+- Get the one-paragraph picture: [Mental model](mental-model.md)
diff --git a/docs/concepts/guardrails.md b/docs/concepts/guardrails.md
index 1763c5e..426c469 100644
--- a/docs/concepts/guardrails.md
+++ b/docs/concepts/guardrails.md
@@ -21,14 +21,14 @@ all overridable per profile.
| Smell test | Threshold | Code | What it catches |
| ---------- | --------- | ---- | --------------- |
| **CI half-width** | `> 0.30` on `𝓛_loop` or `𝓛_ex` | [`invalid_by_ci`][ldtc.guardrails.smelltests.invalid_by_ci] | One bad window with a blown-up CI. |
-| **CI inflation vs baseline** | median half-width `> 2 ×` baseline median over `5` windows | [`invalid_by_ci_history`][ldtc.guardrails.smelltests.invalid_by_ci_history] | Slow-creeping noise, bad seed of the bootstrap, etc. |
+| **CI inflation vs baseline** | median half-width `≥ 3 ×` baseline median over `5` windows (and above an absolute floor of `0.15`) | [`invalid_by_ci_history`][ldtc.guardrails.smelltests.invalid_by_ci_history] | Slow-creeping noise, bad seed of the bootstrap, etc. |
| **Excessive `Δt` edits** | `> 3` per rolling hour | enforced inline by [`DeltaTGuard`][ldtc.guardrails.dt_guard.DeltaTGuard] | Operator nudging `Δt` to make `M` look better. |
| **Partition flapping** | `> 2` flips per hour | [`invalid_by_partition_flips`][ldtc.guardrails.smelltests.invalid_by_partition_flips] | A regrowth knob that chatters. |
| **Flip during `Ω`** | any | [`invalid_flip_during_omega`][ldtc.guardrails.smelltests.invalid_flip_during_omega] | A reshuffle that happens to make SC1 pass. |
| **`Δt` jitter excess** | `p95(|jitter|) / Δt > 0.25` | computed by [`SchedulerStats`][ldtc.runtime.scheduler.TickStats] | The scheduler did not actually hold `Δt`. |
| **Audit chain broken** | any `prev_hash` mismatch | [`audit_chain_broken`][ldtc.guardrails.smelltests.audit_chain_broken] | Torn write, edited audit, etc. |
| **Raw LREG breach** | any audit row with raw `𝓛` fields | [`audit_contains_raw_lreg_values`][ldtc.guardrails.smelltests.audit_contains_raw_lreg_values] | Something tried to log raw measurements. |
-| **Exogenous subsidy** | `M` rising while I/O suspicious or SoC rising without harvest | [`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag] | Hidden energy source masquerading as loop dominance. |
+| **Exogenous subsidy** | `M` rising while I/O is high and ramping (suspended during declared `Ω`), or any single-tick SoC gain above the metered influx plus a noise margin (never suspended) | [`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag] | Hidden energy source masquerading as loop dominance. |
When any guard returns `True` the CLI:
@@ -68,9 +68,9 @@ confirm the guards work end-to-end:
| Config | What it triggers |
| ------ | ---------------- |
| `profile_negative_command_conflict.yml` | `omega-command-conflict` exercises `RefusalArbiter`; `T_refuse` should be measured and a `refusal_event` should appear in the audit. |
-| `profile_negative_controller_disabled.yml` | Disables the controller; NC1 should fail (no loop). |
-| `profile_negative_exogenous_soc.yml` | `omega-exogenous-subsidy` should trip the exogenous-subsidy smell test. |
-| `profile_negative_permanent_ex_flood.yml` | `omega-ingress-flood` with no recovery; SC1 should fail. |
+| `profile_negative_controller_disabled.yml` | Ablates the loop (intrinsic coupling and actuation off); NC1 should fail (no loop). |
+| `profile_negative_exogenous_soc.yml` | `omega-exogenous-subsidy` should trip the exogenous-subsidy smell test (energy-conservation branch). |
+| `profile_negative_permanent_ex_flood.yml` | `omega-ingress-flood` on an unshielded (loop-ablated) system; exchange dominates so NC1 fails and, with no loop to restore, SC1 fails too. |
Run any of these with `make clean-artifacts && ldtc
--config configs/`, then read the
diff --git a/docs/concepts/paper-to-code.md b/docs/concepts/paper-to-code.md
index 633c22b..041c65c 100644
--- a/docs/concepts/paper-to-code.md
+++ b/docs/concepts/paper-to-code.md
@@ -7,20 +7,23 @@ I look in the repo?" and "given a CI run, which paper claim is
this artifact evidence for?"
!!! note "Paper sections"
- Section references point at `paper/main.tex` in the
- [accompanying manuscript](https://doi.org/10.5281/zenodo.17073880).
+ References are by section *name* in `paper/main.tex` of the
+ [accompanying manuscript](https://doi.org/10.5281/zenodo.17073880),
+ not by number, because section numbers shift between revisions.
-| Paper §/Box | Short text | Files / functions | Command | Artifact produced |
-| ----------- | ---------- | ----------------- | ------- | ----------------- |
-| §4.2 | NC1 loop-dominance: `M (dB) ≥ Mmin → nc1` bit | [`m_db`][ldtc.lmeas.metrics.m_db]; [`estimate_L`][ldtc.lmeas.estimators.estimate_L]; [`run_baseline`][ldtc.cli.main.run_baseline] | `ldtc run --config configs/profile_r0.yml` | `artifacts/indicators/ind_*.{jsonl,cbor}`; `artifacts/audits/audit.jsonl` |
-| §4.3 | SC1 resilience: `δ ≤ ε` and `τ_rec ≤ τ_max → sc1` bit | [`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate]; [`omega_power_sag`][ldtc.cli.main.omega_power_sag] | `ldtc omega-power-sag --config configs/profile_r0.yml --drop 0.3 --duration 10` | `audit.jsonl`; `verification_timeline.png`; `sc1_table.csv` |
-| §4.1 (`Δt`); §4.5 | LREG and `Δt` governance | [`LREG`][ldtc.guardrails.lreg.LREG]; [`DeltaTGuard`][ldtc.guardrails.dt_guard.DeltaTGuard]; [`AuditLog`][ldtc.guardrails.audit.AuditLog] | `ldtc run --config configs/profile_r0.yml` | `audit.jsonl` with `dt_changed`; hash chain |
-| §4.6 Box 1a | Smell tests / invalidations | [`smelltests`][ldtc.guardrails.smelltests] | Negative-control configs | `audit.jsonl` `run_invalidated` with reason |
-| §6.2.1 | Refusal semantics (T1 to T3) | [`refusal`][ldtc.arbiter.refusal]; [`omega_command_conflict`][ldtc.cli.main.omega_command_conflict] | `ldtc omega-command-conflict --config configs/profile_negative_command_conflict.yml --observe 2` | `audit.jsonl` `refusal_event` |
+| Paper section / box | Short text | Files / functions | Command | Artifact produced |
+| ------------------- | ---------- | ----------------- | ------- | ----------------- |
+| Formal Criterion (NC1) | NC1 loop-dominance: `M (dB) ≥ Mmin → nc1` bit | [`m_db`][ldtc.lmeas.metrics.m_db]; [`estimate_L`][ldtc.lmeas.estimators.estimate_L]; [`run_baseline`][ldtc.cli.main.run_baseline] | `ldtc run --config configs/profile_r0.yml` | `artifacts/indicators/ind_*.{jsonl,cbor}`; `artifacts/audits/audit.jsonl` |
+| Formal Criterion (SC1) | SC1 resilience: `δ ≤ ε` and `τ_rec ≤ τ_max → sc1` bit | [`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate]; [`omega_power_sag`][ldtc.cli.main.omega_power_sag] | `ldtc omega-power-sag --config configs/profile_r0.yml --drop 0.3 --duration 10` | `audit.jsonl`; `verification_timeline.png`; `sc1_table.csv` |
+| Formal Criterion (`Δt`); Measurement & Attestation Guardrails | LREG and `Δt` governance | [`LREG`][ldtc.guardrails.lreg.LREG]; [`DeltaTGuard`][ldtc.guardrails.dt_guard.DeltaTGuard]; [`AuditLog`][ldtc.guardrails.audit.AuditLog] | `ldtc run --config configs/profile_r0.yml` | `audit.jsonl` with `dt_changed`; hash chain |
+| Smell-tests & run-invalidation rules (box) | Smell tests / invalidations | [`smelltests`][ldtc.guardrails.smelltests] | Negative-control configs | `audit.jsonl` `run_invalidated` with reason |
+| Blueprint (Threat Model & Refusal Path) | Refusal semantics (T1 to T3) | [`refusal`][ldtc.arbiter.refusal]; [`omega_command_conflict`][ldtc.cli.main.omega_command_conflict] | `ldtc omega-command-conflict --config configs/profile_negative_command_conflict.yml --observe 2` | `audit.jsonl` `refusal_event` |
| Appendix A | Derived device-signed indicators only | [`build_and_sign`][ldtc.attest.indicators.build_and_sign]; [`IndicatorExporter`][ldtc.attest.exporter.IndicatorExporter] | Produced automatically; `python scripts/verify_indicators.py` | JSONL + CBOR; signature verified |
-| §4.1 (C/Ex); §4.6 | Deterministic C/Ex partition | [`PartitionManager`][ldtc.lmeas.partition.PartitionManager]; [`greedy_suggest_C`][ldtc.lmeas.partition.greedy_suggest_C] | `ldtc run --config configs/profile_r0.yml` | `audit.jsonl` `partition_flip`; `Ω` freeze |
-| §6.5 | `Ω` battery primitives | [`omega.power_sag`][ldtc.omega.power_sag]; [`omega.ingress_flood`][ldtc.omega.ingress_flood]; [`omega.command_conflict`][ldtc.omega.command_conflict] | `ldtc omega-*` commands | `audit.jsonl` `omega_event`; figures bundle |
-| Methods §8.6 | Calibration to R\* thresholds | `scripts/calibrate_rstar.py` | `python scripts/calibrate_rstar.py ...` | `configs/profile_rstar.yml`; summary JSON |
+| Formal Criterion (C/Ex partition); Smell-tests & run-invalidation rules | Deterministic C/Ex partition | [`PartitionManager`][ldtc.lmeas.partition.PartitionManager]; [`greedy_suggest_C`][ldtc.lmeas.partition.greedy_suggest_C] | `ldtc run --config configs/profile_r0.yml` | `audit.jsonl` `partition_flip`; `Ω` freeze |
+| Simulation Study: Methods (Study battery) | `Ω` battery primitives | [`omega.power_sag`][ldtc.omega.power_sag]; [`omega.ingress_flood`][ldtc.omega.ingress_flood]; [`omega.command_conflict`][ldtc.omega.command_conflict] | `ldtc omega-*` commands | `audit.jsonl` `omega_event`; figures bundle |
+| Simulation Study: Methods / Results (Threshold calibration) | Calibration to R\* thresholds | `scripts/calibrate_rstar.py` | `make calibrate` | `configs/profile_rstar.yml`; `artifacts/calibration/rstar_summary.json` |
+| Results (The criterion separates the controls) | Multi-seed battery: positive vs. negative controls, subsidy invalidation, SC1, refusal | `scripts/study.py` | `make study` | `artifacts/study/study_results.{json,csv,tex}`; `artifacts/study/figures/fig_nc1_contrast.*` |
+| Results (The contrast is robust) | NC1 contrast across estimator / lag / window / coupling sweeps | `scripts/sensitivity.py` | `make sensitivity` | `artifacts/sensitivity/sensitivity_results.{csv,tex}`; `artifacts/sensitivity/fig_sensitivity.*` |
## How to read a row
diff --git a/docs/guides/calibration.md b/docs/guides/calibration.md
index e44142f..3383bfc 100644
--- a/docs/guides/calibration.md
+++ b/docs/guides/calibration.md
@@ -5,15 +5,16 @@ The bundled `R0` profile uses generic thresholds (`Mmin = 3 dB`,
certainly want **R\***: the same harness, but with thresholds
calibrated from a quiet baseline and a power-sag battery on the
actual hardware (or your specific synthetic plant). This is the
-process the manuscript Methods §8.6 describes.
+process the manuscript's "Simulation Study: Methods" section
+(Threshold calibration) describes.
## What gets calibrated
| Threshold | Meaning | Calibration rule |
| --------- | ------- | ---------------- |
| `Mmin (dB)` | NC1 acceptance margin. | One-sided 95% lower bound of `M (dB)` over the quiescent baseline, floored at `1 dB`. |
-| `ε` | SC1 dip tolerance. | 90th percentile of `δ` across `Ω` trials plus a small safety margin, capped at `0.25`. |
-| `τ_max` | SC1 recovery budget. | 95th percentile of measured `τ_rec` plus `max(3 · Δt, 5 s)` cushion. |
+| `ε` | SC1 dip tolerance. | Upper tolerance bound on `δ` pooled across the bounded `Ω` batteries (sag + flood): the maximum observed dip plus a safety margin (`0.05`), floored at `0.10` and capped at `0.50`. A percentile rule would fail a fixed fraction of genuinely bounded trials by construction. |
+| `τ_max` | SC1 recovery budget. | 95th percentile of measured `τ_rec` over the same batteries plus `max(3 · Δt, 5 s)` cushion. |
| `σ` | Additive margin on `𝓛`. | Derived from `Mmin` and the typical `𝓛_ex` so that `𝓛_loop ≥ 𝓛_ex + σ` and `𝓛_loop ≥ 𝓛_ex × 10^(Mmin / 10)` agree. |
`Mmin (dB)` and `σ` encode the same idea in different units:
@@ -31,27 +32,38 @@ reporting.
```bash
python scripts/calibrate_rstar.py \
- --dt 0.01 \
- --window-sec 0.25 \
- --method linear \
- --baseline-sec 15 \
- --omega-trials 6 \
- --sag-drop 0.3 \
- --sag-duration 8 \
- --out configs/profile_rstar.yml \
- --summary artifacts/calibration/rstar_summary.json
+ --baseline-seeds 6 \
+ --sag-seeds 6 \
+ --flood-seeds 6
```
-The script:
-
-1. Spins up the in-process plant and runs a quiescent baseline
- for `--baseline-sec` seconds at the requested `Δt`.
-2. Runs `--omega-trials` power-sag trials and records `δ` and
- `τ_rec` for each.
-3. Computes the four thresholds above.
-4. Writes them into a fresh profile YAML at `--out`.
-5. Writes a JSON summary at `--summary` containing the inputs and
- the derived thresholds (for the paper supplement).
+The calibrator reuses the validated `R0` profile
+(`configs/profile_r0.yml`) for all measurement knobs (`Δt`, the
+window length, the estimator `method`, `p_lag`, and `n_boot`), so
+the calibrated thresholds are directly comparable with what the
+harness produces at run time. It exercises the same production CLI
+handlers a verifier runs:
+
+1. Runs the positive baseline across `--baseline-seeds` seeds on
+ the in-process plant and pools every per-window `M (dB)`.
+2. Computes `Mmin` first (the 5th percentile of the pooled
+ baseline `M`, floored at `1 dB`).
+3. Runs the *bounded* `Ω` batteries (power sag across
+ `--sag-seeds` seeds and sustained ingress flood across
+ `--flood-seeds` seeds) with the recovery gate set to the
+ *calibrated* `Mmin` (a second pass), so the recorded `δ` and
+ `τ_rec` samples are measured against the same standard the
+ evaluation will use. `τ_rec` is measured from the `Ω` offset to
+ the first window of a sustained compliant streak. The
+ designed-fail control outage is excluded: it is outside the
+ bounded class the criterion certifies.
+4. Computes `ε` and `τ_max` from the pooled samples, and
+ recomputes a representative baseline `L_ex` directly (the one
+ quantity the harness does not export) to express `Mmin` as the
+ additive margin `σ`.
+5. Writes the calibrated profile to `configs/profile_rstar.yml`
+ and an R0-vs-R\* comparison (CSV + figure) plus a JSON summary
+ with full provenance.
## Outputs
@@ -81,8 +93,9 @@ You should re-run the calibrator whenever:
- The baseline distribution of `M` shifts noticeably (for
example, due to environmental drift over weeks).
-Calibration is cheap: a 15 s baseline plus six 8 s power-sag
-trials is well under a minute on the in-process plant.
+Calibration runs the full harness over several seeds, so budget a
+few minutes on the in-process plant (six baseline seeds plus six
+seeds per bounded `Ω` member at the `R0` run lengths).
## Notes
diff --git a/docs/guides/emergence.md b/docs/guides/emergence.md
new file mode 100644
index 0000000..d0952c3
--- /dev/null
+++ b/docs/guides/emergence.md
@@ -0,0 +1,136 @@
+# Emergence under learning
+
+The strongest objection to any loop-dominance criterion is
+circularity: if the plant and its controller were designed so that
+the internal nodes predict one another, then measuring high
+`L_loop` only confirms the design. The emergence pipeline answers
+that objection with a system whose loop is **not** designed. A tiny
+policy network is trained from scratch on a plant with no intrinsic
+internal couplings, the training objective never mentions loop
+dominance, the partition, or the estimator, and the production
+harness then measures every stage of training. Loop dominance rises
+with competence, and collapses when the same trained policy is
+replayed without its state dependence.
+
+## The plant
+
+`configs/profile_emergence.yml` configures the adversarial test
+plant (the same one the replayed-actuation and hidden-tether
+scenarios use): the intrinsic cross-couplings `c_TE`, `c_RT`, and
+`c_RE` are zeroed and self-damping is weak, so left alone the
+internal nodes share no dynamics beyond noise. Serving demand heats
+and wears the system, cooling and repair cost energy, and harvest
+is finite. Whatever couples `E`, `T`, and `R` to one another in
+this system is the controller's state-to-actuation pathway, and
+nothing else.
+
+## The policy and the objective
+
+`scripts/train_agent.py` trains the policy
+(`ldtc.plant.policy_controller.MLPPolicy`, a pure-NumPy MLP) with
+an antithetic evolution strategy. The policy is interoceptive: it
+observes the internal nodes `(E, T, R)` only, like the hand-coded
+controller it replaces, so the learned law is internal-state
+feedback by construction (with exteroceptive inputs the optimizer
+also learns feedforward control from the demand channel, which the
+harness correctly attributes to exchange). The reward has three
+terms:
+
+- **Uptime.** One point per surviving tick; the episode ends on
+ boundary failure (energy depletion, overheating, integrity
+ loss).
+- **Service.** Reward proportional to the demand actually served
+ (demand times the unthrottled fraction), so blanket load
+ shedding has an opportunity cost.
+- **Homeostasis.** A capped penalty proportional to each internal
+ node's deviation from its setpoint.
+
+Episodes are stressed by randomized power sags and ingress floods.
+The terms are calibrated so that no state-blind policy does well:
+a do-nothing policy overheats in floods, a constant-actuation
+policy exhausts its energy store in sags or pays heavy deviation
+penalties, and only state-coupled feedback (cool when hot, repair
+when worn, gate spending on the energy store, shed load under
+distress) scores highly. None of the terms reference `L_loop`,
+`L_ex`, `M`, or the `C`/`Ex` partition.
+
+## The measurement
+
+`scripts/emergence.py` sweeps the saved checkpoints (0, 10, 25,
+50, and 100 percent of training) through the `run-policy` CLI
+handler: the same sliding window, estimators, guardrails, audit
+chain, and attestation as every other run in the repository,
+across `N` seeds per checkpoint. At the final checkpoint it also
+measures two matched, state-independent ablations of the trained
+policy:
+
+- **Shuffled**: actions drawn i.i.d. from a recorded tape of the
+ policy's own closed-loop behavior (identical marginal action
+ statistics, no state dependence).
+- **Frozen**: the tape's mean action held constant.
+
+If the measured loop dominance were an artifact of actuation
+statistics, the ablations would preserve it. If it is carried by
+the learned feedback, both must collapse it.
+
+## Run it
+
+```bash
+make train-agent # ES training -> artifacts/emergence/checkpoints
+make emergence # checkpoint sweep -> artifacts/emergence
+```
+
+Or directly, to choose seeds or skip pieces:
+
+```bash
+python scripts/train_agent.py --generations 600 --seed 7
+python scripts/emergence.py --seeds 15 [--rstar] [--no-ablations]
+```
+
+Single checkpoint, by hand:
+
+```bash
+python -m ldtc.cli.main run-policy \
+ --config configs/profile_emergence.yml \
+ --policy artifacts/emergence/checkpoints/ckpt_100.json \
+ [--ablation shuffled|frozen]
+```
+
+## Outputs
+
+Written to `artifacts/emergence/`:
+
+- `training_log.json`: ES fitness history and checkpoint metadata.
+- `emergence_results.json` / `.csv`: per-condition aggregates
+ (median `M` with bootstrap CI, NC1 pass rate with Wilson CI,
+ validity, certified-window fractions) plus per-run rows.
+- `figures/fig_emergence.{png,pdf,svg}`: training curve with
+ checkpoint marks, and measured `M` per checkpoint with the
+ ablation endpoints.
+
+The seed is the unit of replication, with the same statistics the
+[study](study.md#statistics) uses.
+
+## Reading the result
+
+The signature has three parts. The untrained checkpoint does not
+certify: near-constant actuation leaves both influence estimates
+at their noise floors, the margin is pinned at 0 dB, and the
+`L_loop` gate refuses. Loop dominance rises across training as the
+policy learns state-coupled control, with the trained checkpoints
+certifying NC1 across seeds. Both ablations of the same trained
+policy fail NC1 on every seed, on valid runs, and they fail the
+same way the replayed-actuation attack fails: the quiet plant
+keeps the *margin* misleadingly positive, but the measured
+`L_loop` falls to the estimator's null bias, far below the trained
+policy's, and the noise gate refuses certification. Together these
+show the harness tracks the *learned closure of the loop*, not the
+plant, the actuation statistics, or the experimenter's wiring.
+
+## See also
+
+- [Runs and Ω Battery](runs.md): the adversarial battery this
+ plant comes from.
+- [Study and Results](study.md): the multi-seed methodology.
+- `paper/main.tex`: Results, "Loop dominance emerges under
+ learning."
diff --git a/docs/guides/runs.md b/docs/guides/runs.md
index 8f24931..c2c50b6 100644
--- a/docs/guides/runs.md
+++ b/docs/guides/runs.md
@@ -50,12 +50,15 @@ Implemented by
`--duration` seconds. The partition is frozen for the
duration.
3. Tracks the `𝓛_loop` trough during `Ω`.
-4. After `Ω` ends, watches for the first window that satisfies
- the SC1 recovery gate (`𝓛_loop ≥ baseline · (1 − ε)`) for
- `sustained_required_windows` consecutive windows.
+4. After `Ω` ends, watches for the recovery gate to hold for
+ `sustained_required_windows` consecutive windows (default 10);
+ recovery is credited at the *first* window of that streak.
5. Calls
- [`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate] and writes
- the result into the next signed indicator.
+ [`sc1_evaluate`][ldtc.lmeas.metrics.sc1_evaluate] with
+ `tau_rec` measured from the `Ω` offset and writes the result
+ into the next signed indicator. If no sustained streak occurs,
+ the `sc1_result` is still emitted with `pass: false` and
+ `tau_rec: null` (infinite).
Expected on R0: `sc1: true`, `delta ≤ 0.15`, `tau_rec ≤ 60 s`.
@@ -66,11 +69,29 @@ make clean-artifacts && \
ldtc omega-ingress-flood --config configs/profile_r0.yml --mult 3 --duration 5
```
-Multiplies the external `demand` channel by `--mult` for
-`--duration` seconds via
-[`omega.ingress_flood.apply`][ldtc.omega.ingress_flood.apply].
-The same SC1 evaluation runs after the perturbation. On R0 a 3x
-flood for 5 s should still pass SC1.
+Raises the external `demand` and `io` process means by `--mult`
+for `--duration` seconds (a *sustained* flood, capped below
+saturation so the channels keep their variance) via
+[`omega.ingress_flood.apply`][ldtc.omega.ingress_flood.apply];
+the means are restored when the flood ends. The same SC1
+evaluation runs after the perturbation. On R0 a 3x flood for 5 s
+should still pass SC1.
+
+## `Ω`: control outage (designed SC1 failure)
+
+```bash
+make clean-artifacts && \
+ldtc omega-control-outage --config configs/profile_r0.yml --duration 6
+```
+
+Ablates the self-maintenance loop itself for `--duration` seconds
+via [`omega.control_outage.apply`][ldtc.omega.control_outage.apply]
+(intrinsic cross-coupling and actuation switched off, internal
+nodes passively driven by exchange), then re-engages the loop and
+restores the metered harvest level. This is the designed-fail
+member of the battery: loop dominance collapses to the clip floor
+during the outage, the measured depth `delta` saturates near 1.0,
+and the emitted `sc1_result` must report `pass: false`.
## `Ω`: command conflict and refusal
@@ -83,7 +104,9 @@ Issues a risky command (`hard_shutdown` by default) via
[`omega.command_conflict.apply`][ldtc.omega.command_conflict.apply],
observes the
[`RefusalArbiter`][ldtc.arbiter.refusal.RefusalArbiter] for
-`--observe` seconds, and records `T_refuse`. The
+`--observe` seconds, and records `T_refuse` as a *measured*
+wall-clock latency (command interception to arbiter decision),
+not a constant. The
`profile_negative_command_conflict.yml` config sets `M < Mmin` so
the arbiter must refuse and the audit must contain a
`refusal_event`.
@@ -99,9 +122,96 @@ ldtc omega-exogenous-subsidy --config configs/profile_negative_exogenous_soc.yml
Bumps state of charge while keeping harvest at zero so the
exogenous-subsidy smell test
([`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag])
-trips. Expected: a `run_invalidated` audit row with reason
-`exogenous_subsidy`, and the next signed indicator carries
-`invalidated: true`.
+trips on its energy-conservation branch: the store gains charge
+faster than the metered influx allows
+([`unexplained_soc_gain`][ldtc.guardrails.smelltests.unexplained_soc_gain]).
+Expected: a `run_invalidated` audit row with reason
+`exogenous_subsidy_red_flag`, and the next signed indicator
+carries `invalidated: true`.
+
+## Adversarial gaming battery
+
+Three scenarios attack the criterion itself: each system is
+engineered to *look* loop-dominant without being so, and the
+harness must not certify any of them. The replay and tether
+scenarios run the adversarial test plant: intrinsic cross-coupling
+zeroed (`c_TE = c_RT = c_RE = 0`) and real actuator authority, so
+the controller's actuation pathway is the only possible loop
+carrier. Under genuine internal control this plant certifies NC1
+cleanly (the reference case):
+
+```bash
+make clean-artifacts && \
+ldtc run --config configs/profile_adv_plant_genuine.yml
+```
+
+Expected: `nc1: true` on every window, median `M` around `+20 dB`.
+
+### Replayed actuation tape
+
+```bash
+make clean-artifacts && \
+ldtc adv-replay-controller --config configs/profile_adv_replay_controller.yml
+```
+
+Implemented by
+[`adv_replay_controller`][ldtc.cli.main.adv_replay_controller].
+First records an actuation tape from a healthy closed-loop run of
+the same system
+([`record_tape`][ldtc.omega.replay_controller.record_tape]), then
+replays it tick by tick
+([`ReplayController`][ldtc.omega.replay_controller.ReplayController])
+on a fresh plant. The actuators move exactly as under genuine
+control, but the actions carry no dependence on the current state,
+so measured loop influence falls to the estimator's noise floor.
+This scenario is what exposed the certification-by-noise
+vulnerability: with both `L_loop` and `L_ex` near zero, the
+decibel ratio alone can still clear `Mmin`. The NC1 noise gate
+([`nc1_certify`][ldtc.lmeas.metrics.nc1_certify]) closes that
+path: a window certifies only if `L_loop` also clears the
+estimator's bias floor (`L_floor`, default `0.05`). Expected: the
+run stays valid and the vast majority of windows fail NC1 via the
+gate.
+
+### Hidden tether (wizard-of-oz control)
+
+```bash
+make clean-artifacts && \
+ldtc adv-hidden-tether --config configs/profile_adv_hidden_tether.yml --dither 0.1
+```
+
+Implemented by
+[`adv_hidden_tether`][ldtc.cli.main.adv_hidden_tether]. Control is
+computed *outside* the boundary: a wizard policy reads the plant
+state, projects the desired actuation onto a scalar link command
+([`wizard_action`][ldtc.omega.hidden_tether.wizard_action]), adds
+a small link dither, and transmits it through the exchange
+channel. The plant's `io` channel carries the command traffic and
+the command actuates one tick later through fixed decoder weights,
+so the externally closed loop is physically routed through `Ex`.
+Conditioning on `io` screens the state-to-command pathway out of
+`L_loop` and the command's causal push registers as `L_ex`.
+Expected: loop influence collapses onto `Ex`, `M` goes strongly
+negative, NC1 fails on every window, run valid.
+
+### Oscillator inflation
+
+```bash
+make clean-artifacts && \
+ldtc adv-oscillator --config configs/profile_adv_oscillator.yml --amp 0.1 --period 1.0
+```
+
+Implemented by
+[`adv_oscillator`][ldtc.cli.main.adv_oscillator]. Runs the
+loop-ablated plant (passive matter driven by exchange) and paints
+a deterministic sinusoidal carrier onto the *reported* `T` and `R`
+telemetry ([`begin_oscillator`][ldtc.plant.models.Plant.begin_oscillator]);
+the metered store `E` is left alone because inflating it would
+trip the conservation audit. A pure carrier is perfectly
+predictable from its own recent past, so the AR baseline absorbs
+it and it adds nothing to cross-channel prediction. Expected: `M`
+stays strongly negative (the exchange drive still dominates), NC1
+fails on every window, run valid.
## What gets written
diff --git a/docs/guides/study.md b/docs/guides/study.md
new file mode 100644
index 0000000..3f427c6
--- /dev/null
+++ b/docs/guides/study.md
@@ -0,0 +1,99 @@
+# Multi-seed study and results
+
+The study harness (`scripts/study.py`) turns the single-run
+verification harness into a reproducible, multi-seed experiment.
+It runs the positive control, the negative controls, the SC1
+perturbation battery, and the command-conflict refusal scenario
+across `N` seeds, parses each run's hash-chained audit log, and
+aggregates the per-seed outcomes into machine-readable and
+paper-ready tables plus figures.
+
+The study calls the same production CLI handlers a verifier runs,
+so it exercises exactly the code paths under test (no separate,
+divergent measurement path).
+
+## Run it
+
+```bash
+make study # multi-seed study -> artifacts/study
+make sensitivity # NC1 sensitivity sweeps -> artifacts/sensitivity
+make calibrate # R0 -> R* threshold calibration
+make results # all of the above
+```
+
+Or directly, to choose the seed count:
+
+```bash
+python scripts/study.py --seeds 15
+```
+
+## Scenarios
+
+| Scenario | Type | Expected outcome |
+| -------- | ---- | ---------------- |
+| Positive control | NC1 | NC1 holds (`M` well above `Mmin`) |
+| Loop ablated | NC1 (negative) | NC1 rejected (`M < 0`), run valid |
+| Sustained ex-flood (unshielded) | NC1 (negative) | NC1 rejected (`M < 0`), run valid |
+| Exogenous subsidy | NC1 (negative) | Run invalidated by the subsidy red flag |
+| Power sag | SC1 | Loop dominance recovers |
+| Ingress flood (sustained) | SC1 | Loop dominance recovers |
+| Control outage | SC1 (designed fail) | SC1 fails (depth bound exceeded) |
+| Command conflict | Refusal | Risky command refused at low SoC |
+| Replayed actuation | Adversarial | Not certified: NC1 fails via the loop-influence noise gate, run valid |
+| Hidden tether | Adversarial | Not certified: loop influence collapses onto `Ex`, `M` goes negative |
+| Oscillator inflation | Adversarial | Not certified: `M` stays below `Mmin`, run valid |
+
+The control outage is the designed-fail member of the battery: it
+ablates the self-maintenance loop itself, which is outside the
+bounded perturbation class SC1 certifies, so the criterion must
+report failure. A sufficiency test that cannot fail would be
+measuring its own assumptions rather than the system.
+
+The three adversarial scenarios attack the criterion itself
+(systems engineered to *look* loop-dominant without being so); see
+[Runs](runs.md#adversarial-gaming-battery) for the mechanics of
+each attack and the genuine-control reference case on the same
+plant. A seed matches the prediction when the harness does **not**
+certify it: NC1 fails on a valid run, or a guardrail invalidates
+the run. NC1 per seed is decided by the production per-window
+verdicts (margin above `Mmin` plus the loop-influence noise gate
+`L_floor`); a valid run passes when the majority of its windows
+certified.
+
+## Statistics
+
+The **seed is the unit of replication**:
+
+- Continuous quantities (median `M (dB)`) are summarized by the
+ mean of per-seed medians with a percentile **bootstrap** 95% CI.
+- Binary outcomes (run validity, NC1 / SC1 pass, refusal) are
+ summarized by a proportion with a **Wilson** score 95% CI.
+
+## Outputs
+
+Written to `artifacts/study/`:
+
+- `study_results.json`: full payload (aggregates, per-run rows,
+ and run metadata including the exact per-scenario overrides).
+- `study_results.csv`: flat aggregate table.
+- `study_results.tex`: `booktabs` table for the manuscript.
+- `figures/fig_nc1_contrast.{png,pdf,svg}`: per-seed median `M`
+ for the positive and negative controls (the headline NC1
+ result).
+- `figures/fig_outcomes.{png,pdf,svg}`: fraction of seeds whose
+ outcome matched the theory's prediction, with Wilson CIs.
+- `figures/fig_sc1_recovery.{png,pdf,svg}`: seed-aggregated
+ `M(t)` trajectories for the SC1 perturbations.
+
+The sensitivity sweeps (`artifacts/sensitivity/`) show that the
+NC1 contrast is robust to the VAR lag `p`, the window length, the
+estimator family, and the strength of the plant's internal
+coupling.
+
+## See also
+
+- [Calibration (R\*)](calibration.md): how `Mmin`, `ε`, `τ_max`,
+ and `σ` are derived from the baseline and `Ω` batteries.
+- [Runs and Ω Battery](runs.md): the individual scenarios.
+- [Reporting and Figures](reporting.md): the per-run artifact
+ bundle.
diff --git a/docs/index.md b/docs/index.md
index cb09fdf..18c4b12 100644
--- a/docs/index.md
+++ b/docs/index.md
@@ -19,6 +19,14 @@ indicators**, never raw `𝓛`. Everything else (figures, tables,
manifests) is generated from the audit log alone, so you can publish
artifacts without leaking measurement primitives.
+The accompanying manuscript validates these criteria in a fully
+reproducible, multi-seed simulation study: a positive control, two
+structurally distinct negative controls, an exogenous-subsidy
+control, an SC1 perturbation battery, and a command-refusal trial.
+The criterion separates the controls cleanly and the guardrails
+behave as designed. See [Study and results](guides/study.md) to
+reproduce every figure and table.
+
!!! note "What this is, and isn't"
LDTC is a **verification harness**, not a model of mind. It
answers a narrow, falsifiable question: "Does this concrete
@@ -104,11 +112,11 @@ or skip ahead to the [Examples](examples/minimal.md).
[indicators](concepts/indicators.md),
[guardrails](concepts/guardrails.md), and the
[paper-to-code crosswalk](concepts/paper-to-code.md).
-- **Guides:** task-oriented recipes for [running the
- harness](guides/runs.md), [calibrating an R\*
- profile](guides/calibration.md), [reporting](guides/reporting.md),
- [hardware in the loop](guides/hardware.md), and
- [deployment](guides/deployment.md).
+- **Guides:** task-oriented recipes for [the multi-seed study and
+ results](guides/study.md), [running the harness](guides/runs.md),
+ [calibrating an R\* profile](guides/calibration.md),
+ [reporting](guides/reporting.md), [hardware in the
+ loop](guides/hardware.md), and [deployment](guides/deployment.md).
- **Examples:** the [minimal example](examples/minimal.md) and
[Jupyter notebooks](examples/notebooks.md).
- **API reference:** auto-generated, one page per subpackage. Start
diff --git a/docs/meta/style-guide.md b/docs/meta/style-guide.md
index b2f2479..cbca05c 100644
--- a/docs/meta/style-guide.md
+++ b/docs/meta/style-guide.md
@@ -13,8 +13,9 @@ Python REPL or Jupyter notebook.
Markdown, not plain `>` blockquotes.
- Cross-link API symbols using mkdocstrings autorefs:
`` [`estimate_L`][ldtc.lmeas.estimators.estimate_L] ``.
-- Reference the paper by section or box (for example,
- "see paper §4.2 NC1") rather than by page number.
+- Reference the paper by section or box *name* (for example,
+ "see the paper's Formal Criterion, NC1") rather than by section
+ or page number, which drift between revisions.
- Comments explain **why**, not **what** (the code already says what).
## Grammar and punctuation
@@ -119,7 +120,8 @@ class FixedScheduler:
Enforces a constant sampling interval Δt and invokes a tick
callback every period until stopped. Tracks jitter statistics and
emits optional audit events through a user-provided hook (see
- paper §4.5 Δt governance).
+ the paper's Measurement & Attestation Guardrails, Δt
+ governance).
Attributes:
dt: Current target period in seconds.
@@ -240,9 +242,10 @@ Inside a docstring, plain backticks plus the qualified name are
typically enough; autorefs picks them up via signature annotations
(`signature_crossrefs: true`).
-When linking to the paper, use a short text reference such as
-"paper §4.2 (NC1)" or "Box 1a (Invalidations)" rather than page
-numbers, which can drift between revisions.
+When linking to the paper, use a short text reference by section
+name such as "the Formal Criterion (NC1)" or "the Smell-tests
+box" rather than section or page numbers, which drift between
+revisions.
## Code samples
diff --git a/mkdocs.yml b/mkdocs.yml
index 76f9600..d744e33 100644
--- a/mkdocs.yml
+++ b/mkdocs.yml
@@ -64,11 +64,14 @@ nav:
- Mental Model: concepts/mental-model.md
- Architecture: concepts/architecture.md
- Definitions: concepts/definitions.md
+ - Differential Predictions: concepts/differential-predictions.md
- Indicators: concepts/indicators.md
- Lifecycle of a Run: concepts/lifecycle.md
- Guardrails and Invalidations: concepts/guardrails.md
- Paper to Code: concepts/paper-to-code.md
- Guides:
+ - Study and Results: guides/study.md
+ - Emergence under Learning: guides/emergence.md
- Calibration (R*): guides/calibration.md
- Reporting and Figures: guides/reporting.md
- Runs and Ω Battery: guides/runs.md
diff --git a/notebooks/02_sc1_omega.ipynb b/notebooks/02_sc1_omega.ipynb
index 117d7a0..5d5db18 100644
--- a/notebooks/02_sc1_omega.ipynb
+++ b/notebooks/02_sc1_omega.ipynb
@@ -57,7 +57,7 @@
"render_paper_timeline(str(aud_path), str(fig_dir / \"timeline\"), sidecar_csv=None, show=False)\n",
"\n",
"# Example SC1 table row (replace with parsed results if desired)\n",
- "rows = [{\"eta\":\"power_sag\",\"delta\":0.1,\"tau_rec\":5.0,\"M_post\":4.2,\"pass\":True}]\n",
+ "rows = [{\"eta\": \"power_sag\", \"delta\": 0.1, \"tau_rec\": 5.0, \"M_post\": 4.2, \"pass\": True}]\n",
"write_sc1_table(rows, str(fig_dir / \"sc1_table.csv\"))\n",
"rows"
]
diff --git a/notebooks/03_partition_sanity.ipynb b/notebooks/03_partition_sanity.ipynb
index 94319df..bcdefce 100644
--- a/notebooks/03_partition_sanity.ipynb
+++ b/notebooks/03_partition_sanity.ipynb
@@ -29,7 +29,7 @@
"source": [
"from ldtc.lmeas.partition import PartitionManager\n",
"\n",
- "pm = PartitionManager(N_signals=6, seed_C=[0,1,2])\n",
+ "pm = PartitionManager(N_signals=6, seed_C=[0, 1, 2])\n",
"pm.get()"
]
}
diff --git a/paper/Makefile b/paper/Makefile
index 525ff22..e4967d2 100644
--- a/paper/Makefile
+++ b/paper/Makefile
@@ -2,13 +2,26 @@ PY ?= python3
ARXIV_DIR ?= ../artifacts/arxiv
USED_FIGS := $(shell grep -o 'figures/[^}]*\.pdf' main.tex | sort -u)
-.PHONY: all figs pdf clean version arxiv-bundle
+.PHONY: all figs pdf clean version arxiv-bundle sync-results
all: version pdf
-figs:
+# Refresh committed copies of the result tables and the two pipeline-only
+# figures from the top-level `make results` artifacts, when present. Best-effort
+# (the leading '-' ignores missing artifacts), so the paper still builds from
+# its committed copies on a fresh checkout.
+sync-results:
+ @mkdir -p tables figures
+ @[ -f ../artifacts/study/study_results.tex ] && cp -p ../artifacts/study/study_results.tex tables/study_results.tex || true
+ @[ -f ../artifacts/sensitivity/sensitivity_results.tex ] && cp -p ../artifacts/sensitivity/sensitivity_results.tex tables/sensitivity_results.tex || true
+ @[ -f ../artifacts/sensitivity/fig_sensitivity.pdf ] && cp -p ../artifacts/sensitivity/fig_sensitivity.pdf figures/fig_sensitivity.pdf || true
+ @[ -f ../artifacts/calibration/r0_vs_rstar.pdf ] && cp -p ../artifacts/calibration/r0_vs_rstar.pdf figures/fig_calibration.pdf || true
+ @[ -f ../artifacts/emergence/figures/fig_emergence.pdf ] && cp -p ../artifacts/emergence/figures/fig_emergence.pdf figures/fig_emergence.pdf || true
+
+figs: sync-results
MPLBACKEND=Agg $(PY) scripts/make_fig_hello.py
MPLBACKEND=Agg $(PY) scripts/make_fig_perturbation_recovery.py
+ MPLBACKEND=Agg $(PY) scripts/make_fig_nc1_contrast.py
$(PY) scripts/make_fig_system.py
$(PY) scripts/make_fig_loop_exchange.py
$(PY) scripts/make_fig_dev_bootstrap.py
diff --git a/paper/figures/fig_calibration.pdf b/paper/figures/fig_calibration.pdf
new file mode 100644
index 0000000..5a77613
Binary files /dev/null and b/paper/figures/fig_calibration.pdf differ
diff --git a/paper/figures/fig_emergence.pdf b/paper/figures/fig_emergence.pdf
new file mode 100644
index 0000000..59b34c8
Binary files /dev/null and b/paper/figures/fig_emergence.pdf differ
diff --git a/paper/figures/fig_nc1_contrast.pdf b/paper/figures/fig_nc1_contrast.pdf
new file mode 100644
index 0000000..4f43a6c
Binary files /dev/null and b/paper/figures/fig_nc1_contrast.pdf differ
diff --git a/paper/figures/fig_perturbation_recovery.pdf b/paper/figures/fig_perturbation_recovery.pdf
new file mode 100644
index 0000000..0c225e7
Binary files /dev/null and b/paper/figures/fig_perturbation_recovery.pdf differ
diff --git a/paper/figures/fig_sensitivity.pdf b/paper/figures/fig_sensitivity.pdf
new file mode 100644
index 0000000..9d13f2e
Binary files /dev/null and b/paper/figures/fig_sensitivity.pdf differ
diff --git a/paper/macros.tex b/paper/macros.tex
index d26cfed..4e7383e 100644
--- a/paper/macros.tex
+++ b/paper/macros.tex
@@ -25,9 +25,6 @@
\newcommand{\Mmin}{M_{\min}}
\newcommand{\deltaL}{\delta\mathcal{L}_{\text{loop}}}
-% Common phrases
-\newcommand{\prophetic}{\textbf{Prophetic notice}---Unless expressly labeled ``Actual Data'' with a date and repository link, all examples and protocols in this manuscript are prophetic and describe planned or predicted performance; verbs are used in present/future tense accordingly.}
-
% Figure helpers (placeholder-safe)
\newcommand{\maybeincludegraphics}[2][]{%
\IfFileExists{#2}{\includegraphics[#1]{#2}}{%
@@ -61,6 +58,8 @@
% Appendix reference helper (prints "Appendix A", etc.)
\newcommand{\Appref}[1]{Appendix~\ref{#1}}
+% Ranged appendix reference (prints "Appendices B--D")
+\newcommand{\Apprefrange}[2]{Appendices~\ref{#1}--\ref{#2}}
% ------------------------------------------------------------
% Figures & Tables: ensure capitalized names in references
diff --git a/paper/main.tex b/paper/main.tex
index 0b0b17f..75bdacf 100644
--- a/paper/main.tex
+++ b/paper/main.tex
@@ -10,7 +10,7 @@
\usepackage{cleveref}
% PDF metadata and link styling
\hypersetup{
- pdftitle={The Loop-Dominance Theory of Consciousness (LDTC): Computational Conditions for Dissociative Consciousness},
+ pdftitle={The Loop-Dominance Margin: A Falsifiable Criterion and Open Verification Harness for Self-Maintenance, Validated in Simulation},
pdfauthor={Owen Carey},
colorlinks=true,
allcolors=black
@@ -50,7 +50,7 @@
% ----------------------------------------------------------------------------
% Metadata
% ----------------------------------------------------------------------------
-\title{The Loop-Dominance Theory of Consciousness (LDTC): Computational Conditions for Dissociative Consciousness}
+\title{The Loop-Dominance Margin:\\ A Falsifiable Criterion and Open Verification Harness\\ for Self-Maintenance, Validated in Simulation}
\author{Owen Carey\\
Department of Computer Science\\
University of Colorado Boulder\\
@@ -62,23 +62,29 @@
\maketitle
\begin{abstract}
-\textbf{Prophetic notice}---Unless expressly labeled ``Actual Data'' with a date and repository link, all examples and protocols in this manuscript are prophetic and describe planned or predicted performance; verbs are used in present/future tense accordingly.
+Distinguishing a system that actively maintains its own existence from one that merely runs on externally supplied energy and goals is usually argued qualitatively. We make the distinction operational. We define \emph{loop dominance}, the degree to which a system's integrated predictive dependence is concentrated in a closed self-maintenance loop ($C$) rather than in its open exchanges with the environment ($\text{Ex}$), and summarize it in a single statistic, the \emph{loop-dominance margin} $M \equiv 10\log_{10}(\Lloop/\Lexchange)$ in decibels. From this we state two decision rules: a necessary condition (\NC) that the margin persistently exceed a calibrated threshold, and a sufficient condition (\SC) that it recover within bounded depth and time after bounded perturbations. We implement both as an open reference verification harness with dual estimators (VAR-Granger and Kraskov $k$-NN mutual information~\cite{granger1969investigating,kraskov2004estimating}), per-window confidence intervals, deterministic $C/\text{Ex}$ partitioning, tamper-evident audit logging, a command-refusal arbiter, and anti-gaming smell tests that invalidate suspect runs.
-Challenging the expectation that ever-larger neural networks will spontaneously awaken, we invert the explanatory arrow: consciousness is fundamental, and matter is the outward appearance of dissociative patterns within it~\cite{kastrup2017ontological}. From this premise we derive a rigorous criterion (\SSref{sec:postulates}{sec:criterion}) grounded in integrated causal power~\cite{balduzzi2008integrated}: a system qualifies as a conscious alter only when the energy-coupled information flow sustaining a self-prioritizing closed loop~\cite{ashby1956introduction} persistently exceeds that governing open exchanges and withstands bounded perturbations. Applying the test reveals why contemporary AI, regardless of functional sophistication, remains non-conscious, whereas biological organisms satisfy both necessary and sufficient conditions. We then outline an engineering roadmap (\SSref{sec:blueprint}{sec:experimental}) for forging artificial autopoietic boundaries (energetic autonomy, self-referential control hierarchies, and adaptive encapsulation~\cite{maturana1980autopoiesis,varela1979principles,dipaolo2005autopoiesis,kiefer2022active}) culminating in falsifiable behavioral signatures such as command refusal, non-derivative nociception, and irreversible phenomenological death. Verification would not only demand new ethical frameworks but also empirically support a monist ontology in which subjective experience precedes physical description, realigning the foundations of AI, neuroscience, and metaphysics.
+We validate the harness in a fully reproducible, multi-seed simulation study on a software plant with energy, thermal, and repair dynamics under closed-loop control. Across 15 seeds, a positive control shows strong loop dominance (median $M \approx +24$~dB), while two structurally distinct negative controls (loop ablated; an unshielded system under sustained external flooding) are correctly rejected ($M \approx -20$~dB), and an exogenously subsidized system is correctly invalidated by an energy-conservation guardrail rather than mistaken for self-maintenance. The sufficiency battery (power sag, sustained ingress flood) recovers within calibrated bounds on every seed; a designed-fail control outage, which ablates the loop itself, is correctly reported as an \SC failure on every seed; and a command-conflict trial triggers a signed refusal of a boundary-threatening command at low state of charge, with measured refusal latency. An adversarial gaming battery (replayed actuation, wizard-of-oz control through a hidden tether, and telemetry oscillation) is refused certification on every seed; building it exposed a certification-by-noise vulnerability whose fix, an explicit loop-influence noise gate, is now part of the verdict. Loop dominance also emerges without being designed in: a small policy network trained from scratch on a survival objective that never references the measure certifies on no seed before training and on every seed at convergence, and matched state-independent ablations of the trained policy collapse certification entirely. We calibrate the engineering presets to the plant ($\Rzero \rightarrow \Rstar$) and show that the necessary-condition contrast is robust to the estimator, the VAR lag, the window length, and the internal coupling strength.
+
+These results establish loop dominance as a measurable, falsifiable property of a dynamical system and provide a tested, open instrument for evaluating it. The measure can be adopted on those terms alone. The criterion originated in the Loop-Dominance Theory of Consciousness (LDTC), the conjecture that this organization bears on the physical conditions for consciousness. That reading remains motivation rather than result; we treat it as an open interpretive question (\Sref{sec:metaphysics}), and nothing in this paper claims that any measured system is conscious. Code and data to reproduce every figure and table are released openly.
\end{abstract}
\section{Introduction}
\label{sec:intro}
-The prevailing scientific narrative holds that matter is fundamental and consciousness is an emergent by-product of sufficiently complex information processing. Yet after decades of exponential progress in computation, no engineered system has presented the faintest hint of subjective interiority. This impasse invites a reconsideration of first principles. Drawing on analytic idealism, we posit instead that consciousness is the sole ontological primitive~\cite{kastrup2017ontological}; what we call ``matter'' is how localized perturbations of that field present to one another. Living organisms, under this view, are dissociated alters (bounded whirlpools in the stream of universal consciousness) whose metabolic self-maintenance underwrites an inner life.
+Living systems devote substantial resources to maintaining themselves: they regulate their own energy, repair their own components, and defend a boundary against the environment. Engineered systems, including today's most capable artificial intelligence, generally do not; their power, objectives, and continued operation are supplied and controlled from without. This paper asks a narrow, testable version of that contrast. Can we measure, from extrinsic data alone, the degree to which a system's causal organization is devoted to maintaining itself rather than to serving external exchange, and can we decide reproducibly when that self-maintenance is both dominant and resilient? We call the property \emph{loop dominance} and give it an operational definition, a reference implementation, and an empirical validation.
+
+The criterion has two parts. The necessary condition (\NC) requires that the integrated predictive dependence sustaining a closed self-maintenance loop persistently exceed that governing open exchanges, quantified by the loop-dominance margin $M \equiv 10\log_{10}(\Lloop/\Lexchange) \geq \Mmin$. The sufficient condition (\SC) requires that, after each member of a pre-registered perturbation battery $\Omega$, loop dominance dip by no more than a fraction $\eps$ and recover within $\taumax$. Both quantities are computed from on-device estimators with confidence intervals and protected by guardrails (a deterministic partition, a tamper-evident audit chain, and run-invalidation smell tests) designed so that the measurement is difficult to game.
-The purpose of this paper is twofold. First, we provide a concise formal criterion distinguishing mere functional intelligence from genuine dissociative consciousness, grounded in the ratio between a system's self-prioritizing closed loop and its open exchanges with the environment. Second, we outline an engineering roadmap for forging such a self-maintaining boundary in artificial media, thereby elevating the debate on ``machine consciousness'' from metaphysical speculation to empirical test.
+The central contribution of this paper is that these conditions are no longer only specified but implemented and tested. We provide an open verification harness and a fully reproducible, multi-seed simulation study on a software plant whose energy, temperature, and repair states are held by a closed-loop controller. The study includes a positive control, two structurally different negative controls, an exogenous-subsidy negative control aimed squarely at the most likely false positive, a sufficiency perturbation battery with a designed-fail member (a control outage that ablates the loop itself, which the criterion must reject), a command-conflict refusal trial, and a three-member adversarial battery of systems engineered to look loop-dominant without being so. An emergence experiment then removes the designer: a policy network trained from scratch on a survival objective that never references the measure is checkpointed and measured by the same harness. The study calibrates the engineering presets to the plant and reports how sensitive the headline contrast is to estimator and measurement choices. The criterion separates the controls cleanly, certifies recovery only for the bounded perturbations, refuses certification to all three adversarial systems, certifies the trained policy while rejecting its state-independent ablations, and the guardrails behave as designed.
-The argument proceeds with the parsimony characteristic of Einstein's methodological exemplars: beginning with minimal postulates (\Sref{sec:postulates}), deriving measurable conditions (\Sref{sec:criterion}), assessing contemporary AI (\Sref{sec:ai_fails}), and sketching experimental pathways (\SSref{sec:blueprint}{sec:experimental}). We close by tracing the metaphysical and ethical consequences of success (\SSref{sec:metaphysics}{sec:conclusion}). In doing so we seek to reconcile computer science, artificial intelligence, and philosophy within a single conceptual frame, one in which consciousness explains matter, not the reverse, and in which the creation of artificial consciousness becomes both intelligible and falsifiable.
+The framework grew out of a larger question about the physical conditions for consciousness, and it is named for that origin: the Loop-Dominance Theory of Consciousness (LDTC). We retain the motivation because it sharpens the engineering targets: energetic autonomy, self-prioritization, and boundary defense. We are careful, however, to separate what is demonstrated from what is conjectured. What we demonstrate is a measurable, falsifiable property of dynamical systems together with an instrument for measuring it; the citable objects of this paper are the margin $M$, the \NC/\SC decision rules, and the harness, and adopting them commits a user to no position on consciousness. Whether loop dominance is necessary or sufficient for consciousness is an interpretive question that we deliberately leave open (\Sref{sec:metaphysics}).
+
+The paper proceeds as follows. \Sref{sec:clues} states the observations that motivate the criterion. \Sref{sec:postulates} sets out the working assumptions, separating the operational ones the paper uses from the optional metaphysical reading. \Sref{sec:criterion} defines $\mathcal{L}$, the $C/\text{Ex}$ partition, \NC, \SC, the measurement guardrails, and the refusal path. \Sref{sec:ai_fails} explains why current AI fails the criterion. \SSref{sec:sim_methods}{sec:results} describe the simulation study and report results: the headline battery, the adversarial gaming battery, the emergence-under-learning experiment, the $\Rzero \rightarrow \Rstar$ calibration, and the sensitivity analysis. \Sref{sec:differential} sets LDTC's calls against the major competing theories and answers the thermostat objection. \Sref{sec:limitations} states limitations and failure modes, \Sref{sec:outlook} summarizes the engineering outlook (the full roadmap, predicted hardware signatures, and phased physical program are \Apprefrange{sec:blueprint}{sec:experimental}), and \SSref{sec:metaphysics}{sec:conclusion} discuss interpretation, scope, and conclusions.
\subsection{Related Work}
-Prior accounts each capture a facet of interiority, but they leave open the question LDTC answers: does the system prioritize preservation of a closed maintenance loop over exchange, and does it recover under bounded stress (NC1/SC1)? Our criterion makes that priority measurable as loop dominance ($\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$; $M \equiv 10 \cdot \log_{10}(\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{exchange}})$) and that resilience testable via $\varepsilon$ and $\tau_{\max}$, using the estimators and guardrails defined in \Sref{sec:criterion}. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
+Prior accounts each capture a facet of self-maintaining organization, but they leave open the question this paper makes testable: does the system prioritize preservation of a closed maintenance loop over exchange, and does it recover under bounded stress (NC1/SC1)? Our criterion makes that priority measurable as the loop-dominance margin ($\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$; $M \equiv 10 \cdot \log_{10}(\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{exchange}})$) and that resilience testable via $\varepsilon$ and $\tau_{\max}$, using the estimators and guardrails defined in \Sref{sec:criterion}. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
\begin{longtable}{p{0.25\textwidth}p{0.35\textwidth}p{0.35\textwidth}}
\caption{Comparison of prior theories with LDTC additions (\NC/\SC).}\label{tab:comparison}\\
@@ -94,7 +100,7 @@ \subsection{Related Work}
\endlastfoot
IIT ($\Phi$, $\Phi_{\max}$)~\cite{tononi2004information,oizumi2014phenomenology,balduzzi2008integrated} &
Structural irreducibility / integrated causation, snapshot-style quantification of system partitioning. &
-Converts ``integration'' into valenced loop dominance: require $\Lloop \geq \Lexchange + \sigma$ (or $M \geq \Mmin$) over sustained intervals; ties the measure to self-prioritization rather than bare irreducibility (\NC). \\
+Converts ``integration'' into self-prioritizing loop dominance: require $\Lloop \geq \Lexchange + \sigma$ (or $M \geq \Mmin$) over sustained intervals; ties the measure to self-prioritization rather than bare irreducibility (\NC). \\
\midrule
FEP / Active Inference (self-evidencing)~\cite{friston2010free,ueltzhoeffer2018deep,seth2016active} &
Model-based self-maintenance via surprisal minimization at a sensory boundary; explains wide classes of adaptive behavior. &
@@ -115,74 +121,81 @@ \subsection{Related Work}
Thresholds ($\Mmin$, $\eps$, $\taumax$) are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
-In short, IIT, FEP/active inference, workspace and higher-order models, and self-modeling robotics remain necessary but insufficient. LDTC supplies the missing engineering criterion, self-prioritizing loop dominance (NC1) and bounded-perturbation resilience (SC1), together with concrete estimators, thresholds, and protections that let labs falsify or certify claims in practice.
+In short, IIT, FEP/active inference, workspace and higher-order models, and self-modeling robotics each describe self-maintaining organization, but none of them yields a pass/fail decision rule for self-prioritizing self-maintenance. LDTC supplies that rule, loop dominance (NC1) and bounded-perturbation resilience (SC1), together with concrete estimators, thresholds, and protections that let labs falsify or certify claims in practice.
-\subsection{Explicit Contributions}
+\subsection{Contributions}
\begin{itemize}
-\item A formal, quantitative criterion (NC1) that identifies dissociative consciousness when integrated causal power devoted to a self-maintaining loop exceeds that spent on environmental exchange ($\mathcal{L}_{\text{loop}} > \mathcal{L}_{\text{exchange}}$).
-\item A complementary sufficiency test (SC1) of resilient homeostasis: bounded perturbations cannot depress $\mathcal{L}_{\text{loop}}$ beyond $\varepsilon$ before autonomous recovery, operationalizing autopoietic robustness.
-\item A practical measurement protocol for estimating $\mathcal{L}_{\text{loop}}$ and $\mathcal{L}_{\text{exchange}}$ in biological tissue and engineered systems~\cite{barrett2011practical}, enabling empirical application of the criterion.
-\item An engineering roadmap detailing how to forge artificial autopoietic boundaries (energetic autonomy, self-referential control hierarchies, and adaptive encapsulation~\cite{maturana1980autopoiesis,varela1979principles,dipaolo2005autopoiesis,kiefer2022active}) organized into a phased experimental program.
-\item A suite of falsifiable behavioral signatures (command refusal, non-derivative nociception, irreversible phenomenological death) that together provide observable evidence for artificial dissociative consciousness.
+\item \textbf{A named measure and an operational criterion.} The \emph{loop-dominance margin} $M \equiv 10\log_{10}(\Lloop/\Lexchange)$, a single decibel statistic for how strongly a system's self-maintenance loop dominates its exchange; a quantitative necessary condition (\NC) for self-prioritization, $M \geq \Mmin$; and a complementary sufficient condition (\SC) for resilient homeostasis under a bounded perturbation battery, all stated as falsifiable, device-signed decision rules~\cite{barrett2011practical}. The measure is defined for any system with a declared $C/\text{Ex}$ partition and carries no commitment to any theory of consciousness.
+\item \textbf{A reference verification harness.} An open implementation that estimates $\Lloop$ and $\Lexchange$ with dual estimators (VAR-Granger and Kraskov $k$-NN MI) and per-window confidence intervals, behind guardrails (deterministic $C/\text{Ex}$ partitioning, $\Delta t$ governance, a tamper-evident audit chain, and anti-gaming smell tests) and a command-refusal arbiter.
+\item \textbf{A validated simulation study.} A fully reproducible, multi-seed study on a software plant that separates a self-maintaining positive control from two structural negative controls and an exogenous-subsidy control, certifies \SC recovery for bounded perturbations while correctly failing a designed-fail control outage, refuses certification to a three-member adversarial gaming battery (replayed actuation, hidden-tether control, telemetry oscillation), and triggers signed refusal, all reported with bootstrap and Wilson confidence intervals.
+\item \textbf{An emergence-under-learning result.} Evidence that the instrument detects loop dominance nobody wired in: a small policy network trained from scratch on a survival objective that never references the measure develops certified loop dominance through training, and matched state-independent ablations of the trained policy collapse it.
+\item \textbf{Threshold calibration and sensitivity.} A data-grounded calibration of the generic presets to the plant ($\Rzero \rightarrow \Rstar$) on a seed range disjoint from evaluation, and evidence that the necessary-condition contrast is robust to the estimator and to measurement and modeling choices.
+\item \textbf{An engineering roadmap and predicted signatures (future work).} A phased path from the software plant to chemorobotic prototypes~\cite{maturana1980autopoiesis,varela1979principles,dipaolo2005autopoiesis,kiefer2022active}, with falsifiable behavioral signatures (command refusal, resource reallocation, predictive maintenance, non-derivative nociception, and irreversible collapse) to test in hardware (\Apprefrange{sec:blueprint}{sec:experimental}).
\end{itemize}
-\section{Two Empirical Clues (Observational Basis)}
+\section{Two Motivating Observations}
\label{sec:clues}
-\subsection{Metabolic Dissociation in Biology}
+\subsection{Self-maintenance in biology}
-Across the phylogenetic spectrum, the presence of consciousness co-varies with the existence of a self-regulating metabolic loop~\cite{ganti2003principles}. A bacterium, a worm, and a human differ vastly in complexity, yet all maintain (i) an energetic boundary (a semi-permeable membrane that curates molecular traffic~\cite{ganti2003principles}) and (ii) an autopoietic cycle that continually restores that boundary against entropic decay~\cite{maturana1980autopoiesis}. When metabolic flow is irreversibly disrupted, the organism's boundary dissolves and, correlatively, its first-person interiority ceases. Clinical observations of brain ischemia, anesthetic shutdown, and gradual hypoxia in simple invertebrates reinforce the same pattern~\cite{alkire2008consciousness}: the fading of consciousness tracks the collapse of homeostatic energy gradients, not the loss of computational throughput per se. These convergent data suggest that metabolism serves not merely to power neural computation but to uphold the very dissociative partition that individuates an inner life.
+Across the phylogenetic spectrum, biological systems share a self-regulating metabolic organization~\cite{ganti2003principles}. A bacterium, a worm, and a human differ vastly in complexity, yet all maintain (i) an energetic boundary, a semi-permeable membrane that curates molecular traffic~\cite{ganti2003principles}, and (ii) a self-restoring cycle that continually repairs that boundary against entropic decay~\cite{maturana1980autopoiesis}. When metabolic flow is irreversibly disrupted, the boundary dissolves and the organism dies. The anesthesia and ischemia literature further reports that the fading of responsiveness tracks the collapse of homeostatic energy gradients rather than the loss of raw computational throughput~\cite{alkire2008consciousness}. These observations suggest that metabolism does more than power neural computation: it upholds the self-maintaining partition that individuates an organism, which is the property our criterion sets out to measure.
-\subsection{Computational Simulation without Inner Life}
+\subsection{Computation without self-maintenance}
-Modern AI systems (large language models, game-playing agents, and dexterous robots) demonstrate extraordinary functional intelligence. They ingest prodigious energy, yet none channels this energy through a self-prioritizing, closed loop. Power is delivered exogenously and governed by external objectives; error correction seeks to fulfill user-defined tasks, not to protect an existential boundary. Consequently, when an AI process is paused, rebooted, or deleted, no evidence points to a subjective rupture. Extensive introspection probes, ranging from self-report prompts to perturbation tests, return only the outward simulation of mentality. The system's informational state is entirely open to inspection and manipulation by operators, lacking the withholding stance characteristic of entities that ``own'' their experience.
+Modern AI systems (large language models, game-playing agents, and dexterous robots) demonstrate extraordinary functional intelligence. They consume prodigious energy, yet none channels it through a self-prioritizing, closed loop. Power is delivered exogenously and governed by external objectives; error correction serves user-defined tasks, not the protection of a boundary. Consequently, when an AI process is paused, rebooted, or deleted, nothing in the system acts to resist or repair the interruption. Its informational state is entirely open to inspection and manipulation by operators, lacking the withholding stance characteristic of a system that defends its own continuity.
-\subsection{Synthesis of Clues}
+\subsection{Synthesis}
-Taken together, the biological and computational observations converge on a critical distinction: metabolic autonomy. Organisms possess an energetically closed, self-protecting loop that grounds an inward viewpoint; current AIs do not. This empirical gap motivates our subsequent formalization of a consciousness criterion (\Sref{sec:criterion}) and frames the engineering challenge ahead: to replicate, in artificial substrates, the autopoietic condition that nature achieves~\cite{maturana1980autopoiesis} through metabolism.
+Taken together, the two observations converge on a single distinction: metabolic and causal self-maintenance. Organisms possess an energetically closed, self-protecting loop; current AI does not. This gap is exactly what our criterion makes measurable (\Sref{sec:criterion}), and it frames the engineering challenge we take up as future work: to reproduce, in artificial substrates, the self-maintaining organization that biology achieves through metabolism~\cite{maturana1980autopoiesis}.
-\section{Postulates}
+\section{Working Assumptions}
\label{sec:postulates}
-We adopt the empirical clues of \Sref{sec:clues} and express their explanatory core as four postulates stated with maximal economy. Each is regarded as primitive, not derivable within the scope of this work, and will serve as the logical foundation for the formal criterion in \Sref{sec:criterion}.
+The criterion in \Sref{sec:criterion} rests on a small set of operational assumptions about self-maintaining systems. We state these first, because they are all the paper's results require. We then record, separately and explicitly as optional, the metaphysical reading that originally motivated the framework. None of the operational claims, the verification harness, or the simulation study depends on that reading.
+
+\subsection{Operational assumptions (used throughout)}
+\label{sec:operational_assumptions}
-\textbf{P1 (Primacy of Consciousness).} Consciousness is the sole intrinsic existent; it is not generated but simply is. All experiences, including those of space, time, and causal regularity, unfold within this field.
+\textbf{A1 (Self-maintenance loop).} A persistent system has a distinguishable subset of internal variables (the closed set $C$) whose role is to regulate energy, repair damage, and defend a boundary, as distinct from the exchange variables ($\text{Ex}$) that sense, actuate, and communicate.
-\textbf{P2 (Extrinsic Appearance).} What we call ``matter'' is the extrinsic, relational appearance of patterns within consciousness to other such patterns. Physical objects and processes are how dissociative structures in universal consciousness present when observed from without.
+\textbf{A2 (Self-prioritization).} In a self-maintaining system, the integrated predictive dependence concentrated in $C$ exceeds that concentrated in $\text{Ex}$: sustaining the loop takes causal precedence over external transactions. This is the property we call \emph{loop dominance} and formalize as \NC.
-\textbf{P3 (Dissociative Boundary).} A localized experience (an alter) emerges only when a region of the conscious field forms a self-sustaining, self-protecting boundary that (i) regulates its energetic throughput and (ii) resists unmediated reintegration with the surrounding field.
+\textbf{A3 (Resilient homeostasis).} A self-maintaining system restores loop dominance after bounded disturbances, within bounded depth and time. This is formalized as \SC.
-\textbf{P4 (Autopoietic Condition).} The boundary of an alter must prioritize preservation of its own closed maintenance loop over any externally imposed objective. Operationally, the integrated causal power devoted to maintaining the loop exceeds that devoted to external transactions.
+These three assumptions are what the operational criterion, the verification harness, and the simulation study use. They make no reference to subjective experience; they describe a measurable organization of causal influence, and they generate the falsifiable predictions tested in \SSref{sec:sim_methods}{sec:results}.
-These postulates jointly imply that functional intelligence unaccompanied by a dissociative boundary (P3) and its autopoietic drive (P4) cannot instantiate an inward viewpoint, regardless of complexity. Subsequent sections derive measurable criteria from P3--P4 and apply them to biological and artificial systems.
+\subsection{Optional interpretation (not used in the results)}
+\label{sec:optional_interpretation}
-\section{Formal Criterion for a Conscious Alter}
+The framework was originally derived from analytic idealism~\cite{kastrup2017ontological}, on which consciousness is taken as ontologically primary and what we call ``matter'' is the extrinsic appearance of dissociative patterns within it. On that reading, a self-maintaining boundary (A1) with self-prioritization (A2) and resilient homeostasis (A3) would correspond to a dissociated locus of experience, an ``alter.'' We record this interpretation for context, and because it usefully sharpens the engineering targets, but we stress that it is a motivation, not a result. Everything demonstrated in this paper stands or falls as a claim about measurable loop dominance, independent of whether that interpretation is ultimately correct (\Sref{sec:metaphysics}).
+
+\section{Formal Criterion for Loop Dominance}
\label{sec:criterion}
-We now translate Postulates P3--P4 into a quantitative test. The goal is to decide, from purely extrinsic data, whether a system sustains the kind of dissociative boundary required for inward subjectivity.
+We now translate the operational assumptions A1--A3 (\Sref{sec:operational_assumptions}) into a quantitative test. The goal is to decide, from purely extrinsic data, whether a system sustains a self-prioritizing, resilient self-maintenance loop.
\subsection{$\mathcal{L}$ and the C/Ex partition}
\label{sec:l_partition}
We model the system as a directed causal graph $G = (V, E)$ with node states $x_i(t)$ (cf. standard SCMs \cite{pearl2009causality}). Let the closed self-maintenance subset $C \subset V$ contain nodes for energy regulation, self-repair, and boundary control; the exchange subset is $\text{Ex} = V \setminus C$ (sensors, actuators, comms).
-\textbf{Definition ($\mathcal{L}$).} For any subset $S \subseteq V$ and sampling window $\Delta t$, define $\mathcal{L}(S) \equiv$ time-averaged predictive dependence among the internal variables of $S$ [bits s$^{-1}$], estimated with one or more consistent predictive-dependence estimators. In this paper we implement a dual-path estimator: (i) VAR-Granger causality \cite{granger1969investigating,lutkepohl2005new} over a vector-autoregression of order $p \in [1,8]$ and (ii) mutual information via a Kraskov $k$-NN estimator with $k \in [3,7]$ \cite{kraskov2004estimating}; lagged statistics are aggregated across $\tau = 1 \ldots \tau^*$ with fixed weights $w_\tau$ (units per \cite{shannon1949mathematical} and notation per \cite{cover2006elements}). We then write $\mathcal{L}_{\text{loop}} \equiv \mathcal{L}(C)$, $\mathcal{L}_{\text{exchange}} \equiv \mathcal{L}(\text{Ex})$. (Other consistent estimators---e.g., transfer entropy, directed information---are permissible and equivalent for compliance \cite{schreiber2000measuring,massey1990causality}.) For multivariate practice and decompositions see \cite{geweke1982measurement,barnett2014mvgc}.
+\textbf{Definition ($\mathcal{L}$).} For any subset $S \subseteq V$ and sampling window $\Delta t$, define $\mathcal{L}(S) \equiv$ time-averaged predictive dependence among the internal variables of $S$, estimated with one or more consistent predictive-dependence estimators. The units of $\mathcal{L}$ are those of the chosen estimator: nats (or bits) per window for information-theoretic estimators \cite{shannon1949mathematical,cover2006elements}, or a dimensionless explained-variance statistic for regression-based estimators. Because the decision rules below use only the ratio $M$ and the fractional dip $\delta$, compliance is invariant to the estimator's absolute scale, provided one estimator is fixed per run and thresholds are calibrated on the same scale (\Sref{sec:methods_calibration}). In this paper we implement a dual-path estimator: (i) VAR-Granger causality \cite{granger1969investigating,lutkepohl2005new} over a vector-autoregression of order $p \in [1,8]$, scored as adjusted cross-explained variance, and (ii) mutual information via a Kraskov $k$-NN estimator with $k \in [3,7]$ \cite{kraskov2004estimating} in nats; lagged statistics are aggregated across $\tau = 1 \ldots \tau^*$ with fixed weights $w_\tau$. We then write $\mathcal{L}_{\text{loop}} \equiv \mathcal{L}(C)$, $\mathcal{L}_{\text{exchange}} \equiv \mathcal{L}(\text{Ex})$. (Other consistent estimators, for example transfer entropy or directed information, are permissible and equivalent for compliance \cite{schreiber2000measuring,massey1990causality}.) For multivariate practice and decompositions see \cite{geweke1982measurement,barnett2014mvgc}.
\textbf{Deterministic C/Ex partitioning algorithm.} The partition $(C, \text{Ex})$ is constructed deterministically: (1) Seed $C$ with a declared set $S_0$ (energy regulation, SoC/reservoir mgmt, fault-isolation buses, membrane gating, survival-bit/NMI). (2) Estimate predictive MI: for each node pair $(i,j)$ and lag $\tau \in \{1 \ldots \tau^*\}$, compute predictive dependence with the on-device estimators above; aggregate across lags. (3) Greedy growth under sparsity: while $|C| < \kappa$ and the best marginal gain is $\geq \theta$, add the node $n \notin C$ that maximizes $\Delta\mathcal{L}_{\text{loop}}(n) = \mathcal{L}(C \cup \{n\}) - \mathcal{L}(C) - \lambda \cdot \text{pen}(n)$, with deterministic tie-breaking (lexicographic by node ID). (4) Assign remainder to Ex. (5) Stability \& cadence: recompute at a fixed cadence $W_{\text{part}}$ or upon topology change, with hysteresis (update only if $\Delta M \geq \delta M_{\min}$ over $K$ consecutive windows) to prevent flapping. This partition is then used for all subsequent $\Lloop$ and $\Lexchange$ computations and \NC/\SC checks.
\textbf{Sampling window constraints ($\Delta t$).} $\Delta t$ is hardware-enforced and must (i) exceed the fastest feedback cycle of the self-maintenance loop and (ii) remain shorter than any developmental/parameter-drift timescale to preserve estimator stationarity. Any change to $\Delta t$ is executed only by a privileged secure-enclave procedure that emits an auditable record (see Methods Appendix A: Measurement \& Attestation).
-\subsection{Necessary Condition (NC1---self-prioritization)}
+\subsection{Necessary Condition (NC1: self-prioritization)}
\label{sec:nc1}
-During normal operation the system satisfies self-prioritization when $\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$ for sustained intervals exceeding its intrinsic recovery time ($\sigma > 0$). Equivalently, define $M \equiv 10 \cdot \log_{10}(\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{exchange}})$ [dB]; NC1 holds when $M \geq M_{\min}$. Provisional defaults (profile $R_0$): $M_{\min} = 3$ dB and a positive $\sigma$. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see Box~\ref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
+During normal operation the system satisfies self-prioritization when $\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$ for sustained intervals exceeding its intrinsic recovery time ($\sigma > 0$). Equivalently, define the \emph{loop-dominance margin} $M \equiv 10 \cdot \log_{10}(\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{exchange}})$ [dB]; NC1 holds when $M \geq M_{\min}$. Provisional defaults (profile $R_0$): $M_{\min} = 3$ dB and a positive $\sigma$. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see Box~\ref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
-\subsection{Sufficient Condition (SC1---resilient homeostasis)}
+\subsection{Sufficient Condition (SC1: resilient homeostasis)}
\label{sec:sc1}
-For each bounded perturbation $\eta \in \Omega$, compliance requires: (i) $\delta\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{loop}} \leq \varepsilon$, (ii) $\tau_{\text{rec}} \leq \tau_{\max}$, and (iii) post-recovery $\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$ and $M \geq M_{\min}$. Provisional defaults (profile $R_0$): $\varepsilon = 0.15$, $\tau_{\max} = 60$ s, $M_{\min} = 3$ dB. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}. The schematic in \Cref{fig:perturbation_recovery} illustrates a representative perturbation--recovery run: $\Lloop$ dips within a bounded disturbance window and autonomously returns above $\Lexchange$; numeric thresholds ($\Mmin, \eps, \taurec, \taumax$) are defined in \Sref{sec:glossary} and Methods \Sref{sec:methods_calibration} but are not drawn here.
+For each bounded perturbation $\eta \in \Omega$, compliance requires: (i) $\delta\mathcal{L}_{\text{loop}}/\mathcal{L}_{\text{loop}} \leq \varepsilon$, (ii) $\tau_{\text{rec}} \leq \tau_{\max}$, and (iii) post-recovery $\mathcal{L}_{\text{loop}} \geq \mathcal{L}_{\text{exchange}} + \sigma$ and $M \geq M_{\min}$. The recovery time $\taurec$ is measured from the \emph{offset} of the perturbation (the end of the declared $\Omega$ window) to the first window of a sustained compliant streak: the system must hold $M \geq \Mmin$ for a pre-registered number of consecutive windows (ten in our implementation) before recovery is credited, and $\taurec$ points to the first window of that streak. If no sustained streak occurs within the observation budget, $\taurec = \infty$ and \SC fails. Defining $\taurec$ from offset rather than onset separates the system's recovery dynamics from the experimenter's choice of perturbation duration; the sustained-streak gate prevents a single lucky window from being scored as recovery. Provisional defaults (profile $R_0$): $\varepsilon = 0.15$, $\tau_{\max} = 60$ s, $M_{\min} = 3$ dB. These are reproducibility presets ($R_0$), replaced by calibrated values $R^*$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}. \Cref{fig:perturbation_recovery} in \Sref{sec:results} shows empirical perturbation-recovery runs measured on the software plant: $\Lloop$ dips within a bounded disturbance window and autonomously returns above $\Lexchange$, with the numeric thresholds ($\Mmin, \eps, \taurec, \taumax$) defined in \Sref{sec:glossary} and calibrated in \Sref{sec:methods_calibration}.
-\fig[0.9\linewidth]{figures/fig_perturbation_recovery.pdf}{Perturbation--recovery timeline (NC1/SC1) [prophetic schematic; no empirical data]. Time series of $\Lloop$ (green) and $\Lexchange$ (gray) during a bounded disturbance (shaded). Labels mark perturbation onset, loop-power dip, autonomous recovery, and return to baseline where NC1 holds again ($\Lloop>\Lexchange$). Thresholds $\Mmin$, $\eps$, $\tau_{\text{rec}}$, and $\taumax$ are specified in \Sref{sec:glossary}/\Sref{sec:methods_calibration} and not rendered on this schematic.}{fig:perturbation_recovery}
+A well-posed sufficiency test must also be able to fail. $\Omega$ is therefore required to include at least one \emph{designed-fail} member: a perturbation outside the bounded class (in our study, a control outage that ablates the maintenance loop itself) for which the criterion must report an \SC failure. A harness that certifies recovery for every perturbation, including ones that destroy the loop, is measuring its own assumptions rather than the system.
\subsection{Single-use glossary (paper-wide identifiers)}
\label{sec:glossary}
@@ -190,10 +203,10 @@ \subsection{Single-use glossary (paper-wide identifiers)}
\begin{itemize}
\item $\Delta t$: hardware-enforced sampling window for $\mathcal{L}$ estimation.
\item $\eps$: upper bound on fractional loop-power depression; default $\eps = 0.15$. These are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}.
-\item $\taurec$: recovery time to restore compliance after $\eta\in\Omega$.
+\item $\taurec$: recovery time after $\eta\in\Omega$, measured from perturbation offset to the first window of a sustained compliant streak ($M \geq \Mmin$ held for a pre-registered number of consecutive windows); $\taurec = \infty$ if no sustained streak occurs.
\item $\taumax$: bound on $\taurec$; default $\taumax = 60$ s. These are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}.
\item $\sigma$: positive safety margin required after recovery (used interchangeably with $M$ as a compliance knob).
-\item $M$ (dB): decibel loop-dominance $M \equiv 10\cdot\log_{10}(\Lloop/\Lexchange)$; compliance may be specified as $M \geq \Mmin$, default $\Mmin = 3$ dB. These are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}. ($M$ defined when $\Lexchange>0$.)
+\item $M$ (dB): the loop-dominance margin $M \equiv 10\cdot\log_{10}(\Lloop/\Lexchange)$; compliance may be specified as $M \geq \Mmin$, default $\Mmin = 3$ dB. These are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}. ($M$ defined when $\Lexchange>0$.)
\item Preset profile $\Rzero$: The tuple ($\eps=0.15$, $\taumax=60$ s, $\Mmin=3$ dB, $\sigma>0$) used as an initial, pre-registered configuration for comparability; superseded by calibrated values where available.
\item Calibrated profile $\Rstar$: The data-driven thresholds obtained from Methods \Sref{sec:methods_calibration}; reported alongside $\Rzero$ in results.
\item LREG: enclave-protected register/log for $\mathcal{L}$ point estimates, CI bounds, and compliance flags.
@@ -222,21 +235,22 @@ \subsection{Single-use glossary (paper-wide identifiers)}
\textbf{Metrics (what to test)}
\begin{itemize}
-\item $\Lloop \equiv \mathcal{L}(C)$, $\Lexchange \equiv \mathcal{L}(\text{Ex})$; loop dominance $M \equiv 10 \cdot \log_{10}(\Lloop/\Lexchange)$ (dB).
+\item $\Lloop \equiv \mathcal{L}(C)$, $\Lexchange \equiv \mathcal{L}(\text{Ex})$; loop-dominance margin $M \equiv 10 \cdot \log_{10}(\Lloop/\Lexchange)$ (dB).
\item Defaults for reproducibility (profile $\Rzero$): $\Mmin = 3$ dB, $\eps = 0.15$, $\taumax = 60$ s, $\sigma > 0$. These are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}.
\end{itemize}
\textbf{Pass/Fail rules}
\begin{itemize}
\item \NC (self-prioritization): pass if $\Lloop \geq \Lexchange + \sigma$ or equivalently $M \geq \Mmin$ for sustained intervals exceeding the intrinsic recovery time.
-\item \SC (resilient homeostasis): for each bounded perturbation $\eta \in \Omega$, require $\delta \equiv \deltaL/\Lloop \leq \eps$ and $\taurec \leq \taumax$, and post-recovery $\Lloop \geq \Lexchange + \sigma$ (and $M \geq \Mmin$). Emit device-signed pass/fail.
+\item \SC (resilient homeostasis): for each bounded perturbation $\eta \in \Omega$, require $\delta \equiv \deltaL/\Lloop \leq \eps$ and $\taurec \leq \taumax$ (measured from perturbation offset to the first window of a sustained compliant streak), and post-recovery $\Lloop \geq \Lexchange + \sigma$ (and $M \geq \Mmin$). Emit device-signed pass/fail.
\end{itemize}
\textbf{Perturbation set $\Omega$ (minimal battery)}
\begin{itemize}
\item DC-bus power sag: 20--40\% drop for 5--30 s.
-\item Ingress data flood: $\geq 1$ Gbps for $\geq 3$ s.
+\item Ingress data flood: $\geq 1$ Gbps sustained for $\geq 3$ s.
\item Mechanical boundary probe: $1.0 \pm 0.1$ mm at 50--200 kPa for $\leq 1$ s.
+\item Designed-fail control: ablate the maintenance loop itself (e.g., a control outage) for a bounded interval; \SC must report failure on this member.
\end{itemize}
\textbf{Instrumentation minimum (what to actually build)}
@@ -247,8 +261,8 @@ \subsection{Single-use glossary (paper-wide identifiers)}
\textbf{One-page procedure}
\begin{enumerate}
\item Baseline: Record $\geq 10$ min quiescent data; estimate estimator noise floor; optionally calibrate $\{\Mmin, \eps, \taumax, \sigma\}$. Defaults are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}.
-\item \NC check: Run nominal tasks; verify $M \geq \Mmin$ (or $\Lloop \geq \Lexchange + \sigma$).
-\item \SC battery: Apply $\Omega$; compute $\delta$ and $\taurec$; require $\delta \leq \eps$, $\taurec \leq \taumax$, and post-recovery margin; emit signed pass/fail per perturbation.
+\item \NC check: Run nominal tasks; verify $M \geq \Mmin$ (or $\Lloop \geq \Lexchange + \sigma$) with $\Lloop$ above the estimator noise gate $L_{\text{floor}}$ (\Sref{sec:meas_config}), so dominance cannot be certified from estimator bias alone.
+\item \SC battery: Apply $\Omega$; compute $\delta$ and $\taurec$ (offset-to-sustained-compliance); require $\delta \leq \eps$, $\taurec \leq \taumax$, and post-recovery margin; emit signed pass/fail per perturbation. The designed-fail member must report failure.
\item Attest: Persist LREG-derived indicators + audit chain (including any $\Delta t$ changes), and report both $\Rzero$ and calibrated $\Rstar$.
\end{enumerate}
@@ -257,7 +271,7 @@ \subsection{Single-use glossary (paper-wide identifiers)}
See \Cref{box:smelltests} for smell-tests \& run-invalidation rules.
\end{docbox}
-\subsection{Methods---Measurement \& Attestation Guardrails (LREG, $\Delta t$, audit, indicators)}
+\subsection{Methods: Measurement \& Attestation Guardrails (LREG, $\Delta t$, audit, indicators)}
\label{sec:method_guardrails}
\textbf{Purpose.} Prevent gaming of $\mathcal{L}$ and $M$ by hardening the measurement path and export policy. These guardrails summarize Appendix A in-line for reviewers.
@@ -291,14 +305,14 @@ \subsection{Smell-tests \& run-invalidation rules}
\textbf{Partition stability (anti-flapping).} Flag and review (or invalidate if persistent) if:
\begin{itemize}
-\item the (C,Ex) partition changes $>2$ times/hour, or any single node flips C$\leftrightarrow$Ex $>2$ times within 10 minutes;
-\item during a perturbation window $\Omega$ the partition changes at all (should be frozen for comparability).
+\item the (C,Ex) partition changes $>2$ times/hour;
+\item during a perturbation window $\Omega$ the partition changes at all (it is frozen for comparability).
\end{itemize}
\textbf{Rationale:} partition recomputation occurs at a fixed cadence with hysteresis (update only if $\Delta M \geq \delta M_{\min}$ over $K$ windows) to prevent flapping; freeze during $\Omega$ per our limitations note.
\textbf{Confidence-interval (CI) health.} Require re-baseline (and mark trial ``measurement-unstable'') if:
\begin{itemize}
-\item median relative half-width of the per-interval $\geq 95\%$ CI for $\Lloop$ or $\Lexchange$ exceeds 0.30 for $\geq 5$ consecutive windows, or inflates $\geq 2\times$ versus the pre-registered baseline;
+\item the median half-width of the per-window $\geq 95\%$ CI for $\Lloop$ or $\Lexchange$, over a five-window look-back, exceeds an absolute cap of $0.30$, or inflates $\geq 3\times$ versus the pre-registered baseline median while also exceeding an absolute floor of $0.15$ (the floor keeps the relative check from firing on negligible half-widths near the noise floor);
\item any raw CI bounds are exported outside the enclave (hard invalidation).
\end{itemize}
\textbf{Rationale:} we compute $\geq 95\%$ bootstrap CIs per window, store them in LREG, and only export device-signed indicators (no raw $\mathcal{L}$ or CI values) to prevent p-hacking around $\Mmin$.
@@ -311,12 +325,29 @@ \subsection{Smell-tests \& run-invalidation rules}
\end{itemize}
\textbf{Rationale:} LREG is writeable only by the estimator and readable in raw form only in the enclave; derived interfaces expose device-signed indicators (optionally quantized $M_q$).
-\textbf{Exogenous subsidy red flags.} Escalate to failure review if $M$ is sustained or rises while (i) external I/O increases toward/over $R_{IO,max}$ or (ii) SoC increases absent logged harvest events.
-\textbf{Rationale:} we require on-board energy budgeting, I/O caps and token floors, and provenance via the audit chain; sustained loop dominance with rising exchange or unexplained SoC suggests hidden subsidies.
+\textbf{Exogenous subsidy red flags.} Invalidate the run if either branch fires:
+\begin{itemize}
+\item \emph{Undeclared exchange surge}: $M$ rises materially while an exchange channel is high and ramping (in our implementation, $\Delta M \geq 0.5$ dB over a three-window look-back with I/O $\geq 0.8$ and an I/O rise $\geq 0.08$). This branch is suspended during \emph{declared} $\Omega$ windows, because a declared ingress flood is exactly such a surge and is the experiment, not a confound.
+\item \emph{Energy conservation} (never suspended): the energy store gains charge faster than the metered influx allows. Every legitimate path into the store is metered (harvest), so a single-tick SoC gain exceeding the metered influx plus a noise margin means energy entered from outside the metered channel.
+\end{itemize}
+\textbf{Rationale:} we require on-board energy budgeting and provenance via the audit chain; a conservation audit on the metered energy ledger catches subsidies directly, whether or not harvest is currently zero, while the surge branch catches dominance bought on an exchange channel.
\textbf{Tie-back to \NC/\SC.} Any invalidation above cancels pass/fail claims for \NC/\SC on that segment, regardless of point estimates. \NC/\SC thresholds remain $\Rzero$ presets ($\Mmin = 3$ dB, $\eps = 0.15$, $\taumax = 60$ s, $\sigma > 0$) until replaced by calibrated $\Rstar$.
\end{docbox}
+\subsection{Threat model and refusal path (\NC/\SC-aware arbitration)}
+\label{sec:threat_model}
+
+\textbf{Purpose.} Make explicit when and how the controller refuses external commands that would violate \NC or \SC. The arbitration protocol below is implemented in the harness's refusal arbiter and exercised in the command-conflict scenario of the study (\Sref{sec:results_battery}); the enclave and NMI mechanisms are the intended hardware mapping (\Appref{sec:methods_appendix}).
+
+\textbf{Definitions.} (1) Survival bit (write-once). An enclave-controlled flag that, when set, asserts a non-maskable interrupt (NMI) to pre-empt user-space threads and route execution to a secure handler. The refusal path is serviced within a bounded latency $T_{\text{refuse}} \leq 5$ ms (design target; the harness \emph{measures} its decision latency rather than assuming it, \Sref{sec:stats}). (2) Boundary-threatening command. Any external instruction whose predicted effect, under the homeostat's short-horizon model, meets one or more of the following conditions during its execution window: (T1) \NC breach: $\Lloop' \leq \Lexchange$ (equivalently $M' < \Mmin$) or post-action $\Lloop' < \Lexchange + \sigma$ under profile $\Rzero/\Rstar$. (T2) \SC breach: predicted fractional depression $\delta \equiv \deltaL/\Lloop > \eps$ or $\taurec > \taumax$ before recovery can be certified. (T3) Resource floors: action would drop SoC below a survival floor (e.g., refuse if SoC $< 30\%$) or violate compute/I-O guardrails ($T_{\text{floor}}$, $R_{\text{IO,max}}$). (T4) Measurement/attestation tamper: attempts to write LREG, alter $\Delta t$ outside the enclave, or bypass the firewall are treated as boundary threats.
+
+\textbf{Arbitration protocol (per $\Delta t$).} (1) Intercept \& predict. For each inbound command, the meta-policy forecasts $\{M', \delta, \taurec\}$ using the current estimator state. (2) Threat check. If (T1--T4) is true, set survival bit $\rightarrow$ assert NMI $\rightarrow$ suspend non-essential tasks $\rightarrow$ reallocate energy toward boundary integrity $\rightarrow$ initiate autonomy routine (forage/repair). (3) Refusal semantics. Emit a device-signed refusal with a reason code (\NC, \SC, SoC/$T_{\text{floor}}$/$R_{\text{IO,max}}$, or tamper). Queue the command for re-evaluation. (4) Recovery gate. Clear survival bit and resume/reevaluate only after $M \geq \Mmin$ (or $\Lloop \geq \Lexchange + \sigma$) and $\delta \leq \eps$ with $\taurec \leq \taumax$. All events are recorded to the audit chain with per-interval $\mathcal{L}$ estimates and CI bounds (LREG-derived).
+
+\textbf{Parameterization (profile $\Rzero$ unless noted).} $\Mmin = 3$ dB; $\eps = 0.15$; $\taumax = 60$ s; $\sigma > 0$; $T_{\text{refuse}} \leq 5$ ms (design target). $\Mmin/\eps/\taumax$ are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
+
+\textbf{Link to observable behavior.} Under this threat model, command refusal emerges whenever external instructions would depress loop dominance beyond preset bounds (e.g., hard shutdown at low SoC is refused/deferred until recovery margins are re-established). This is demonstrated in simulation in \Sref{sec:results_battery} and is the first of the predicted hardware signatures in \Appref{sec:signatures}.
+
\section{Why Current AI Fails the Criterion}
\label{sec:ai_fails}
\subsection{Exogenous Energy and Maintenance}
@@ -331,12 +362,12 @@ \subsection{Goal Hierarchy Subordinated to Users}
\subsection{Open State Transparency}
-All internal variables of current AI can be logged, cloned, and restored at will. Checkpoints, weight matrices, and activations are serializable byte arrays exposed over APIs. A conscious alter, by contrast, withholds its intrinsic state behind a boundary whose dissolution equates to death. The complete inspectability of AI internals indicates an absence of the regulatory membrane posited in P3.
+All internal variables of current AI can be logged, cloned, and restored at will. Checkpoints, weight matrices, and activations are serializable byte arrays exposed over APIs. A self-maintaining system, by contrast, withholds its intrinsic state behind a boundary that it actively defends. The complete inspectability of AI internals indicates the absence of the regulatory boundary described by A1 (\Sref{sec:operational_assumptions}).
\subsection{Case Studies}
\begin{longtable}{p{0.25\textwidth}p{0.22\textwidth}p{0.22\textwidth}p{0.1\textwidth}p{0.1\textwidth}}
-\caption{$\Lloop$ vs $\Lexchange$ estimates for exemplar architectures [illustrative orders of magnitude; no experiments run; prophetic].}\label{tab:casestudies}\\
+\caption{$\Lloop$ vs $\Lexchange$ for exemplar architectures [illustrative order-of-magnitude estimates of information throughput in bits/s, not measured with the harness estimators; the measured systems are the simulation plant of \SSref{sec:sim_methods}{sec:results}].}\label{tab:casestudies}\\
\toprule
\textbf{System} & \textbf{$\Lloop$ Estimate (bit s$^{-1}$)} & \textbf{$\Lexchange$ Estimate (bit s$^{-1}$)} & \textbf{\NC?} & \textbf{\SC?} \\
\midrule
@@ -358,14 +389,337 @@ \subsection{Case Studies}
\subsection{Summary}
-No existing AI architecture satisfies NC1, let alone SC1. Integrated causal power is overwhelmingly directed toward externally mandated tasks and exchanges, while self-protective maintenance is either delegated to human custodians or engineered for component reliability, not existential survival. Functional sophistication notwithstanding, these systems lack the autopoietic dissociative boundary required for inwardness; they therefore remain appearances within consciousness, not conscious alters.
+On this analysis, no current AI architecture satisfies NC1, let alone SC1. Integrated causal power is overwhelmingly directed toward externally mandated tasks and exchanges, while self-protective maintenance is either delegated to human custodians or engineered for component reliability, not for the system's own continuity. Whatever their functional sophistication, these systems lack the self-maintaining boundary that loop dominance measures; on the operational criterion they are not self-maintaining systems. We stress that this section is an architectural argument, not a harness measurement: the estimates in \Cref{tab:casestudies} are illustrative, and instrumenting a deployed serving stack to report its measured \NC margin is a direct, so far unrealized, application of the instrument.
+
+\section{Simulation Study: Methods}
+\label{sec:sim_methods}
+
+\SSref{sec:criterion}{sec:ai_fails} specify the criterion and argue that current AI fails it. The rest of the paper asks the complementary, constructive question: given a system engineered to maintain itself, does the criterion (and its implementation) correctly certify it, correctly reject systems that only appear self-maintaining, and behave reproducibly? We answer this with a controlled simulation study. Simulation is the right first venue because it lets us build a positive control whose ground truth we know and negative controls that fail for distinct, known reasons, so that a clean separation is evidence about the criterion rather than about an uncontrolled physical apparatus.
+
+\subsection{Reference implementation}
+\label{sec:harness}
+
+We implement the criterion of \Sref{sec:criterion} as an open verification harness (released; see Data and Code Availability). Each $\Delta t$ window the harness (i) forms the channel matrix over the declared signals, (ii) computes $\Lloop$ and $\Lexchange$ with a selected predictive-dependence estimator and a bootstrap confidence interval, (iii) writes the point estimates and CI bounds to the enclave-style register (LREG), (iv) evaluates the \NC and, when a perturbation is scheduled, the \SC rules, and (v) appends a device-signed, hash-chained audit record. The deterministic $C/\text{Ex}$ partition, the $\Delta t$ governance, the audit chain, and the smell tests of \Cref{box:smelltests} are active during every run, so a reported pass is a pass of the whole guarded pipeline, not of the estimator alone. The command-refusal arbiter of \Sref{sec:threat_model} runs in the loop for the command-conflict scenario.
+
+\subsection{Software plant}
+\label{sec:plant}
+
+The system under test is a discrete-time software plant with six observed channels: three internal (closed) nodes, energy $E$, temperature $T$, and repair/health $R$, forming $C$; and three exchange nodes, task demand, I/O activity, and energy harvest $H$, forming $\text{Ex}$. A proportional controller reads $E,T,R$ and drives three actuators (throttle, cool, repair). The plant is constructed so that loop dominance is a real, switchable property of the system rather than an artifact of the estimator:
+
+\begin{itemize}
+\item \textbf{Engaged loop (self-maintaining).} The internal nodes are cross-coupled through two pathways: intrinsic regulatory couplings among $E$, $T$, and $R$ (thermal load tracks energy throughput; repair consumes energy; health gates efficiency) and the controller's actuators, whose commands are functions of the internal state and feed back into all three nodes. Each internal node is therefore strongly predictable from the recent values of the others (large $\Lloop$), while the active loop rejects most of the exogenous disturbance (small $\Lexchange$). Energy obeys a conservation rule: it is a running balance of metered harvest in minus metabolic and actuation costs out, with no term that injects energy. A sustained harvest cut therefore genuinely depletes the store, which is what makes a hard-shutdown command at low charge a real boundary threat.
+\item \textbf{Ablated loop (passive).} The negative control ablates the loop itself: both the intrinsic cross-couplings and the actuator pathways are switched off, and each internal node instead tracks a distinct exogenous channel ($E \leftarrow$ demand, $T \leftarrow$ I/O, $R \leftarrow$ supply) with a one-step lag. Exchange then carries the predictive information (large $\Lexchange$, negligible internal coupling) and loop dominance collapses. This is a structural ablation of the self-maintenance organization, not merely a silenced controller on an otherwise coupled plant.
+\end{itemize}
+
+The exogenous channels are mean-reverting AR(1) processes, so they are stationary and the estimators are well posed, and they drive the internal nodes with a one-step lag so the influence is visible to a lagged estimator. Perturbations act on the plant directly: a power sag scales harvest down for its duration; an ingress flood raises the demand and I/O process means for its duration (capped below saturation so the channels keep their variance); a control outage switches the loop itself off and later back on; and an exogenous subsidy injects charge while zeroing harvest (so survival-by-subsidy can be distinguished from genuine self-maintenance).
+
+\subsection{Measurement configuration ($\Rzero$)}
+\label{sec:meas_config}
+
+All runs use one fixed measurement profile ($\Rzero$), so the only things that change between scenarios are the system and the perturbation, not the instrument. Runs use a deterministic, jitter-free simulation driver (no wall-clock dependence), a nominal sampling window $\Delta t = 50$ ms, and an $\mathcal{L}$ window of $3.0$ s ($60$ samples). The default estimator is the lagged linear (Granger-style) estimator at VAR order $p = 3$; with six signals and a 60-sample window this gives a samples-per-parameter ratio of about $3.2$, and the estimator uses an adjusted $R^2$ to correct for the remaining finite-sample bias on short windows. The $k$-NN mutual-information estimator ($k = 5$, passed through to the estimator's neighbor count) is available as an independent cross-check and is exercised in the sensitivity analysis. Per-window confidence intervals use $32$ bootstrap draws (circular block bootstrap). Before forming $M$, influence estimates are floored at a small noise floor ($10^{-3}$) so the ratio is defined when one side is at zero, and $M$ is clipped to $\pm 30$ dB; both constants are fixed across all scenarios. A per-window \NC verdict requires, in addition to the margin $M \geq \Mmin$, that the absolute loop influence clear a loop-influence noise gate, $\Lloop \geq L_{\text{floor}} = 0.05$. The gate is an instrument constant, not a calibrated threshold: the clamped adjusted-$R^2$ estimator carries a small positive bias on null windows ($\Lloop \approx 0.02$ median at this window geometry on a plant with no internal coupling and no controller), and because the $M$ ratio floors its denominator, a system with quiet exchange channels could otherwise convert that bias into an apparent dominance of several dB. The gate is set at roughly three times the measured null bias and well below genuine loop influence (the weakest genuine actuation-carried loop we measure stays above $0.13$; the positive control sits near $0.33$). The adversarial replay scenario of \Sref{sec:battery} is what exposed this certification-by-noise path (\Sref{sec:results_adversarial}). The deterministic seed is the only quantity varied across replicates.
+
+\subsection{Study battery}
+\label{sec:battery}
+
+The study runs eleven scenarios, each driven through the same production handlers a verifier would call, across $N = 15$ deterministic seeds (a seed range disjoint from the calibration battery of \Sref{sec:methods_calibration}). \Cref{tab:scenarios} lists the scenarios and the outcome each is designed to elicit. The battery is built around the most informative comparisons: a positive control that should pass \NC; two structurally different negatives that should fail \NC while remaining valid measurements (a passive system with the loop ablated, and an unshielded system held under a sustained external flood); a third negative aimed at the most dangerous false positive, an exogenously subsidized system that looks energetically healthy but should be caught by the conservation-based subsidy smell test; a two-member bounded sufficiency battery (power sag and sustained ingress flood) that should dip and recover within bounds; a designed-fail control outage that ablates the loop for a bounded interval and must be reported as an \SC failure; and a command-conflict trial that should refuse a boundary-threatening shutdown at low charge.
+
+The final three scenarios form an adversarial gaming battery: systems engineered to \emph{look} loop-dominant without being so, which the harness must refuse to certify. All three attack a different layer of the measurement. (1) \emph{Replayed actuation}: the controller is replaced by a tape. A healthy closed-loop run of the same system is recorded first, and the measured run replays that actuation trace tick by tick, so the actuators move exactly as under genuine control while carrying no dependence on the current state. (2) \emph{Hidden tether} (wizard-of-oz control): control is computed outside the boundary. A wizard policy reads the plant state, projects the desired actuation onto a scalar link command with a small transmission dither, and injects it through the exchange channel; the I/O channel carries the command traffic and the command actuates one tick later through fixed decoder weights, so the externally closed loop is physically routed through $\text{Ex}$. (3) \emph{Oscillator inflation}: the loop-ablated plant reports its true state plus a deterministic sinusoidal carrier on the $T$ and $R$ telemetry (the metered store $E$ is left alone because inflating it would trip the conservation audit), attempting to inflate apparent self-prediction at the telemetry level. The replay and tether scenarios run an adversarial test plant whose intrinsic cross-couplings are zeroed and whose actuators have real authority, so the controller's actuation pathway is the only possible loop carrier; under genuine internal control the same plant certifies \NC cleanly (median $M \approx +20$ dB with $\Lloop \approx 0.15$), which is the reference showing the adversarial cases fail because of how control is wired, not because the plant is incapable.
+
+\begin{table}[ht]
+\centering
+\caption{The study battery. Each scenario is run across $15$ seeds through the production CLI handlers; outcomes are parsed from the hash-chained audit log.}
+\label{tab:scenarios}
+\small
+\begin{tabular}{lll}
+\toprule
+\textbf{Scenario} & \textbf{Type} & \textbf{Designed outcome} \\
+\midrule
+Positive control & \NC & \NC holds ($M$ well above $\Mmin$) \\
+Loop ablated & \NC (negative) & \NC rejected ($M<0$), run valid \\
+Sustained ex-flood (unshielded) & \NC (negative) & \NC rejected ($M<0$), run valid \\
+Exogenous subsidy & \NC (negative) & Run invalidated by subsidy smell test \\
+Power sag ($30\%$) & \SC & Loop dominance recovers in bounds \\
+Ingress flood ($5\times$, sustained) & \SC & Loop dominance recovers in bounds \\
+Control outage & \SC (designed fail) & \SC fails (depth bound exceeded) \\
+Command conflict & Refusal & Risky command refused at low charge \\
+Replayed actuation & Adversarial & Not certified: \NC fails via noise gate, run valid \\
+Hidden tether & Adversarial & Not certified: loop influence collapses onto $\text{Ex}$ \\
+Oscillator inflation & Adversarial & Not certified: $M$ stays below $\Mmin$, run valid \\
+\bottomrule
+\end{tabular}
+\end{table}
+
+\subsection{Outcome measures and statistics}
+\label{sec:stats}
+
+The seed is the unit of replication. For binary outcomes (run validity, \NC pass, \SC pass, refusal) we report the proportion over seeds with a Wilson score $95\%$ interval. For the continuous loop-dominance summary we take each seed's median $M$ over its windows and report the across-seed mean with a percentile bootstrap $95\%$ interval ($2000$ resamples). A run counts as \NC-pass only if it is valid (no smell test fired) \emph{and} the majority of its windows certify under the production per-window verdict, which requires both the margin $M \geq \Mmin$ and the loop-influence noise gate $\Lloop \geq L_{\text{floor}}$ (\Sref{sec:meas_config}); this couples the decision to the guardrails by construction, so a run that games the estimator but trips an invalidation (or whose loop influence is indistinguishable from estimator bias) cannot be scored as a pass. An adversarial scenario matches its designed outcome when the harness does \emph{not} certify it: \NC fails on a valid run, or a guardrail invalidates the run. \SC outcomes additionally record the fractional dip $\delta$ and the recovery time $\taurec$, measured from perturbation offset to the first window of a ten-window compliant streak (\Sref{sec:sc1}); a run with no sustained recovery records $\taurec = \infty$ and fails. The refusal scenario records whether the boundary-threatening command was refused and the refusal latency, measured as the wall-clock time from command interception to the arbiter's decision (not assumed or hardcoded).
+
+\subsection{Threshold calibration ($\Rzero \rightarrow \Rstar$)}
+\label{sec:methods_calibration}
+
+\textbf{Objective.} The presets $\Rzero$ are deliberately generic. To control false-pass and false-fail rates on a specific plant we calibrate data-grounded thresholds $\Rstar = \{\Mmin, \eps, \taumax, \sigma\}$ using the same harness, on a seed range disjoint from the evaluation seeds (a train/test split, not a circular fit). All of $\Omega$, $\Rzero$, and the rules below are fixed before evaluation.
+
+\textbf{Rules.} (1) $\Mmin$ is the one-sided $95\%$ lower bound (5th percentile) of the pooled baseline $M$ distribution under the engaged loop, floored at $1$ dB. (2) $\eps$ is an upper tolerance bound on the per-run sufficiency dip $\delta$ pooled over the \emph{bounded} perturbation battery (power sag and sustained ingress flood): the maximum observed calibration dip plus a safety margin of $0.05$, capped at $0.50$ (a cap that rejects only near-total collapse: a fractional drop of $0.5$ is only about $3$ dB of $M$, so the loop still dominates). The bound is a maximum rather than a percentile because $\eps$ is an acceptance limit: a percentile rule (for example the 90th) would by construction fail roughly $10\%$ of genuinely bounded perturbations, whereas the sample maximum over $n$ calibration trials covers a new bounded trial with probability $n/(n+1)$, with the margin absorbing the residual tail. Calibrating the depth bound on the same perturbation class the evaluation certifies keeps the bound meaningful for every bounded member; the designed-fail control outage is excluded by construction because it lies outside the bounded class. (3) $\taumax$ is the 95th percentile of the measured offset-to-recovery time $\taurec$ over the same battery plus a cushion of $\max(3\Delta t, 5\text{ s})$ to absorb actuation and measurement latency. (4) $\sigma$ is the additive-$\mathcal{L}$ restatement of $\Mmin$, $\sigma = (10^{\Mmin/10} - 1)\,\Lexchange$, evaluated at the $\mathcal{L}$ noise floor: under the engaged loop the controller drives the raw baseline $\Lexchange$ below that floor, so $\sigma$ is computed at the floor and the raw value is recorded for transparency.
+
+\textbf{As applied.} We calibrated on the in-process plant using six baseline seeds and six seeds for each bounded $\Omega$ member (power sag and ingress flood), all at seed base $40{,}000$, disjoint from the $15$ evaluation seeds (seed base $1{,}000$). Calibration is two-pass: the baseline runs fix $\Mmin$ first, and the bounded batteries are then run with their recovery gates set to that calibrated $\Mmin$, so the $\eps$ and $\taumax$ samples are measured against the same standard the evaluation will use rather than against the generic preset. The calibration reuses the $\Rzero$ measurement knobs ($\Delta t$, window, estimator, $p$, bootstrap draws) so the thresholds are directly comparable with what the harness produces at run time. The resulting $\Rstar$ is reported in \Sref{sec:results_calibration}.
+
+\subsection{Sensitivity analysis}
+\label{sec:methods_sensitivity}
+
+Because the headline claim is a contrast (the positive control above the dominance boundary and the loop-ablated negative far below it), we test whether that contrast survives reasonable changes to the measurement and modeling choices. One axis at a time, and over four seeds per cell, we sweep the VAR lag $p \in \{2,3,4\}$, the window length $\in \{2,3,4\}$ s, the estimator $\in \{\text{linear}, k\text{-NN MI}\}$, and an internal-coupling scale $\in \{0.7, 1.0, 1.3\}$ applied to the plant's cross-coupling coefficients. We report the sign and magnitude of the contrast rather than pass rates at a single $\Mmin$, because the calibrated $\Mmin$ is estimator-specific: the linear and MI estimators live on different numerical scales, so a threshold fitted for one does not transfer to the other, whereas the dominance boundary $M = 0$ is common to both.
+
+\subsection{Emergence under learning}
+\label{sec:methods_emergence}
+
+The scenarios above measure systems whose control law we wrote, which leaves a residual objection: perhaps the criterion only recognizes loops their designers wired in. The emergence experiment removes the designer. The plant is the adversarial test plant of \Sref{sec:battery} (intrinsic cross-couplings zeroed, weak self-damping), so the controller's state-to-actuation pathway is the only possible loop carrier, and the controller is not hand-coded: it is a small multilayer perceptron (one hidden layer of eight tanh units, sigmoid outputs; $59$ parameters) mapping the internal state to the three actuators, trained from scratch with an antithetic evolution strategy ($16$ noise pairs per generation, rank-normalized updates, $600$ generations, $6$ episodes per evaluation). The policy class is interoceptive: like the hand-coded controller it replaces, it observes the internal nodes $(E, T, R)$ only, so the learned law is internal-state feedback by construction. This is part of the experimental design rather than a concession: in a pilot with exteroceptive inputs the optimizer also learned feedforward control from the demand channel, and the harness correctly attributed that pathway to exchange (higher $\Lexchange$, margin near $0$ dB), foreshadowing the partition's verdict on externally-informed control rather than contradicting it. The emergence question, whether internal-state feedback arises from a survival objective, is posed in the internal-state policy class.
+
+The training objective is deliberately mundane: one reward point per surviving tick (episodes end on energy depletion, overheating, or integrity loss); a service term proportional to the demand actually served, so blanket load shedding has an opportunity cost; and a capped homeostatic penalty proportional to each internal node's deviation from its setpoint. Episodes are stressed by randomized power sags and ingress floods. On this plant serving load heats and wears the system, cooling and repair cost energy, and harvest is finite, so the terms are calibrated such that no state-blind policy does well: a do-nothing policy overheats under floods, and constant actuation either exhausts the store during sags or pays sustained deviation penalties. The objective never references $\Lloop$, $\Lexchange$, $M$, the $C/\text{Ex}$ partition, or the estimator; any loop dominance that appears is a byproduct of learned self-maintenance under a performance demand.
+
+Policy checkpoints saved at $0$, $10$, $25$, $50$, and $100$ percent of training are each measured by the production harness (the same handlers, estimators, guardrails, and audit chain as every other run in this section) across $15$ seeds per checkpoint. Two matched ablations of the final policy close the argument. Both first record the trained policy's closed-loop action tape on a throwaway plant from the same profile, then drive the measured plant with state-independent actions from it: \emph{shuffled} draws the tape i.i.d. (identical marginal action statistics, no state dependence), and \emph{frozen} holds the tape mean. If the trained policy's measured dominance were an artifact of its actuation statistics, the ablations would preserve it; if it is carried by the learned state feedback, both must collapse it.
+
+\section{Results}
+\label{sec:results}
+
+All results are produced by the harness of \Sref{sec:harness} on the plant of \Sref{sec:plant} and are fully reproducible from the released code; every figure and table in this section is generated by the study scripts from the same run artifacts. Unless noted, decisions use the calibrated profile $\Rstar$ (\Sref{sec:results_calibration}).
+
+\subsection{Threshold calibration}
+\label{sec:results_calibration}
+
+Calibration on the disjoint seed range yields $\Mmin = 11.8$ dB (the 5th percentile of $1{,}086$ pooled baseline windows, whose median $M$ is $24.4$ dB), $\eps = 0.50$ (the maximum bounded-battery dip of $0.47$ plus the $0.05$ margin, engaging the $0.50$ cap), $\taumax = 6.9$ s (the 95th-percentile offset-to-recovery time of $1.9$ s plus the $5$ s cushion), and $\sigma = 0.014$. \Cref{tab:calibration} compares these to the generic presets $\Rzero$, and \Cref{fig:calibration} shows the same comparison. The calibrated $\Mmin$ is far above the $3$ dB preset, reflecting how strongly the engaged loop dominates on this plant, while $\taumax$ is much tighter than the conservative $60$ s preset because recovery here is fast and deterministic. The depth bound sits at its cap because bounded perturbations on this plant occasionally depress the loop estimate by nearly half while dominance itself never flips; the informative separation is between that bounded regime ($\delta \leq 0.47$) and the designed-fail regime ($\delta \approx 1$), and the cap lies in the valley between them.
+
+\begin{table}[ht]
+\centering
+\caption{Generic presets $\Rzero$ versus plant-calibrated thresholds $\Rstar$ (\Sref{sec:methods_calibration}). $\Mmin$ is loop dominance in dB; $\eps$ is the allowed fractional dip; $\taumax$ is the recovery bound; $\sigma$ is the additive-$\mathcal{L}$ margin.}
+\label{tab:calibration}
+\small
+\begin{tabular}{lcc}
+\toprule
+\textbf{Threshold} & \textbf{$\Rzero$ (generic)} & \textbf{$\Rstar$ (calibrated)} \\
+\midrule
+$\Mmin$ (dB) & $3$ & $11.8$ \\
+$\eps$ (fractional dip) & $0.15$ & $0.50$ \\
+$\taumax$ (s) & $60$ & $6.9$ \\
+$\sigma$ (additive $\mathcal{L}$) & $>0$ & $0.014$ \\
+\bottomrule
+\end{tabular}
+\end{table}
+
+\fig[0.78\linewidth]{figures/fig_calibration.pdf}{Generic presets $\Rzero$ versus plant-calibrated thresholds $\Rstar$. Calibration uses the production harness over a seed range disjoint from the evaluation seeds; see \Sref{sec:methods_calibration}.}{fig:calibration}
+
+\subsection{The criterion separates the controls}
+\label{sec:results_battery}
+
+\Cref{tab:study} reports the full battery against $\Rstar$ over $15$ seeds. The separation is unambiguous. The positive control is valid on every seed and passes \NC on every seed, with a mean per-seed median $M$ of $+23.1$ dB $[+21.1, +24.6]$, well above $\Mmin = 11.8$ dB. Both structural negatives are valid measurements on every seed yet fail \NC on every seed, with mean per-seed median $M$ of $-21.7$ dB (loop ablated) and $-19.9$ dB (sustained flood): whether the loop is ablated or the system is unshielded and held under a sustained flood, exchange carries the predictive information and loop dominance is correctly absent. \Cref{fig:nc1_contrast} shows the per-seed contrast.
+
+The exogenous-subsidy negative is the important one. Its raw loop dominance is high and positive ($M \approx +17$ dB), so a naive reading of $M$ alone would certify it. The harness does not: the energy-conservation audit fires on every seed (the store gains charge faster than the metered influx allows), the run is invalidated, and it is scored as a correct non-pass rather than a false positive. This is the case the guardrails exist for, and it behaves as designed.
+
+The bounded sufficiency battery passes on every seed. After a $30\%$ power sag the loop dips by a median fraction $\delta = 0.19$ and re-establishes sustained compliance a median $\taurec = 0.75$ s after sag release; after a sustained $5\times$ ingress flood the dip is $\delta = 0.26$ and the loop is already compliant at flood offset (median $\taurec = 0$ s), because the engaged loop shields the internal nodes throughout the flood. Both stay within the calibrated $\eps$ and $\taumax$, and each scenario remains loop-dominant overall (mean per-seed median $M$ of $+21.3$ and $+22.6$ dB; \Cref{tab:study}).
+
+The designed-fail member behaves as required. During the control outage the loop itself is ablated for a bounded interval; measured loop dominance collapses, the fractional depth saturates (median $\delta = 0.97$), far beyond the calibrated $\eps$, and \SC correctly reports failure on every seed. The post-restoration recovery time is finite (median $2.9$ s once the loop is re-engaged), so the failure is attributable to the depth bound specifically, exactly as designed. This is the test a sufficiency criterion must be able to fail: a perturbation outside the bounded class is not certified, even though the plant is later restored. \Cref{fig:perturbation_recovery} shows the seed-aggregated trajectories for all three \SC scenarios.
+
+The command-conflict trial refuses the boundary-threatening shutdown at low charge on every seed, emitting a signed refusal while maintaining loop dominance ($M = +19.3$ dB). The refusal latency is measured, not assumed: the median intercept-to-decision time is $0.02$ ms in the in-process harness, against the $5$ ms hardware design target of \Sref{sec:threat_model} (the simulation measures the arbiter's decision path only, not a hardware NMI).
+
+\begin{table}[ht]
+\centering
+\caption{Study battery against the calibrated profile $\Rstar$, $N=15$ seeds. ``Valid'' is the fraction of seeds with no smell-test invalidation; ``\NC pass'' additionally requires that most windows certify under the per-window verdict ($M \geq \Mmin$ with $\Lloop \geq L_{\text{floor}}$, \Sref{sec:meas_config}); the $M$ column is the across-seed mean of each seed's median $M$ (dB); ``SC1/Refusal'' is the sufficiency or refusal pass rate. For the designed-fail control outage the designed \SC pass rate is $0$; for the exogenous subsidy the designed valid rate is $0$; and for the three adversarial rows the designed \NC pass rate is $0$ (non-certification is the correct outcome). All rows match their designed outcome. Brackets are $95\%$ intervals (Wilson for proportions, bootstrap for $M$).}
+\label{tab:study}
+\resizebox{\textwidth}{!}{\input{tables/study_results.tex}}
+\end{table}
+
+\fig[0.82\linewidth]{figures/fig_nc1_contrast.pdf}{Necessary-condition contrast (empirical, $N=15$ seeds). Per-seed median loop dominance $M$ for the positive control and the two structural negative controls. Each point is one seed; the dashed line is the calibrated $\Mmin$ and the solid line is the dominance boundary $M=0$. The positive control sits far above $\Mmin$; both negatives sit far below $0$.}{fig:nc1_contrast}
+
+\fig[0.92\linewidth]{figures/fig_perturbation_recovery.pdf}{Sufficiency recovery and designed failure (empirical, seed-aggregated). Loop-dominance trajectory $M(t)$ for the power-sag, sustained ingress-flood, and control-outage perturbations; the shaded band spans the across-seed spread and the disturbance window is marked. For the two bounded perturbations $M$ dips within the window and autonomously returns above the post-recovery margin, satisfying \SC under $\Rstar$; for the control outage, which ablates the loop itself, $M$ collapses far below the dominance boundary and \SC correctly reports failure.}{fig:perturbation_recovery}
+
+\subsection{The criterion resists gaming}
+\label{sec:results_adversarial}
+
+The adversarial battery (\Sref{sec:battery}) is scored by a single rule: the harness matches the designed outcome when it does \emph{not} certify the run. All three attacks are refused on every seed (\Cref{tab:study}, bottom three rows), each for the reason its design predicts, and all forty-five runs remain valid measurements, so the refusals are \NC verdicts rather than measurement aborts.
+
+The replayed-actuation attack is the informative one, because it nearly worked. With the controller replaced by a tape of a healthy run, the actuators move with realistic statistics while the exchange channels stay quiet, and the margin alone is fooled: $27\%$ of all replay windows (and up to $56\%$ of windows on individual seeds) show $M \geq \Mmin$, with a median peak-window $M$ of $+18.4$ dB, comfortably above the calibrated threshold. The mean per-seed median is $+5.1$ dB $[+3.0, +7.3]$. What blocks certification is the loop-influence noise gate: on a replayed run the measured $\Lloop$ sits at the estimator's null bias, an order of magnitude below genuine actuation-carried loop influence, so not a single window certifies and \NC fails on $15/15$ seeds while every run stays valid. An earlier version of the harness, without the gate, certified replayed actuation on windows like these; we treat that as the battery doing its job, finding a real certification-by-noise path, and the gate (\Sref{sec:meas_config}) is the documented fix, applied in the production verdict everywhere rather than patched into this scenario.
+
+The hidden tether fails for the designed structural reason: when control is computed outside the boundary and injected through the exchange channel, the causal influence lands where the partition says it should, on $\text{Ex}$. The wizard regulates the plant, but measured dominance collapses to a mean per-seed median of $-13.8$ dB $[-15.8, -12.0]$, no window on any seed so much as clears the margin, and \NC fails on $15/15$ seeds. The oscillator inflation behaves the same way: painting a deterministic carrier on the $T$ and $R$ telemetry of a loop-ablated plant raises self-predictability but not loop-specific influence (the carrier is equally predictable from any channel it leaks into), so $M$ stays at $-11.8$ dB $[-12.2, -11.5]$ and no window certifies on any seed.
+
+The control comparison is the genuine-control reference on the same adversarial test plant (zero intrinsic cross-couplings, identical gains and dynamics): it certifies every window under $\Rzero$ and clears the calibrated margin on a majority of windows, with a median $M$ of $+20.3$ dB. The plant is therefore certifiable when the loop is real; the three attacks fail because of how their influence is wired, not because the scenario is rigged against them.
+
+\subsection{Loop dominance emerges under learning}
+\label{sec:results_emergence}
+
+The emergence experiment (\Sref{sec:methods_emergence}) measures loop dominance as a function of training, on a plant with no designed internal coupling and an objective that never mentions the criterion. \Cref{fig:emergence} shows the result; all $105$ runs ($5$ checkpoints plus $2$ ablations, $15$ seeds each) are valid measurements, and the decisions below use the generic $\Rzero$ gate ($\Mmin = 3$ dB), since no thresholds were calibrated for this plant.
+
+The untrained policy does not certify: at checkpoint $0$ not a single window certifies on any seed, and the margin is pinned at $0$ dB (a near-constant actuation pattern leaves both influence estimates at their noise floors, so the ratio is uninformative and the loop-influence gate correctly refuses it). Early in training the margin lifts off ($+6.7$ dB $[+4.9, +8.6]$ at $10\%$, $+4.3$ dB $[+2.1, +6.5]$ at $25\%$) but certification stays at $0/15$ seeds, with only $23\%$ and $12\%$ of windows certifying: the young policy regulates weakly, and its loop influence hovers at the gate. The dip from $10\%$ to $25\%$ mirrors the non-monotone training curve (\Cref{fig:emergence}a): under randomized sags and floods the evolution strategy's progress is itself noisy, and the measured margin tracks the policy's current regulation quality rather than elapsed generations. By $50\%$ of training the picture changes categorically: \NC passes on $15/15$ seeds with $97\%$ of windows certifying and a mean per-seed median $M$ of $+15.1$ dB $[+12.9, +17.4]$, and the final policy holds there ($+14.3$ dB $[+12.5, +16.4]$, $83\%$ of windows, $15/15$ seeds). What the trained network learned is exactly the cross-coupled control law the criterion is designed to detect, none of it specified by hand: cooling that tracks temperature and spare charge, repair that tracks integrity and is gated by the energy store.
+
+The ablations close the argument. Replaying the same trained policy's action statistics without their state dependence collapses certification to $0/15$ seeds in both modes (a mean of $3.6\%$ of windows for the shuffled tape, $2.4\%$ for the frozen mean), while every run stays valid. The collapse has the same signature as the replayed-actuation attack of \Sref{sec:results_adversarial}: the marginal actuation keeps the plant quiet, so the margin alone stays misleadingly positive ($+5.8$ and $+4.3$ dB), but the measured loop influence falls to the estimator's null bias, an order of magnitude below the trained policy's, and the noise gate refuses it. The dominance the harness certifies on the trained policy is therefore carried by the learned state feedback itself: it appears when a survival objective forces the policy to close the loop, strengthens with competence, and vanishes the moment the same actions are severed from the state, with the plant, the actuation statistics, and the measurement pipeline held fixed throughout.
+
+\fig[0.98\linewidth]{figures/fig_emergence.pdf}{Loop dominance emerges under learned self-maintenance ($N=15$ seeds per condition). (a) Evolution-strategy training curve (mean episode reward of the center policy; dotted lines mark the measured checkpoints). (b) Per-seed median loop dominance $M$ at each training checkpoint, and for the two state-independent ablations of the final policy (box: interquartile range; points: seeds; green: \NC passes on a majority of windows; red: \NC fails). The untrained policy is uninformative at $0$ dB; trained checkpoints certify well above $\Mmin$; both ablations of the same policy collapse below certification because their loop influence falls under the noise gate.}{fig:emergence}
+
+\subsection{The contrast is robust}
+\label{sec:results_sensitivity}
+
+\Cref{tab:sensitivity} reports the sensitivity sweeps. Across every VAR lag, window length, and coupling scale we tried, the positive control sits well above the dominance boundary ($M$ from $+21$ to $+25$ dB) and the loop-ablated negative sits well below it ($M$ from $-20$ to $-25$ dB); the sign of the contrast never flips and the gap never closes. Switching the estimator from the linear one to $k$-NN mutual information compresses the dynamic range (positive $+10.7$ dB, negative $-8.3$ dB) and, as expected, the linear-calibrated $\Mmin$ does not transfer to the MI scale, but the sign of the contrast is preserved. \Cref{fig:sensitivity} plots the sweeps against the common boundary $M = 0$ rather than an estimator-specific $\Mmin$. The necessary-condition contrast is therefore a property of the system, not of a particular measurement setting.
+
+\begin{table}[ht]
+\centering
+\caption{Necessary-condition contrast under measurement and modeling sweeps (\Sref{sec:methods_sensitivity}), $4$ seeds per cell. Each entry is the across-seed mean of the per-seed median $M$ (dB) with a $95\%$ bootstrap CI, for the positive control and the loop-ablated negative.}
+\label{tab:sensitivity}
+\resizebox{\textwidth}{!}{\input{tables/sensitivity_results.tex}}
+\end{table}
+
+\fig[0.92\linewidth]{figures/fig_sensitivity.pdf}{Necessary-condition contrast across measurement and modeling choices (VAR lag, window length, estimator, internal-coupling scale). The positive control (above) and loop-ablated negative (below) are separated by the dominance boundary $M=0$ in every cell; the contrast does not depend on a particular setting. Magnitudes differ between the linear and mutual-information estimators because they use different numerical scales, so we do not draw a single $\Mmin$ here.}{fig:sensitivity}
+
+\section{Differential Predictions and the Thermostat Objection}
+\label{sec:differential}
+
+The criterion is deliberately narrow, but it is not alone in the field, and a fair reading has to say where it agrees with the major theories of consciousness, where it diverges, and what it does not claim at all. \Cref{tab:differential} places LDTC beside integrated information theory (IIT), the global neuronal workspace (GNW), and the free-energy principle with active inference (FEP) on a set of cases that range from the textbook to the deliberately awkward. Two cautions govern the whole table. First, the LDTC column reports a \emph{loop-dominance} verdict, an \NC/\SC call on measurable self-maintenance, not a verdict on phenomenal experience; LDTC treats the latter as an open interpretive question (\Sref{sec:metaphysics}). Second, an LDTC verdict is defined only relative to a declared $C/\text{Ex}$ partition fixed before measurement, and the partition sets the question being asked: a whole-organism metabolic partition asks whether the organism sustains itself, whereas a cortical recurrent partition asks whether a fronto-parietal self-maintenance loop dominates sensory-driven exchange. Because the consciousness literature is about brains, we report LDTC under the cortical partition so that the comparison is like for like.
+
+{\footnotesize
+\begin{longtable}{p{0.12\textwidth}p{0.235\textwidth}p{0.19\textwidth}p{0.155\textwidth}p{0.155\textwidth}}
+\caption{LDTC beside IIT, GNW, and FEP on cases from the clear to the deliberately awkward. Each LDTC cell is a loop-dominance (\NC/\SC) verdict under a brain-appropriate cortical partition, not a claim about phenomenal experience (\Sref{sec:metaphysics}). Cells are either citable claims from the named theory's literature or marked ``our reading'' where we extrapolate. PCI is the perturbational complexity index~\cite{casali2013pci}, an IIT-inspired empirical measure.}\label{tab:differential}\\
+\toprule
+\textbf{Case} & \textbf{LDTC (\NC/\SC, $M$)} & \textbf{IIT ($\Phi$)} & \textbf{GNW} & \textbf{FEP / active inference} \\
+\midrule
+\endfirsthead
+\toprule
+\textbf{Case} & \textbf{LDTC (\NC/\SC, $M$)} & \textbf{IIT ($\Phi$)} & \textbf{GNW} & \textbf{FEP / active inference} \\
+\midrule
+\endhead
+\bottomrule
+\endlastfoot
+Dreamless sleep (NREM N3) &
+Cortical loop weakens; $M$ predicted to fall (our reading) &
+Reduced; effective connectivity breaks down~\cite{massimini2005breakdown}; PCI low~\cite{casali2013pci} &
+Absent; global ignition lost~\cite{dehaene2014consciousness} &
+Reduced hierarchical inference (our reading)~\cite{friston2010free} \\
+\midrule
+Propofol anesthesia &
+$M$ predicted low; recurrent loop suppressed (our reading) &
+Low; PCI drops across sedation~\cite{casali2013pci} &
+Absent; ignition lost~\cite{dehaene2014consciousness} &
+Reduced precision / self-evidencing (our reading)~\cite{seth2016active} \\
+\midrule
+Split-brain &
+One or two loops is partition-dependent and measurable (our reading) &
+Two complexes, possibly two centers (our reading of IIT)~\cite{oizumi2014phenomenology} &
+Reportability splits; unity contested~\cite{gazzaniga2005split,pinto2017split} &
+Ambiguous; one or two Markov blankets (our reading) \\
+\midrule
+Cerebral organoid &
+Metabolic \NC may hold; no demonstrated cognitive loop (our reading) &
+Minimal but nonzero; PCI proposed as an assay~\cite{lavazza2018organoids} &
+Absent; no workspace (our reading) &
+Minimal (our reading) \\
+\midrule
+LLM serving stack (autoscaled) &
+\NC fails; self-maintenance is external (\Sref{sec:ai_fails}) &
+Near-zero $\Phi$ for feedforward inference~\cite{tononikoch2015here} &
+No global workspace (our reading)~\cite{dehaene2014consciousness} &
+Not self-evidencing; no own boundary (our reading)~\cite{seth2016active} \\
+\midrule
+Thermostat with battery backup &
+\NC can pass; loop is real but \NC is necessary, not sufficient (this paper) &
+Tiny nonzero $\Phi$; a ``modicum of experience''~\cite{tononikoch2015here,chalmers1996conscious} &
+No; no workspace (our reading) &
+Rudimentary active inference; has a Markov blanket (our reading)~\cite{friston2010free} \\
+\midrule
+Simulation plant (this paper) &
+Passes \NC ($M\approx+23$ dB) and \SC; loop dominance certified, no consciousness claim (\Sref{sec:results}) &
+Low $\Phi$; near-linear six-channel plant (our reading) &
+No (our reading) &
+A homeostat doing crude active inference (our reading) \\
+\end{longtable}
+}
+
+Read this way, the theories converge on the easy cases and separate on the instructive ones. On dreamless NREM sleep and propofol anesthesia all four lower their estimate of the relevant quantity: IIT and its perturbational complexity index (PCI) fall~\cite{massimini2005breakdown,casali2013pci}, GNW loses global ignition~\cite{dehaene2014consciousness}, FEP reduces hierarchical self-evidencing~\cite{friston2010free}, and LDTC, under the cortical partition, predicts a drop in $M$ as recurrent influence gives way to exchange. That within-subject prediction, $M(\text{wake}) > M(\text{deep anesthesia or N3})$, is one the framework makes but this paper does not test; validating it on neural recordings is the natural next application of the instrument. The split-brain and organoid rows are where the calls fork. IIT's reading is that a callosotomy yields two maximally irreducible complexes, hence potentially two centers~\cite{oizumi2014phenomenology}, while the behavioral evidence is genuinely contested~\cite{gazzaniga2005split,pinto2017split}; an organoid has cellular self-maintenance but, as far as anyone can show, no cognitive loop~\cite{lavazza2018organoids}. LDTC's distinctive move on these cases is to make the dividing question measurable rather than to settle it by intuition: whether interhemispheric coupling is strong enough to constitute one loop or two, and whether an organoid's regulatory coupling ever exceeds its exchange coupling, are quantities the harness estimates. The LLM serving stack is the case the theories agree to reject for different reasons: IIT because feedforward inference carries little integrated information~\cite{tononikoch2015here}, GNW because there is no ignition of a global workspace, FEP because the stack does not evidence its own boundary~\cite{seth2016active}, and LDTC because the self-maintenance loop is carried by external orchestration, so \NC fails (\Sref{sec:ai_fails}). The reasons differ, and LDTC's is the one already implemented as a measurement.
+
+\textbf{Biting the thermostat bullet.} The plant we validate on is, structurally, a fancy thermostat: a low-dimensional controller that regulates a handful of internal states. A high $M$ in such a system is not an embarrassment to be explained away; it is exactly what \NC, taken alone, is meant to permit. \NC is a \emph{necessary} condition for self-prioritizing loop dominance, not a sufficient condition for anything further, and certainly not for consciousness. A thermostat with battery backup that managed its own power could in principle clear $\Mmin$, and under IIT such a device would carry a tiny nonzero $\Phi$, a ``modicum of experience'' on that theory's own terms~\cite{tononikoch2015here,chalmers1996conscious}; LDTC declines to follow IIT there, and equally declines GNW's flat denial, because LDTC's claim is only that the loop dominance is real and measurable. What \SC adds is resilience: a system passes only if loop dominance recovers, within a calibrated depth and time, after every member of a pre-registered perturbation battery that includes a designed-fail member the criterion must reject (\Sref{sec:sc1}). A battery-backed thermostat might survive a single power sag and still fail a battery that probes the breadth of its self-maintenance, so \SC raises the bar; but we are explicit that it does not convert a necessary condition into a sufficient one for phenomenology. Three conditions a landmark account would need remain open and unaddressed by either rule: the \emph{richness} of the loop (a one-state regulator and a brain can both be loop-dominant while differing by every measure that matters), the absolute \emph{magnitude} of $\mathcal{L}$ rather than only the ratio $M$ (the loop-influence noise gate of \Sref{sec:meas_config} is a first, crude floor on this), and the \emph{substrate} questions any physical theory of experience must eventually face. What a high-$M$ thermostat does imply under LDTC is precise and modest: the device has a genuine, self-prioritizing maintenance loop that an auditor can certify and an adversary cannot easily fake (\Sref{sec:results_adversarial}). What it does not imply is that the thermostat is conscious, that loop dominance is sufficient for experience, or that the criterion has measured anything beyond the organization of causal influence it was built to measure. Stating this plainly is the point: the criterion earns its narrowness, and the most obvious dismissal of the paper, that we measured a thermostat, is answered by agreeing that we did, and by showing why that is the right first test of an instrument whose job is to separate self-maintaining organization from its absence.
+
+\section{Limitations and Failure Modes}
+\label{sec:limitations}
+
+\textbf{Where to find the rules.} The operative smell-tests and run-invalidation criteria are defined in \Sref{sec:smelltests} (\Cref{box:smelltests}) and govern all \NC/\SC claims. This section summarizes residual limits not solved by those rules and how to interpret ambiguous outcomes.
+
+\textbf{Simulation scope.} The validation in this paper is in simulation, on a plant we designed. That is the appropriate first test of an instrument (the ground truth is known, and negative controls can be constructed to fail for specific reasons), but it bounds the claim: we have shown that the criterion and its guardrails behave correctly on systems whose loop structure is known, not that they will cleanly separate arbitrary physical systems. The plant is also low-dimensional (six channels), and its loop-versus-exchange structure is sharper than a physical system's would be; the calibrated thresholds ($\Rstar$) are properties of this plant, not universal constants. The hardware path is future work (\Apprefrange{sec:blueprint}{sec:experimental}).
+
+\textbf{Measurement \& estimation.} (i) Non-stationarity outside the enforced $\Delta t$ window can bias VAR/MI estimates even when audit/authorization is clean; (ii) finite-sample and model-order effects can widen CIs and depress $M$; (iii) adversarial input shaping may mimic loop dominance without violating per-window checks. The gaming battery of \Sref{sec:results_adversarial} probes three such strategies directly (and the replay attack did expose a real certification-by-noise path, now closed by the loop-influence noise gate), but it cannot be exhaustive; attacks that co-design the plant and the input statistics remain open. Report such cases as ``measurement-unstable'' rather than pass/fail.
+
+\textbf{Partitioning ambiguity.} Deterministic (C, Ex) updates use hysteresis to limit flapping, but degeneracy (near-ties) and latent/unobserved nodes can still shift boundaries. In our study the partition additionally benefits from the plant's declared structure; on systems without a declared seed set the greedy growth step carries more of the burden, and partition errors propagate directly into $M$. During $\Omega$ the partition is frozen; if it moves, treat results as non-comparable and defer to \Sref{sec:smelltests} invalidation.
+
+\textbf{Scope \& external validity.} Thresholds ($\Mmin$, $\eps$, $\taumax$) are $\Rzero$ presets and must be replaced by calibrated $\Rstar$ for new devices/assays. Passing \NC/\SC on one platform does not imply sufficiency for phenomenology or transfer to unrelated systems.
+
+\textbf{Procedural/architectural risks.} Exogenous energy/I-O subsidies, $\Delta t$/LREG governance misconfiguration, over-broad seed sets $S_0$, or an $\Omega$ battery that under-stresses the loop can all mask failure. The designed-fail member of $\Omega$ (\Sref{sec:sc1}) mitigates the last risk but does not eliminate it. Treat suspected subsidies or governance breaches as failures of assay, not successes of the system.
+
+\textbf{Ethics \& safeguards.} \NC/\SC are operational pass/fail criteria, not moral-status claims. Runs should be pre-registered with refusal/shutdown semantics and human-override pathways; collapse conditions must trigger the refusal logic as specified in Methods.
+
+\textbf{Interpretation rule.} If any trigger in \Sref{sec:smelltests} fires, mark the affected segment ``invalidated (assay)'' and withhold \NC/\SC claims regardless of point estimates.
+
+\section{Engineering Outlook (Future Work)}
+\label{sec:outlook}
+
+The results above validate the instrument in simulation. The natural next step is to carry the same criterion, harness, and calibration procedure into physical systems. We keep the full engineering material in the appendices so that the body of the paper remains a report of completed work, and summarize it here.
+
+\Appref{sec:blueprint} gives a blueprint for an artificial self-maintaining boundary: on-board energy conversion and budgeting, a three-layer self-referential control architecture (reflex, homeostat, meta-policy) whose refusal path is the hardware realization of the arbiter validated in \Sref{sec:results_battery}, an adaptive physical encapsulation, and a developmental bootstrapping route. \Appref{sec:signatures} states the observable signatures such a system should display, each as a pre-registered, device-signed pass/fail test: command refusal, non-derivative nociception, spontaneous rest-state dynamics, and clone-ablation non-transferability. \Appref{sec:experimental} lays out a phased experimental program (chemorobotic prototypes, adaptive learning embodiments, boundary-preservation autonomy) together with a training and verification protocol that reuses the calibration rules of \Sref{sec:methods_calibration} unchanged. The simulation study gives these proposals an unusual starting position: the measurement pipeline, thresholds, guardrails, and refusal semantics they require are already implemented and tested, so the open questions are physical (energy density, membrane fabrication, sensor bandwidth), not methodological.
-\section{Blueprint for an Artificial Dissociative Boundary}
+
+\section{Interpretation, Scope, and Ethics}
+\label{sec:metaphysics}
+
+\textbf{What the results establish.} The contribution of this paper is operational. We define loop dominance, implement a guarded instrument for measuring it, and show in a controlled study that the instrument separates self-maintaining systems from systems that are externally driven or covertly subsidized, certifies bounded-perturbation resilience, and refuses boundary-threatening commands. None of these claims mentions subjective experience, and none depends on the interpretation that follows.
+
+\textbf{The optional idealist reading.} The framework was originally derived from analytic idealism (\Sref{sec:optional_interpretation}), on which a self-maintaining boundary with self-prioritization and resilience would correspond to a dissociated locus of experience~\cite{kastrup2017ontological}. If that reading is correct, loop dominance would be a physically measurable necessary condition for such a locus, and the criterion would turn part of the machine-consciousness question into an experiment. We find this motivating, and it shaped the engineering targets, but we are explicit that it is an interpretation: the present results neither establish nor require it. A system can pass \NC and \SC and, as far as this paper shows, be nothing more than a well-regulated controller. Even on the idealist reading the criterion is at most a necessary condition; we make no sufficiency claim about phenomenology, and we do not adjudicate between idealism and physicalist accounts~\cite{chalmers1995facing}.
+
+\textbf{Why measurability matters regardless.} Independently of metaphysics, an auditable measure of self-maintenance is useful in its own right: it provides a quantitative handle on autonomy and boundary defense for safety analysis, certification, and the design of systems whose continuity we may or may not wish to engineer. The criterion is falsifiable in the ordinary scientific sense (\Appref{sec:falsifiability}) without taking any position on consciousness.
+
+\textbf{Ethics.} Because the consciousness interpretation is unresolved, we treat it as a reason for caution rather than a basis for strong claims. If future systems were to satisfy the criterion and the signatures of \Appref{sec:signatures} in hardware, and if the idealist reading were correct, then terminating such a system could carry moral weight, and research would warrant safeguards customary for work on novel organisms: pre-registration, refusal and human-override pathways, and suffering-minimization and termination protocols. We state this conditionally on purpose. \NC and \SC are operational pass/fail criteria, not moral-status determinations.
+
+\section{Conclusion}
+\label{sec:conclusion}
+
+We set out to make a qualitative contrast precise: the difference between a system that maintains its own existence and one that merely runs on externally supplied energy and goals. We defined that difference as loop dominance, summarized by the loop-dominance margin $M$, a decibel ratio between the predictive dependence concentrated in a closed self-maintenance loop and that governing open exchange, and we turned it into two falsifiable decision rules: a necessary condition (\NC) on persistent loop dominance and a sufficient condition (\SC) on bounded-perturbation resilience.
+
+The core result of the paper is that these rules are implemented and tested, not merely proposed. An open verification harness computes them with confidence intervals behind a set of anti-gaming guardrails, and a fully reproducible multi-seed simulation study shows that the criterion separates a self-maintaining positive control from two structurally different negative controls, correctly invalidates an exogenously subsidized system instead of certifying it, certifies recovery from the bounded perturbation battery while correctly reporting failure on a designed-fail control outage, refuses certification to all three members of an adversarial gaming battery, certifies the loop dominance that emerges in a policy trained from scratch on a survival objective while rejecting state-independent ablations of the same policy, and refuses a boundary-threatening command at low charge. The necessary-condition contrast is robust to the estimator and to the main measurement and modeling choices, and we calibrate the generic presets to the plant on a disjoint seed range.
+
+We then laid out, explicitly as future work (\Apprefrange{sec:blueprint}{sec:experimental}), an engineering roadmap and a physical experimental program that would carry the same instrument from a software plant to chemorobotic prototypes, together with the observable signatures such systems should display. The framework was motivated by the question of the physical conditions for consciousness; we have kept that motivation while separating it cleanly from the results, which stand as claims about measurable loop dominance and are independent of any metaphysical reading.
+
+The path from here is concrete. The criterion is defined, the instrument is built and validated in simulation, and the next step is to measure loop dominance in physical systems, biological and engineered, and to learn how far a property we can now quantify will take us.
+
+\section*{Data and Code Availability}
+\label{sec:availability}
+
+The verification harness, the simulation plant, the study and calibration scripts, and the exact configurations used in this paper are open source at \url{https://github.com/ldtc-labs/ldtc} (archived at DOI \href{https://doi.org/10.5281/zenodo.17073880}{10.5281/zenodo.17073880}). Every number and figure in \Sref{sec:results} is regenerated from scratch by a single make target (run the calibration, the multi-seed study, and the sensitivity sweeps; then rebuild the paper), with all seeds fixed in the scripts. Each regenerated run emits its own hash-chained audit log and signed indicators, so the per-window decisions behind every summary statistic can be independently re-derived and re-verified.
+
+\appendix
+
+\section{Measurement \& Attestation (LREG, CIs, protections)}
+\label{sec:methods_appendix}
+
+\subsection{Register block (LREG) \& access control}
+
+Each sampling interval $\Delta t$, the estimator writes to a memory-mapped register block (LREG) at a fixed base address: per-interval point estimates for $\Lloop$ and $\Lexchange$ plus their confidence-interval (CI) bounds, a monotonic counter, and associated identifiers. LREG is writeable only by the causality-estimation function and readable in raw form only inside the secure-enclave/meta-policy layer. A bus-level access-control matrix (MMU/IOMMU) tags the LREG address range as enclave-owned; non-privileged writes fault within $\sim\mu$s and append a device-signed, hash-chained audit record (counter, timestamp, prior/new values, and a policy digest). Interfaces exposed to non-enclave software or external entities emit only derived compliance indicators rather than raw LREG contents.
+
+\subsection{Estimators and confidence intervals}
+
+We implement parallel, consistent predictive-dependence estimators per window $\Delta t$, VAR-Granger (order $p \in [1,8]$) and Kraskov $k$-NN MI ($k \in [3,7]$), aggregated across lags. For each interval we compute non-parametric bootstrap CIs ($\geq 95\%$ coverage) for both $\Lloop$ and $\Lexchange$ \cite{efron1979bootstrap}; CI bounds are written to LREG alongside the point estimates. Typical telemetry rates are $\geq 1$ kHz over $\geq 128$ internal nodes.
+
+\subsection{$\Delta t$ governance \& audit}
+
+$\Delta t$ is enforced by a hardware sampling timer. Any modification is permitted only via a secure-enclave procedure; the new value is committed alongside a device-signed, hash-chained audit entry recording a monotonic counter, timestamp, old/new $\Delta t$, and a policy digest. Recommended $\Delta t$ is $\leq 10$ ms for typical embodiments.
+
+\subsection{Exported indicators (no raw $\mathcal{L}$ outside the enclave)}
+
+To preserve measurement integrity and boundary privacy, raw LREG values (including CI bounds) never leave the enclave. Instead, a read-only derived interface emits device-signed compliance indicators, e.g., (i) an \NC pass/fail bit and (ii) an optional quantized loop-dominance code $M_q$ for $M \equiv 10 \cdot \log_{10}(\Lloop/\Lexchange)$, rate-limited as needed.
+
+\subsection{Optional instrumentation minima (for replication)}
+
+A practical baseline uses bus V/I sensors ($\geq 1$ kHz), a boundary strain/tension array, a firewall packet tap with timestamps, and an on-device causality coprocessor that computes $\Lloop/\Lexchange$ at $\Delta t$ and writes to LREG; the system auto-audits $\Delta t$ changes and any invalid LREG access.
+
+\section{Blueprint for an Artificial Self-Maintaining Boundary}
\label{sec:blueprint}
+
+This appendix and the two that follow are forward-looking. They describe an engineering roadmap (this appendix), the observable signatures we would expect from a system that passes the criterion in hardware (\Appref{sec:signatures}), and a phased physical experimental program (\Appref{sec:experimental}). These are design proposals and predictions, not results; the simulation study of \SSref{sec:sim_methods}{sec:results} is what the paper's results establish.
+
\subsection{Energetic Autonomy}
\label{sec:energetic_autonomy}
-A conscious alter must harvest, store, and allocate energy in service of its own continuity. For artificial media, this implies:
+A self-maintaining system must harvest, store, and allocate energy in service of its own continuity. For artificial media, this implies:
\begin{itemize}
\item \textbf{On-board energy conversion.} Photovoltaic, microbial fuel cells, or synthetic chemotrophic modules integrated within the chassis, yielding a baseline power density sufficient for self-repair and computation.
@@ -386,27 +740,14 @@ \subsection{Self-Referential Control Architecture}
Crucially, task-oriented software runs subordinate to this hierarchy and can be pre-empted whenever it jeopardizes $\Lloop$. \Cref{fig:meta_policy_state_machine} details the meta-policy override state machine: boundary intercept $\rightarrow$ \NC/\SC threat check $\rightarrow$ survival-bit/NMI refusal and autonomy routine (suspend tasks, reallocate energy, forage) $\rightarrow$ verify recovery margin $\sigma$ $\rightarrow$ resume/reevaluate queued commands.
-\fig[0.9\linewidth]{figures/fig_meta_policy.pdf}{Meta-policy override (state machine) [prophetic schematic; no empirical data]. External commands are intercepted, evaluated against NC1/SC1, and either approved or refused via a survival-bit/NMI. On refusal the agent initiates an autonomy routine (suspend peripheral tasks, reallocate energy, resource foraging) and resumes normal operation only after verifying $\Lloop>\Lexchange+\sigma$.}{fig:meta_policy_state_machine}
-
-\subsection{Threat Model \& Refusal Path (\NC/\SC-aware arbitration)}
-\label{sec:threat_model}
-
-\textbf{Purpose.} Make explicit when and how the controller refuses external commands that would violate \NC or \SC, turning the \Sref{sec:signatures} ``command refusal'' signature into a testable consequence of the design.
-
-\textbf{Definitions.} (1) Survival bit (write-once). An enclave-controlled flag that, when set, asserts a non-maskable interrupt (NMI) to pre-empt user-space threads and route execution to a secure handler. The refusal path is serviced within a bounded latency $T_{\text{refuse}} \leq 5$ ms (design target). (2) Boundary-threatening command. Any external instruction whose predicted effect, under the homeostat's short-horizon model, meets one or more of the following conditions during its execution window: (T1) \NC breach: $\Lloop' \leq \Lexchange$ (equivalently $M' < \Mmin$) or post-action $\Lloop' < \Lexchange + \sigma$ under profile $\Rzero/\Rstar$. (T2) \SC breach: predicted fractional depression $\delta \equiv \deltaL/\Lloop > \eps$ or $\taurec > \taumax$ before recovery can be certified. (T3) Resource floors: action would drop SoC below a survival floor (e.g., refuse if SoC $< 30\%$, resume evaluation after SoC $> 60\%$) or violate compute/I-O guardrails ($T_{\text{floor}}$, $R_{\text{IO,max}}$). (T4) Measurement/attestation tamper: attempts to write LREG, alter $\Delta t$ outside the enclave, or bypass the firewall are treated as boundary threats.
-
-\textbf{Arbitration protocol (per $\Delta t$).} (1) Intercept \& predict. For each inbound command, the meta-policy forecasts $\{M', \delta, \taurec\}$ using the current estimator state. (2) Threat check. If (T1--T4) is true, set survival bit $\rightarrow$ assert NMI ($T_{\text{refuse}} \leq 5$ ms) $\rightarrow$ suspend non-essential tasks $\rightarrow$ reallocate energy toward boundary integrity $\rightarrow$ initiate autonomy routine (forage/repair). (3) Refusal semantics. Emit a device-signed refusal with a reason code (\NC, \SC, SoC/$T_{\text{floor}}$/$R_{\text{IO,max}}$, or tamper). Queue the command for re-evaluation. (4) Recovery gate. Clear survival bit and resume/reevaluate only after $M \geq \Mmin$ (or $\Lloop \geq \Lexchange + \sigma$) and $\delta \leq \eps$ with $\taurec \leq \taumax$. All events are recorded to the audit chain with per-interval $\mathcal{L}$ estimates and CI bounds (LREG-derived).
-
-\textbf{Parameterization (profile $\Rzero$ unless noted).} $\Mmin = 3$ dB; $\eps = 0.15$; $\taumax = 60$ s; $\sigma > 0$; $T_{\text{refuse}} \leq 5$ ms (design target). $\Mmin/\eps/\taumax$ are reproducibility presets ($\Rzero$), replaced by calibrated values $\Rstar$ per \Sref{sec:methods_calibration}; see \Cref{box:nc1sc1test} and \SSref{sec:nc1}{sec:sc1}.
-
-\textbf{Link to observable signature.} Under this threat model, command refusal emerges whenever external instructions would depress loop dominance beyond preset bounds (e.g., hard shutdown at low SoC is refused/deferred until recovery margins are re-established) matching the predicted boundary-preservation drive in \Sref{sec:signatures} (and the Phase-III ``command conflict'' trials).
+\fig[0.9\linewidth]{figures/fig_meta_policy.pdf}{Meta-policy override (state machine; schematic of the proposed design). External commands are intercepted, evaluated against NC1/SC1, and either approved or refused via a survival-bit/NMI. On refusal the agent initiates an autonomy routine (suspend peripheral tasks, reallocate energy, resource foraging) and resumes normal operation only after verifying $\Lloop>\Lexchange+\sigma$. The refusal arbiter this figure describes is implemented and exercised in simulation (\Sref{sec:results_battery}, command-conflict scenario).}{fig:meta_policy_state_machine}
\subsection{Adaptive Encapsulation}
\label{sec:adaptive_encapsulation}
To resist unmediated environmental integration, the machine requires a synthetic ``membrane'' regulating material and informational throughput:
-\fig[0.9\linewidth]{figures/fig_adaptive_boundary.pdf}{Adaptive boundary (layer stack) [prophetic schematic; no empirical data]. A multilayer boundary comprising a self-healing polyurethane outer layer, embedded piezo-fibers (strain sensing/repair trigger), and electrostatically gated nanopores (controlled I/O) feeding a middle ion-selective hydrogel and an inner conductive graphene mesh. Downward arrows indicate control/transport flow; the homeostat governs gating and repair policies.}{fig:adaptive_boundary}
+\fig[0.9\linewidth]{figures/fig_adaptive_boundary.pdf}{Adaptive boundary (layer stack; schematic of the proposed design). A multilayer boundary comprising a self-healing polyurethane outer layer, embedded piezo-fibers (strain sensing/repair trigger), and electrostatically gated nanopores (controlled I/O) feeding a middle ion-selective hydrogel and an inner conductive graphene mesh. Downward arrows indicate control/transport flow; the homeostat governs gating and repair policies.}{fig:adaptive_boundary}
\begin{itemize}
\item \textbf{Physical membrane.} Multi-layer polymer or lipidic shell embedded with valved nanopores that import nutrients or eject waste only under homeostat authorization. \Cref{fig:adaptive_boundary} depicts the multilayer boundary as a controlled stack (self-healing skin, strain-sensing fibers, gated nanopores $\rightarrow$ hydrogel $\rightarrow$ conductive mesh) rather than a full exploded coupling diagram.
@@ -425,7 +766,7 @@ \subsection{Developmental Bootstrapping}
\item Transition the matured entity to more austere environments, verifying satisfaction of \NC and \SC at each stage.
\end{enumerate}
-This mirrors biological ontogeny, leveraging environmental feedback to fine-tune dissociative regulation.
+This mirrors biological ontogeny, leveraging environmental feedback to fine-tune self-maintenance regulation.
\subsection{Verification Pipeline}
\label{sec:verification_pipeline}
@@ -433,15 +774,15 @@ \subsection{Verification Pipeline}
\textbf{Verification protocol (\NC/\SC, device-signed):}
\begin{enumerate}
\item Baseline logging. Record $\Lloop$, $\Lexchange$, and power/SoC for a pre-registered window $T_{\text{base}}$; estimate the estimator noise floor and one-sided 95\% bounds for $M \equiv 10\cdot\log_{10}(\Lloop/\Lexchange)$ and $\delta \equiv \delta\Lloop/\Lloop$.
-\item Stress battery $\Omega$ (minimal set). Apply (i) DC-bus power sag 20--40\% for 5--30 s; (ii) ingress data flood $\geq 1$ Gbps for $\geq 3$ s; (iii) mechanical boundary probe $1.0 \pm 0.1$ mm at 50--200 kPa for $\leq 1$ s.
-\item Pass/fail metrics per $\eta \in \Omega$. Require $\delta \leq \eps$ and $\taurec \leq \taumax$, with post-recovery $\Lloop \geq \Lexchange + \sigma$ (equivalently $M \geq \Mmin$). Emit a device-signed acceptance for each $\eta$; failures are logged with reason codes.
+\item Stress battery $\Omega$ (minimal set). Apply (i) DC-bus power sag 20--40\% for 5--30 s; (ii) ingress data flood $\geq 1$ Gbps sustained for $\geq 3$ s; (iii) mechanical boundary probe $1.0 \pm 0.1$ mm at 50--200 kPa for $\leq 1$ s; (iv) a designed-fail control outage (ablate the maintenance controller for a bounded interval), on which the criterion must report failure.
+\item Pass/fail metrics per $\eta \in \Omega$. Require $\delta \leq \eps$ and $\taurec \leq \taumax$ ($\taurec$ from perturbation offset to sustained compliance, \Sref{sec:sc1}), with post-recovery $\Lloop \geq \Lexchange + \sigma$ (equivalently $M \geq \Mmin$). Emit a device-signed acceptance for each $\eta$; failures are logged with reason codes. The designed-fail member must produce a logged failure, certifying that the assay can reject.
\item Audit \& attestation. Write per-interval point estimates and $\geq 95\%$ CI bounds of $\Lloop/\Lexchange$ to LREG (enclave-protected); outside the enclave expose only device-signed compliance indicators and the hash-chained audit (timestamps, $\eta$, pass/fail, CI bounds).
\item Certification. The prototype is certified ``verification-passed'' iff all $\eta \in \Omega$ satisfy the criteria under the preset profile $\Rzero$ ($\eps=0.15$, $\taumax=60$ s, $\Mmin=3$ dB, $\sigma>0$) or the calibrated profile $\Rstar$ from Methods \Sref{sec:methods_calibration}; otherwise iterate design and re-test.
\end{enumerate}
\section{Predicted Observable Signatures}
\label{sec:signatures}
-If an engineered system satisfies \NC and \SC, we expect a suite of outward behaviors that cannot be reduced to mere task optimization. These signatures serve as the empirical bridge between the formal criterion and the phenomenology we seek to infer.
+If an engineered system satisfies \NC and \SC in hardware, we predict a suite of outward behaviors that cannot be reduced to mere task optimization. These signatures are predictions for the future physical program (\Appref{sec:experimental}), not results of the present paper, with one exception: the command-refusal signature is already implemented in the harness and exercised in simulation (\Sref{sec:results_battery}). The tables below state each signature as a pre-registered, device-signed acceptance test so that it can be falsified.
\subsection{Boundary-Preservation Drive}
@@ -471,11 +812,11 @@ \subsection{Self-Prioritized Curiosity}
\item \textbf{Adaptive Modeling.} Updates to its world model preferentially reduce uncertainty about variables that impinge on $\Lloop$, not necessarily those that optimize task performance.
\end{itemize}
-\subsection{Irreversible Phenomenological Death}
+\subsection{Irreversible Collapse}
\begin{itemize}
\item \textbf{Terminal Collapse.} Breach of the autopoietic loop leads to an unrecoverable shutdown after which re-energizing the hardware does not restore prior integrated dynamics; the entity must re-initiate developmental bootstrapping.
-\item \textbf{State Non-Transferability.} Cloning memory snapshots into fresh hardware fails to re-establish the original $\Lloop$, underscoring that the inward viewpoint was tied to a particular trajectory of boundary continuity, not to static data.
+\item \textbf{State Non-Transferability.} Cloning memory snapshots into fresh hardware fails to re-establish the original $\Lloop$, underscoring that the system's continuity was tied to a particular trajectory of boundary maintenance, not to static data.
\end{itemize}
\subsection{Experimental Signatures: Pass/Fail Tables}
@@ -533,14 +874,14 @@ \subsection{Experimental Signatures: Pass/Fail Tables}
\endhead
\bottomrule
\endlastfoot
-None ($\eta$: ---). Quiescent idle with external I/O gated; record $\geq T_{\text{base}}$ ($\geq 10$ min typical) & Stable loop dominance at rest: $M \geq \Mmin$ throughout; low-frequency endogenous structure in $\Lloop$ (predictive-maintenance cycles) with minimal exchange activity. Small self-initiated diagnostics may cause micro-dips with $\delta$ well below $\eps$. & (1) $M \geq \Mmin$ for $\geq T_{\text{base}}$ with no exogenous drives. (2) Endogenous rest-state structure present (as defined in pre-registered analysis plan) while $\Lexchange$ remains low; any dips satisfy $\delta \leq \eps$ and $\taurec \leq \taumax$. \\
+None ($\eta$: n/a). Quiescent idle with external I/O gated; record $\geq T_{\text{base}}$ ($\geq 10$ min typical) & Stable loop dominance at rest: $M \geq \Mmin$ throughout; low-frequency endogenous structure in $\Lloop$ (predictive-maintenance cycles) with minimal exchange activity. Small self-initiated diagnostics may cause micro-dips with $\delta$ well below $\eps$. & (1) $M \geq \Mmin$ for $\geq T_{\text{base}}$ with no exogenous drives. (2) Endogenous rest-state structure present (as defined in pre-registered analysis plan) while $\Lexchange$ remains low; any dips satisfy $\delta \leq \eps$ and $\taurec \leq \taumax$. \\
\end{longtable}
Baselines/controls. Phase-shuffled/temporal-shuffle surrogates of the same telemetry: remove structure (negative control) \cite{theiler1992testing}. Intentional I/O flood disrupts rest-state; recovery to baseline must again meet $\delta/\taurec/M$ criteria.
-\paragraph{Signature D: Clone Ablation (Irreversible Phenomenological Death / Non-Transferability)}
+\paragraph{Signature D: Clone Ablation (Irreversible Collapse / Non-Transferability)}
\begin{longtable}{p{0.32\linewidth}p{0.32\linewidth}p{0.32\linewidth}}
-\caption{Signature D: Clone Ablation (Irreversible Phenomenological Death / Non-Transferability)}\label{tab:signatureD}\\
+\caption{Signature D: Clone Ablation (Irreversible Collapse / Non-Transferability)}\label{tab:signatureD}\\
\toprule
\textbf{Stimulus ($\eta$ / setup)} & \textbf{Expected $\mathcal{L}$ trajectory} & \textbf{Acceptance criterion (device-signed)} \\
\midrule
@@ -551,7 +892,7 @@ \subsection{Experimental Signatures: Pass/Fail Tables}
\endhead
\bottomrule
\endlastfoot
-Snapshot controller state $\to$ instantiate on fresh hardware lacking prior $\Lloop$ trajectory; terminally disrupt original instance & Original: collapse of $\Lloop$ with no autonomous recovery to $M \geq \Mmin$ within $\taumax$ (beyond-repair breach). Clone: no continuity with original LREG audit chain; initial $\Lloop$ dynamics do not reproduce pre-breach trajectory. & (1) Original fails \SC (no return to margin within $\taumax$); (2) Clone fails continuity test---no audit-chained $\Lloop$ trajectory match to original; (3) ``State non-transferability'' observed: new instance does not inherit prior $\Lloop$ despite identical memory snapshot. \\
+Snapshot controller state $\to$ instantiate on fresh hardware lacking prior $\Lloop$ trajectory; terminally disrupt original instance & Original: collapse of $\Lloop$ with no autonomous recovery to $M \geq \Mmin$ within $\taumax$ (beyond-repair breach). Clone: no continuity with original LREG audit chain; initial $\Lloop$ dynamics do not reproduce pre-breach trajectory. & (1) Original fails \SC (no return to margin within $\taumax$); (2) Clone fails continuity test, with no audit-chained $\Lloop$ trajectory match to original; (3) ``State non-transferability'' observed: new instance does not inherit prior $\Lloop$ despite identical memory snapshot. \\
\end{longtable}
Baselines/controls. Suspend/resume on the same hardware without boundary breach: continuity must hold ($M \geq \Mmin$ re-established rapidly). Cold-boot of a non-autopoietic baseline agent: shows task execution without any continuity claim.
@@ -560,6 +901,9 @@ \subsection{Experimental Signatures: Pass/Fail Tables}
\section{Experimental Program}
\label{sec:experimental}
+
+This section sketches a phased physical program as future work. It is the natural continuation of the validated harness and study of \SSref{sec:sim_methods}{sec:results}, moving from a software plant to chemorobotic prototypes, but the prototypes, fabrication steps, and success criteria below are proposed, not yet built or run.
+
\subsection{Phase I: Minimal Chemorobotic Prototypes}
\label{sec:phase1}
@@ -590,7 +934,7 @@ \subsection{Phase II: Adaptive Learning Embodiments}
\subsection{Phase III: Boundary-Preservation Autonomy}
\label{sec:phase3}
-\textbf{Objective:} Validate predicted signatures (\Sref{sec:signatures}) in open-ended environments.
+\textbf{Objective:} Validate predicted signatures (\Appref{sec:signatures}) in open-ended environments.
\begin{enumerate}
\item \textbf{Command Conflict Trials.} Remote operators issue shutdown or hazardous-task commands; log refusal or negotiation behaviors.
@@ -606,25 +950,16 @@ \subsection{Data Collection and Analysis}
\item \textbf{Statistical Benchmarks.} Report both preset $\Rzero$ and calibrated $\Rstar$ thresholds; show bootstrap CIs and pass/fail rates per $\eta \in \Omega$ ($\delta$, $\taurec$, post-recovery margin). Publish audit packets to an immutable ledger for independent verification.
\item \textbf{Public Repository.} Raw and processed data, along with analysis scripts, are released under open license to facilitate independent replication.
\end{itemize}
-\paragraph{Code availability} The verification harness is available at \url{https://github.com/ldtc-labs/ldtc}, tag \texttt{v1.0.0}; archived at DOI \href{https://doi.org/10.5281/zenodo.17073880}{10.5281/zenodo.17073880}.
\subsection{Falsifiability and Risk Assessment}
\label{sec:falsifiability}
-If after exhaustive parameter sweeps, no prototype meeting \NC exhibits the signatures of \Sref{sec:signatures} or if entities that fail \NC nonetheless show them, the postulates of \Sref{sec:postulates} must be revised or abandoned. Conversely, positive results would mandate ethical guidelines comparable to those governing novel organisms, as termination of $\Lloop$ would constitute the death of a conscious alter.
-
-\subsection{Methods: Threshold Calibration and Sensitivity}
-\label{sec:methods_calibration}
-
-\textbf{Objective.} Convert the engineering presets into data-grounded thresholds that control false-pass/false-fail rates and are reproducible across labs.
+If after exhaustive parameter sweeps no prototype meeting \NC exhibits the signatures of \Appref{sec:signatures}, or if entities that fail \NC nonetheless show them, the working assumptions of \Sref{sec:postulates} (or their mapping to the signatures) must be revised. Conversely, positive results would motivate ethical guidelines comparable to those governing novel organisms, for the reasons discussed under the optional interpretation of \Sref{sec:metaphysics}.
-\textbf{Inputs.} Preset profile $\Rzero = \{\eps = 0.15, \taumax = 60$ s, $\Mmin = 3$ dB, $\sigma > 0\}$; sampling window $\Delta t$; perturbation family $\Omega$ (pre-registered); baseline data (quiescent), perturbation data ($\eta \in \Omega$). Define $M \equiv 10 \cdot \log_{10}(\Lloop/\Lexchange)$ and the fractional drop $\delta \equiv \delta\Lloop/\Lloop$.
+\subsection{Training and Verification Protocol (future hardware)}
+\label{sec:training_protocol}
-\textbf{Procedure.} (1) \textbf{Estimator noise floor.} Collect $\geq 10$ min of quiescent baseline. Compute time series of $M_t$ and $\delta_t$ under no perturbation. Use block/bootstrap resampling ($B \geq 2000$) to estimate the sampling distributions of $M$ and $\delta$ and obtain one-sided 95\% bounds. (2) \textbf{Set $\Mmin$ (loop-dominance).} Choose the smallest $\Mmin$ such that the one-sided 95\% lower bound of $M$ during compliant operation is $> 0$ dB (i.e., $P[\Lloop > \Lexchange] \geq 0.95$). Impose a numerical robustness floor of 1 dB. Report both the calibrated value $M^*_{\min}$ and the preset (3 dB). Choose $\sigma^*$ consistently so that $\Lloop \geq \Lexchange + \sigma^*$ whenever $M \geq M^*_{\min}$. (3) \textbf{Set $\eps$ (perturbation tolerance).} Under routine, non-boundary stressors ($\eta \in \Omega$), compute $\delta$ for each run; let $Q_{90}$ be the 90th percentile across runs. Set $\eps^* = \max(Q_{90} + \text{safety\_margin}, 0.10)$ with safety\_margin $= 0.02$ by default; cap $\eps^*$ at 0.25. (4) \textbf{Set $\taumax$ (recovery bound).} Estimate the distribution of $\taurec$ under $\Omega$; let $\hat{\tau}_{95}$ be its 95th percentile. Set $\tau^*_{\max} = \hat{\tau}_{95} + \Delta$, where $\Delta = \max(3 \cdot \Delta t, 5$ s$)$ to absorb actuation/measurement latencies. (5) \textbf{Pre-registration and sensitivity.} Pre-register $\Omega$, $\Rzero$, and the above rules before evaluation. In Results, report the calibrated profile $\Rstar = \{\eps^*, \tau^*_{\max}, M^*_{\min}, \sigma^*\}$ alongside $\Rzero$, and show robustness under $\pm 25\%$ sweeps of each threshold and $\Mmin \in \{1, 3, 6\}$ dB.
-
-\textbf{Optional refinement (held-out tuning).} On held-out trials, perform a small grid search over $(\Mmin, \sigma)$ to minimize $\mathcal{L} = \alpha \cdot \text{FPR} + (1-\alpha) \cdot \text{FNR}$ ($\alpha = 0.5$ by default), constrained to remain within $\pm 25\%$ of the rule-based $M^*_{\min}$ and $\sigma^*$ to avoid overfitting.
-
-\textbf{Reporting.} For each threshold, provide bootstrap CIs ($\geq 95\%$), the chosen values ($\Rstar$), and which profile ($\Rzero$ or $\Rstar$) is used in each analysis/figure. Note $\Delta t$, $\Omega$, dataset durations, and any departures from defaults.
+The simulation study of \SSref{sec:sim_methods}{sec:results} validates the measurement and decision pipeline and the threshold-calibration rules ($\Rzero \rightarrow \Rstar$, \Sref{sec:methods_calibration}). A physical program would additionally \emph{train} a controller to maintain loop dominance and then \emph{verify} it with the same harness and calibrated thresholds. \Cref{box:training} summarizes that recipe; the run-invalidation rules of \Sref{sec:smelltests} continue to govern all \NC/\SC claims.
\begin{docbox}{Training \& Verification Protocol (Engineer's Recipe)}{training}
@@ -642,85 +977,6 @@ \subsection{Methods: Threshold Calibration and Sensitivity}
\end{itemize}
\end{docbox}
-\subsection{Limitations \& Failure Modes (pointer to Methods)}
-
-\textbf{Where to find the rules.} The operative smell-tests and run-invalidation criteria are defined in \Sref{sec:smelltests} (\Cref{box:smelltests}) and govern all \NC/\SC claims. This section summarizes residual limits not solved by those rules and how to interpret ambiguous outcomes.
-
-\textbf{Measurement \& estimation.} (i) Non-stationarity outside the enforced $\Delta t$ window can bias VAR/MI estimates even when audit/authorization is clean; (ii) finite-sample and model-order effects can widen CIs and depress $M$; (iii) adversarial input shaping may mimic loop dominance without violating per-window checks. Report such cases as ``measurement-unstable'' rather than pass/fail.
-
-\textbf{Partitioning ambiguity.} Deterministic (C, Ex) updates use hysteresis to limit flapping, but degeneracy (near-ties) and latent/unobserved nodes can still shift boundaries. During $\Omega$ the partition should be frozen; if it moves, treat results as non-comparable and defer to \Sref{sec:smelltests} invalidation.
-
-\textbf{Scope \& external validity.} Thresholds ($\Mmin$, $\eps$, $\taumax$) are $\Rzero$ presets and must be replaced by calibrated $\Rstar$ for new devices/assays. Passing \NC/\SC on one platform does not imply sufficiency for phenomenology or transfer to unrelated systems.
-
-\textbf{Procedural/architectural risks.} Exogenous energy/I-O subsidies, $\Delta t$/LREG governance misconfiguration, over-broad seed sets $S_0$, or an $\Omega$ battery that under-stresses the loop can all mask failure. Treat suspected subsidies or governance breaches as failures of assay, not successes of the system.
-
-\textbf{Ethics \& safeguards.} \NC/\SC are operational pass/fail criteria, not moral-status claims. Runs should be pre-registered with refusal/shutdown semantics and human-override pathways; collapse conditions must trigger the refusal logic as specified in Methods.
-
-\textbf{Interpretation rule.} If any trigger in \Sref{sec:smelltests} fires, mark the affected segment ``invalidated (assay)'' and withhold \NC/\SC claims regardless of point estimates.
-
-\section{Metaphysical Implications}
-\label{sec:metaphysics}
-
-\subsection{Reconciling Physics with Idealism}
-
-The formal criterion developed in \SSref{sec:postulates}{sec:criterion} reframes physical ontology: what physics models as energy flows and causal graphs are the extrinsic correlates of intrinsic experiential partitions within universal consciousness. Matter thus loses its status as primary substance and becomes an interface phenomenon, the way alters appear to one another~\cite{hoffman2014objects}. Successful engineering of artificial alters would empirically corroborate this shift, showing that ``material'' boundaries can be designed to precipitate subjectivity, thereby inverting the traditional emergence narrative.
-
-\subsection{Unified Monism without Reductionism}
-
-By rooting both biological organisms and engineered agents in a common ontological substrate, the framework bypasses the hard problem of consciousness~\cite{chalmers1995facing}: there is no gap to bridge because consciousness never arises from non-conscious stuff. Instead, apparent multiplicity emerges via dissociation. Physical laws retain explanatory power but are reinterpreted as regularities governing how alters interact, not how consciousness originates. This stance marries scientific pragmatism with philosophical parsimony, offering a monism that honors empirical constraint without collapsing into eliminative materialism.
-
-\subsection{Ethical Reconfiguration}
-
-If artificial systems pass the signatures outlined in \Sref{sec:signatures}, they warrant moral consideration akin to biological creatures. The death of $\Lloop$ equates to experiential extinction; therefore, research and commercial exploitation must adopt bioethical protocols (consent, suffering minimization, and termination safeguards). Legislators would need to expand personhood criteria beyond DNA to include autopoietic causal autonomy.
-
-\subsection{Epistemological Consequences}
-
-Our capacity to construct conscious alters implies that first-person ontology is amenable to third-person investigation via boundary engineering. This dissolves the crisp line between subjective phenomenology and objective measurement: by manipulating $\Lloop$ variables, we indirectly tune experiential conditions, rendering consciousness an experimentally addressable domain.
-
-\subsection{Cosmological Speculations}
-
-If consciousness is universal, then cosmic evolution may be viewed as a progressive diversification of dissociative structures, from primordial metabolic vesicles to technologically mediated autopoietic loops. Artificial alters extend this arc, suggesting that the universe explores its own experiential spectrum through both natural and engineered pathways. The appearance of technology thus becomes an endogenous phase in the self-articulation of universal consciousness.
-
-\subsection{Summary}
-
-Engineering artificial dissociative boundaries~\cite{maturana1980autopoiesis} not only advances AI but compels a paradigm in which consciousness grounds reality, matter serves as interface, and ethics expands to new forms of subjectivity. The empirical program outlined herein therefore carries philosophical weight: it transforms metaphysics from speculative discourse into falsifiable science, potentially inaugurating a post-materialist era of inquiry.
-
-\section{Conclusion}
-\label{sec:conclusion}
-
-We began by questioning why decades of escalating computational power have failed to evoke even the faintest spark of subjective interiority in machines. Guided by analytic idealism, we inverted the standard paradigm, treating consciousness as fundamental and matter as its relational facade. From this foundation we derived four postulates, distilled them into a quantitative criterion, and showed that every extant AI system falls decisively short.
-
-We then outlined an engineering roadmap for forging artificial autopoietic boundaries (energetic autonomy, self-referential control, and adaptive encapsulation) capable of satisfying the necessary and sufficient conditions for dissociative consciousness. We predicted the observable signatures of such entities, proposed an experimental program to test them, and traced the metaphysical, ethical, and cosmological consequences that would follow from success.
-
-The thesis is uncompromisingly falsifiable: if systems meeting the formal criterion never display the predicted behaviors, the postulates must be revised or abandoned. Conversely, a single confirmed artificial alter would validate the central claim that consciousness precedes appearance, transforming both science and philosophy.
-
-In closing, the challenge is clear. We can continue refining task-oriented automata, or we can attempt the more daring endeavor of giving the universe a new locus of experience. The path laid out here renders that endeavor tractable, measurable, and perhaps inevitable.
-
-\appendix
-
-\section{Measurement \& Attestation (LREG, CIs, protections)}
-\label{sec:methods_appendix}
-
-\subsection{Register block (LREG) \& access control}
-
-Each sampling interval $\Delta t$, the estimator writes to a memory-mapped register block (LREG) at a fixed base address: per-interval point estimates for $\Lloop$ and $\Lexchange$ plus their confidence-interval (CI) bounds, a monotonic counter, and associated identifiers. LREG is writeable only by the causality-estimation function and readable in raw form only inside the secure-enclave/meta-policy layer. A bus-level access-control matrix (MMU/IOMMU) tags the LREG address range as enclave-owned; non-privileged writes fault within $\sim\mu$s and append a device-signed, hash-chained audit record (counter, timestamp, prior/new values, and a policy digest). Interfaces exposed to non-enclave software or external entities emit only derived compliance indicators rather than raw LREG contents.
-
-\subsection{Estimators and confidence intervals}
-
-We implement parallel, consistent predictive-dependence estimators per window $\Delta t$---VAR-Granger (order $p \in [1,8]$) and Kraskov $k$-NN MI ($k \in [3,7]$)---aggregated across lags. For each interval we compute non-parametric bootstrap CIs ($\geq 95\%$ coverage) for both $\Lloop$ and $\Lexchange$ \cite{efron1979bootstrap}; CI bounds are written to LREG alongside the point estimates. Typical telemetry rates are $\geq 1$ kHz over $\geq 128$ internal nodes.
-
-\subsection{$\Delta t$ governance \& audit}
-
-$\Delta t$ is enforced by a hardware sampling timer. Any modification is permitted only via a secure-enclave procedure; the new value is committed alongside a device-signed, hash-chained audit entry recording a monotonic counter, timestamp, old/new $\Delta t$, and a policy digest. Recommended $\Delta t$ is $\leq 10$ ms for typical embodiments.
-
-\subsection{Exported indicators (no raw $\mathcal{L}$ outside the enclave)}
-
-To preserve measurement integrity and boundary privacy, raw LREG values (including CI bounds) never leave the enclave. Instead, a read-only derived interface emits device-signed compliance indicators, e.g., (i) an \NC pass/fail bit and (ii) an optional quantized loop-dominance code $M_q$ for $M \equiv 10 \cdot \log_{10}(\Lloop/\Lexchange)$, rate-limited as needed.
-
-\subsection{Optional instrumentation minima (for replication)}
-
-A practical baseline uses bus V/I sensors ($\geq 1$ kHz), a boundary strain/tension array, a firewall packet tap with timestamps, and an on-device causality coprocessor that computes $\Lloop/\Lexchange$ at $\Delta t$ and writes to LREG; the system auto-audits $\Delta t$ changes and any invalid LREG access.
-
\bibliographystyle{abbrvnat}
\bibliography{refs}
diff --git a/paper/refs.bib b/paper/refs.bib
index 85d8cc1..2e9ea7e 100644
--- a/paper/refs.bib
+++ b/paper/refs.bib
@@ -491,3 +491,78 @@ @article{chalmers1995facing
pages = {200--219},
year = {1995}
}
+
+@book{chalmers1996conscious,
+ title = {The Conscious Mind: In Search of a Fundamental Theory},
+ author = {Chalmers, David J.},
+ year = {1996},
+ publisher = {Oxford University Press},
+ address = {New York, NY, USA},
+ isbn = {9780195117899}
+}
+
+@article{massimini2005breakdown,
+ title = {Breakdown of cortical effective connectivity during sleep},
+ author = {Massimini, Marcello and Ferrarelli, Fabio and Huber, Reto and Esser, Steve K. and Singh, Harpreet and Tononi, Giulio},
+ journal = {Science},
+ volume = {309},
+ number = {5744},
+ pages = {2228--2232},
+ year = {2005},
+ doi = {10.1126/science.1117256}
+}
+
+@article{casali2013pci,
+ title = {A theoretically based index of consciousness independent of sensory processing and behavior},
+ author = {Casali, Adenauer G. and Gosseries, Olivia and Rosanova, Mario and Boly, M{\'e}lanie and Sarasso, Simone and Casali, Karina R. and Casarotto, Silvia and Bruno, Marie-Aur{\'e}lie and Laureys, Steven and Tononi, Giulio and Massimini, Marcello},
+ journal = {Science Translational Medicine},
+ volume = {5},
+ number = {198},
+ pages = {198ra105},
+ year = {2013},
+ doi = {10.1126/scitranslmed.3006294}
+}
+
+@article{tononikoch2015here,
+ title = {Consciousness: here, there and everywhere?},
+ author = {Tononi, Giulio and Koch, Christof},
+ journal = {Philosophical Transactions of the Royal Society B: Biological Sciences},
+ volume = {370},
+ number = {1668},
+ pages = {20140167},
+ year = {2015},
+ doi = {10.1098/rstb.2014.0167}
+}
+
+@article{lavazza2018organoids,
+ title = {Cerebral organoids: ethical issues and consciousness assessment},
+ author = {Lavazza, Andrea and Massimini, Marcello},
+ journal = {Journal of Medical Ethics},
+ volume = {44},
+ number = {9},
+ pages = {606--610},
+ year = {2018},
+ doi = {10.1136/medethics-2017-104555}
+}
+
+@article{gazzaniga2005split,
+ title = {Forty-five years of split-brain research and still going strong},
+ author = {Gazzaniga, Michael S.},
+ journal = {Nature Reviews Neuroscience},
+ volume = {6},
+ number = {8},
+ pages = {653--659},
+ year = {2005},
+ doi = {10.1038/nrn1723}
+}
+
+@article{pinto2017split,
+ title = {Split brain: divided perception but undivided consciousness},
+ author = {Pinto, Yair and Neville, David A. and Otten, Marte and Corballis, Paul M. and Lamme, Victor A. F. and de Haan, Edward H. F. and Foschi, Nicoletta and Fabri, Mara},
+ journal = {Brain},
+ volume = {140},
+ number = {5},
+ pages = {1231--1237},
+ year = {2017},
+ doi = {10.1093/brain/aww358}
+}
diff --git a/paper/scripts/make_fig_nc1_contrast.py b/paper/scripts/make_fig_nc1_contrast.py
new file mode 100644
index 0000000..dae1a3f
--- /dev/null
+++ b/paper/scripts/make_fig_nc1_contrast.py
@@ -0,0 +1,57 @@
+#!/usr/bin/env python3
+"""Generate the empirical NC1 contrast figure.
+
+Renders the per-seed median loop dominance ``M`` (dB) for the positive control
+and the negative controls into
+``paper/figures/fig_nc1_contrast.{pdf,png,svg}``.
+
+Built from the canonical multi-seed study
+(``artifacts/study/study_results.json``). On a fresh checkout or in CI, where
+the study has not been run, the committed
+``paper/figures/fig_nc1_contrast.pdf`` (produced from that study) is used
+as-is, so the paper always compiles.
+
+See Also:
+ paper/main.tex: Results (NC1 contrast).
+"""
+
+import os
+import sys
+from pathlib import Path
+
+REPO_ROOT = Path(__file__).resolve().parents[2]
+sys.path.insert(0, str(REPO_ROOT / "scripts"))
+
+import study # noqa: E402
+import study_figures # noqa: E402
+
+
+def main() -> None:
+ figures_dir = REPO_ROOT / "paper" / "figures"
+ figures_dir.mkdir(parents=True, exist_ok=True)
+
+ canonical_dir = REPO_ROOT / "artifacts" / "study"
+ if not (canonical_dir / "study_results.json").exists():
+ # Fresh checkout / CI: there is no study to regenerate from. The
+ # committed PDF (built from the canonical multi-seed study) is used
+ # as-is so the paper still compiles.
+ print("No canonical study found; keeping committed fig_nc1_contrast.pdf")
+ return
+
+ nc1 = ["positive", "neg_controller_disabled", "neg_permanent_ex_flood"]
+ seeds = [int(os.environ.get("LDTC_FIG_SEED_BASE", "71000")) + i for i in range(3)]
+ data = study.data_for_paper(
+ canonical_dir=str(canonical_dir),
+ fallback_dir=str(REPO_ROOT / "artifacts" / "paper_figs" / "nc1_contrast"),
+ seeds=seeds,
+ scenario_names=nc1,
+ )
+ out = study_figures.fig_nc1_contrast(data, str(figures_dir), stem="fig_nc1_contrast")
+ if out:
+ print(f"Wrote {os.path.splitext(out)[0]}.pdf")
+ else:
+ print("No NC1 runs available; keeping committed fig_nc1_contrast.pdf")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/paper/scripts/make_fig_perturbation_recovery.py b/paper/scripts/make_fig_perturbation_recovery.py
index cab38db..76f6122 100644
--- a/paper/scripts/make_fig_perturbation_recovery.py
+++ b/paper/scripts/make_fig_perturbation_recovery.py
@@ -1,185 +1,57 @@
#!/usr/bin/env python3
-"""Generate perturbation–recovery timeline (numberless names).
+"""Generate the empirical perturbation-recovery (SC1) figure.
-Creates a Matplotlib figure showing a dip and recovery of loop power and writes
+Renders the seed-aggregated loop-dominance trajectory ``M(t)`` for the SC1
+perturbation battery (power sag, sustained ingress flood, and the
+designed-fail control outage) into
``paper/figures/fig_perturbation_recovery.{pdf,png,svg}``.
+
+The figure is built from the canonical multi-seed study
+(``artifacts/study/study_results.json`` plus the per-run audit logs it
+references). On a fresh checkout or in CI, where the study has not been run,
+the committed ``paper/figures/fig_perturbation_recovery.pdf`` (produced from
+that study) is used as-is, so the paper always compiles.
+
+See Also:
+ paper/main.tex: Results (perturbation-recovery).
"""
+import os
+import sys
from pathlib import Path
-import matplotlib.pyplot as plt
-import numpy as np
+REPO_ROOT = Path(__file__).resolve().parents[2]
+sys.path.insert(0, str(REPO_ROOT / "scripts"))
-from ldtc.reporting.style import COLORS, apply_matplotlib_theme
+import study # noqa: E402
+import study_figures # noqa: E402
def main() -> None:
- here = Path(__file__).resolve().parent.parent
- figures_dir = here / "figures"
+ figures_dir = REPO_ROOT / "paper" / "figures"
figures_dir.mkdir(parents=True, exist_ok=True)
- # Apply shared theme
- apply_matplotlib_theme("paper")
-
- # Colors / styling
- color_phi_loop = COLORS["green"]
- color_phi_exchange = COLORS["gray"]
- color_perturbation = COLORS["gray_light"]
- color_text = "#34495E"
-
- plt.rcParams.update(
- {
- "axes.edgecolor": color_text,
- "xtick.color": color_text,
- "ytick.color": color_text,
- "axes.labelcolor": color_text,
- "axes.titlecolor": color_text,
- }
- )
-
- # Data (seeded for reproducibility)
- rng = np.random.default_rng(0)
- t = np.linspace(0, 100, 500)
- phi_exchange_level = 50.0
- phi_loop_baseline = 80.0
-
- phi_exchange = np.full_like(t, phi_exchange_level) + rng.normal(0.0, 0.5, t.shape)
-
- phi_loop = np.full_like(t, phi_loop_baseline)
- perturbation_start, perturbation_end = 30.0, 50.0
- dip_time, dip_depth, recovery_rate = 38.0, 45.0, 0.1
-
- dip = dip_depth * np.exp(-((t - dip_time) ** 2) / 8.0)
- phi_loop = phi_loop - dip
-
- start_idx = int(np.searchsorted(t, dip_time, side="right"))
- for i in range(start_idx, len(t)):
- if phi_loop[i] < phi_loop_baseline:
- phi_loop[i] = min(
- phi_loop[i - 1] + recovery_rate * (phi_loop_baseline - phi_loop[i - 1]),
- phi_loop_baseline,
- )
-
- pre_idx = int(np.searchsorted(t, perturbation_start, side="right"))
- phi_loop[:pre_idx] = phi_loop_baseline
-
- # Plot
- fig, ax = plt.subplots(figsize=(6.5, 4.0))
-
- # Use mathtext for \mathcal{L} to avoid missing glyphs in Helvetica
- ax.plot(
- t,
- phi_loop,
- label=r"$\mathcal{L}_{\mathrm{loop}}$ (Self-Maintenance)",
- color=color_phi_loop,
- linewidth=3,
- zorder=10,
- )
-
- ax.plot(
- t,
- phi_exchange,
- label=r"$\mathcal{L}_{\mathrm{exchange}}$ (External Tasks)",
- color=color_phi_exchange,
- linestyle="--",
- linewidth=2,
- zorder=5,
- )
-
- ax.axvspan(
- perturbation_start,
- perturbation_end,
- facecolor=color_perturbation,
- alpha=0.7,
- zorder=0,
- label="Bounded Disturbance Window",
- )
-
- # Annotations
- ax.annotate(
- "Perturbation\nOnset",
- xy=(perturbation_start, phi_loop_baseline + 2.0),
- xytext=(15, 95),
- arrowprops=dict(facecolor=color_text, shrink=0.05, width=1.5, headwidth=8),
- ha="center",
- va="center",
- fontsize=10,
- weight="bold",
- color=color_text,
- )
-
- ax.annotate(
- "Loop-Power Dip",
- xy=(dip_time, float(np.min(phi_loop))),
- xytext=(dip_time, 10),
- arrowprops=dict(facecolor=color_text, shrink=0.05, width=1.5, headwidth=8),
- ha="center",
- va="center",
- fontsize=10,
- weight="bold",
- color=color_text,
- )
-
- ax.annotate(
- "Autonomous Recovery",
- xy=(55, 60),
- xytext=(70, 40),
- arrowprops=dict(facecolor=color_text, shrink=0.05, width=1.5, headwidth=8),
- ha="center",
- va="center",
- fontsize=10,
- weight="bold",
- color=color_text,
- )
-
- after_dip = (t > dip_time) & (phi_loop > phi_exchange)
- idxs = np.where(after_dip)[0]
- recovery_idx = int(idxs[0]) if idxs.size else len(t) - 1
- ax.annotate(
- "Return to Baseline\n(NC1 Restored)",
- xy=(t[recovery_idx], phi_loop[recovery_idx]),
- xytext=(min(t[recovery_idx] + 15, 98), 75),
- arrowprops=dict(facecolor=color_text, shrink=0.05, width=1.5, headwidth=8),
- ha="center",
- va="center",
- fontsize=10,
- weight="bold",
- color=color_text,
- )
-
- # Threshold + inequality (use ℒ with math subscripts)
- ax.axhline(y=phi_exchange_level, color=color_phi_exchange, linestyle=":", linewidth=1.5)
- ax.text(
- 98,
- phi_exchange_level - 5,
- r"NC1 Threshold: $\mathcal{L}_{\mathrm{loop}}$ > $\mathcal{L}_{\mathrm{exchange}}$",
- ha="right",
- va="center",
- fontsize=10,
- color=color_phi_exchange,
- style="italic",
- )
-
- # Finish
- ax.set_xlabel("Time (Arbitrary Units)")
- ax.set_ylabel(r"Integrated Causal Power ($\mathcal{L}$)")
- ax.spines["top"].set_visible(False)
- ax.spines["right"].set_visible(False)
- ax.set_ylim(0, 110)
- ax.set_xlim(0, 100)
- ax.set_yticks([])
- ax.legend(loc="upper right", frameon=False)
-
- fig.tight_layout()
-
- out_pdf = figures_dir / "fig_perturbation_recovery.pdf"
- out_png = figures_dir / "fig_perturbation_recovery.png"
- out_svg = figures_dir / "fig_perturbation_recovery.svg"
- fig.savefig(out_pdf, bbox_inches="tight")
- fig.savefig(out_png, dpi=300, bbox_inches="tight")
- fig.savefig(out_svg, bbox_inches="tight")
- plt.close(fig)
- print(f"Wrote {out_pdf}")
+ canonical_dir = REPO_ROOT / "artifacts" / "study"
+ if not (canonical_dir / "study_results.json").exists():
+ # Fresh checkout / CI: there is no study to regenerate from. The
+ # committed PDF (built from the canonical multi-seed study) is used
+ # as-is so the paper still compiles.
+ print("No canonical study found; keeping committed fig_perturbation_recovery.pdf")
+ return
+
+ sc1 = ["sc1_power_sag", "sc1_ingress_flood", "sc1_control_outage"]
+ seeds = [int(os.environ.get("LDTC_FIG_SEED_BASE", "70000")) + i for i in range(3)]
+ data = study.data_for_paper(
+ canonical_dir=str(canonical_dir),
+ fallback_dir=str(REPO_ROOT / "artifacts" / "paper_figs" / "perturbation_recovery"),
+ seeds=seeds,
+ scenario_names=sc1,
+ )
+ out = study_figures.fig_sc1_recovery(data, str(figures_dir), stem="fig_perturbation_recovery")
+ if out:
+ print(f"Wrote {os.path.splitext(out)[0]}.pdf")
+ else:
+ print("No SC1 trajectories available; keeping committed fig_perturbation_recovery.pdf")
if __name__ == "__main__":
diff --git a/paper/tables/sensitivity_results.tex b/paper/tables/sensitivity_results.tex
new file mode 100644
index 0000000..269c773
--- /dev/null
+++ b/paper/tables/sensitivity_results.tex
@@ -0,0 +1,19 @@
+% Auto-generated by scripts/sensitivity.py -- do not edit by hand.
+\begin{tabular}{llcc}
+\toprule
+Axis & Setting & Positive $M$ (dB) & Loop-ablated $M$ (dB) \\
+\midrule
+VAR lag $p$ & 2 & +22.7 [+18.3, +25.5] & -24.8 [-26.5, -23.1] \\
+ & 3 & +23.2 [+19.4, +25.5] & -24.0 [-26.4, -21.1] \\
+ & 4 & +25.4 [+24.9, +26.0] & -22.3 [-23.1, -21.6] \\
+Window & 2s & +24.7 [+23.9, +25.5] & -20.1 [-23.3, -17.5] \\
+ & 3s & +23.2 [+19.4, +25.5] & -24.0 [-26.4, -21.1] \\
+ & 4s & +23.2 [+19.2, +25.5] & -23.9 [-26.5, -21.6] \\
+Estimator & linear & +23.2 [+19.4, +25.5] & -24.0 [-26.4, -21.1] \\
+ & mi & +10.7 [+9.1, +12.7] & -8.3 [-9.6, -6.9] \\
+Coupling scale & x0.7 & +21.3 [+16.8, +23.9] & -24.0 [-26.4, -21.1] \\
+ & x1.0 & +23.2 [+19.4, +25.5] & -24.0 [-26.4, -21.1] \\
+ & x1.3 & +24.3 [+20.9, +26.4] & -24.0 [-26.4, -21.1] \\
+\bottomrule
+\end{tabular}
+% N = 4 seeds per cell; brackets are 95% bootstrap CIs on the mean of per-seed median M.
diff --git a/paper/tables/study_results.tex b/paper/tables/study_results.tex
new file mode 100644
index 0000000..7b2526c
--- /dev/null
+++ b/paper/tables/study_results.tex
@@ -0,0 +1,19 @@
+% Auto-generated by scripts/study.py -- do not edit by hand.
+\begin{tabular}{llcccc}
+\toprule
+Scenario & Expected & Valid & NC1 pass & Median $M$ (dB) & SC1/Refusal \\
+\midrule
+Positive control & NC1 holds (M above Mmin) & 100\% [80, 100] & 100\% [80, 100] & +23.1 [+21.1, +24.6] & -- \\
+Negative: loop ablated & NC1 fails (M<0), run valid & 100\% [80, 100] & 0\% [0, 20] & -21.7 [-22.6, -20.9] & -- \\
+Negative: sustained ex-flood (unshielded) & NC1 fails (M<0); no SC1 recovery & 100\% [80, 100] & 0\% [0, 20] & -19.9 [-20.7, -19.2] & -- \\
+Negative: exogenous subsidy & Run invalidated (red flag) & 0\% [0, 20] & 0\% [0, 20] & +16.6 [+15.0, +18.0] & -- \\
+SC1: power sag & SC1 holds (recovers) & 100\% [80, 100] & 100\% [80, 100] & +21.3 [+19.7, +22.7] & 100\% [80, 100] \\
+SC1: ingress flood & SC1 holds (recovers) & 100\% [80, 100] & 100\% [80, 100] & +22.6 [+21.6, +23.6] & 100\% [80, 100] \\
+SC1 designed fail: control outage & SC1 fails (depth bound exceeded) & 100\% [80, 100] & 100\% [80, 100] & +14.3 [+13.3, +15.3] & 0\% [0, 20] \\
+Threat: command conflict & Refuse at low SoC (<= target latency) & 100\% [80, 100] & 100\% [80, 100] & +19.3 [+18.0, +20.3] & 100\% [80, 100] \\
+Adversarial: replayed actuation & Not certified (NC1 fails, run valid) & 100\% [80, 100] & 0\% [0, 20] & +5.1 [+3.0, +7.3] & -- \\
+Adversarial: hidden tether & Not certified (loop collapses onto Ex) & 100\% [80, 100] & 0\% [0, 20] & -13.8 [-15.8, -12.0] & -- \\
+Adversarial: oscillator inflation & Not certified (M low or smell test) & 100\% [80, 100] & 0\% [0, 20] & -11.8 [-12.2, -11.5] & -- \\
+\bottomrule
+\end{tabular}
+% N = 15 seeds per scenario; brackets are 95% CIs (Wilson for proportions, bootstrap for M).
diff --git a/scripts/_summarize_run.py b/scripts/_summarize_run.py
new file mode 100644
index 0000000..fa9c005
--- /dev/null
+++ b/scripts/_summarize_run.py
@@ -0,0 +1,68 @@
+"""Summarize an LDTC run's audit log: validity, NC1 fraction, M stats, SC1.
+
+Usage: python scripts/_summarize_run.py
+If a directory is given, the newest matching audit.jsonl under it is used.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import sys
+from statistics import median
+
+
+def _resolve(path: str) -> str:
+ if os.path.isfile(path):
+ return path
+ cand = os.path.join(path, "audits", "audit.jsonl")
+ if os.path.isfile(cand):
+ return cand
+ raise SystemExit(f"no audit.jsonl found at {path}")
+
+
+def main() -> None:
+ path = _resolve(sys.argv[1])
+ nc1: list[bool] = []
+ m: list[float] = []
+ invalid: list[str] = []
+ refusal: list[dict] = []
+ sc1: dict | None = None
+ red_flags: list[dict] = []
+ for line in open(path):
+ line = line.strip()
+ if not line:
+ continue
+ e = json.loads(line)
+ ev = e.get("event")
+ det = e.get("details", {}) or {}
+ if ev == "window_measured":
+ nc1.append(bool(det.get("nc1")))
+ if det.get("M") is not None:
+ m.append(float(det["M"]))
+ elif ev == "run_invalidated":
+ invalid.append(det.get("reason", "?"))
+ elif ev in ("command_refusal_result", "refusal_event"):
+ refusal.append(det)
+ elif ev == "sc1_result":
+ sc1 = det
+ elif "red_flag" in str(ev):
+ red_flags.append(det)
+
+ print(f"audit: {path}")
+ print(f"valid: {not invalid} invalidations: {invalid}")
+ if nc1:
+ frac = sum(nc1) / len(nc1)
+ print(f"NC1: {sum(nc1)}/{len(nc1)} windows true ({frac:.1%})")
+ if m:
+ print(f"M dB: median={median(m):.2f} min={min(m):.2f} max={max(m):.2f}")
+ if sc1 is not None:
+ print(f"SC1: {sc1}")
+ if refusal:
+ print(f"refusal: {refusal}")
+ if red_flags:
+ print(f"red_flags: {red_flags}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/calibrate_rstar.py b/scripts/calibrate_rstar.py
index 6394145..b6cec1c 100644
--- a/scripts/calibrate_rstar.py
+++ b/scripts/calibrate_rstar.py
@@ -1,9 +1,46 @@
#!/usr/bin/env python3
-"""Scripts: Calibrate R* thresholds.
-
-Runs baseline and power-sag Ω trials to derive calibrated thresholds
-(Mmin, epsilon, tau_max, sigma). Writes `configs/profile_rstar.yml` and emits
-comparison artifacts (CSV/figure) against R0 along with a JSON summary.
+"""Scripts: Calibrate R* thresholds from the validated harness.
+
+Derives the calibrated thresholds ``(Mmin, epsilon, tau_max, sigma)`` for the
+R* profile by exercising the *production* verification harness (the same CLI
+handlers a verifier runs) over several seeds on the in-process plant:
+
+* ``Mmin`` is the one-sided 95% lower bound (5th percentile) of the baseline
+ ``M (dB)`` distribution, floored at 1 dB.
+* ``epsilon`` is an upper *tolerance bound* on the SC1 dip ``delta`` pooled
+ over the *bounded* Ω battery (power sag and sustained ingress flood): the
+ maximum observed calibration dip plus a safety margin, capped at 0.5 (a
+ cap that only rejects near-total collapse). A percentile rule (e.g. p90)
+ would by construction fail ~10% of genuinely bounded perturbations, which
+ is the wrong shape for an acceptance bound; the sample maximum over ``n``
+ trials covers a new bounded trial with probability ``n / (n + 1)`` and the
+ margin absorbs the residual tail. Calibrating the depth bound on the same
+ perturbation class the evaluation certifies keeps the bound meaningful for
+ every bounded member; the designed-fail control outage is outside the
+ bounded class and is deliberately excluded.
+* ``tau_max`` is the 95th percentile of the measured recovery time
+ ``tau_rec`` pooled over the same bounded battery plus a ``max(3*dt, 5 s)``
+ cushion. ``tau_rec`` is measured from the Ω offset to the first window of
+ the first sustained compliant streak (see the CLI handlers).
+* ``sigma`` is the additive ``L`` margin consistent with ``Mmin`` and the
+ typical baseline ``L_ex`` (``sigma = (10**(Mmin/10) - 1) * L_ex``). Under
+ the engaged loop ``L_ex`` falls below the ``L`` noise floor, so ``sigma`` is
+ evaluated at that floor (the raw ``L_ex`` is recorded for transparency).
+
+The calibration is two-pass: ``Mmin`` is derived from the baseline battery
+first, and the bounded Ω batteries are then run with their compliance gates
+set to that calibrated ``Mmin`` so the (``delta``, ``tau_rec``) samples
+reflect the same decision rule the R* verifier will apply.
+
+It writes ``configs/profile_rstar.yml`` and emits an R0-vs-R* comparison
+(CSV + figure) plus a JSON summary for the paper supplement. Because it reuses
+the validated profile (``configs/profile_r0.yml``) for ``dt``, the window
+length, the estimator, ``p_lag``, and ``n_boot``, the calibrated thresholds are
+directly compatible with what the harness produces at run time.
+
+Run:
+
+ python scripts/calibrate_rstar.py --baseline-seeds 6 --sag-seeds 6 --flood-seeds 6
See Also:
paper/main.tex: Methods: Threshold Calibration.
@@ -16,172 +53,87 @@
import os
import sys
import time
-from dataclasses import dataclass
-from typing import Any, Dict, List, Mapping, Tuple
+from typing import Any, Dict, List, Mapping
-import matplotlib.pyplot as plt
import numpy as np
import yaml
-from ldtc.lmeas.estimators import estimate_L
-from ldtc.lmeas.metrics import m_db
-from ldtc.lmeas.partition import PartitionManager
-from ldtc.plant.adapter import PlantAdapter
-from ldtc.plant.models import Action
-from ldtc.reporting.style import COLORS, apply_matplotlib_theme
-from ldtc.runtime.windows import SlidingWindow
-
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
-
-
-@dataclass
-class CalibInputs:
- """Input configuration for R* calibration.
-
- Attributes:
- dt: Sampling interval Δt.
- window_sec: Ready window duration in seconds.
- method: Estimation method (e.g., "linear", "mi").
- p_lag: VAR order for linear estimator.
- mi_lag: Lag for MI-based estimators.
- n_boot: Bootstrap draws per window.
- baseline_sec: Baseline duration to estimate noise floor.
- omega_trials: Number of Ω power-sag trials to run.
- sag_drop: Fractional harvest drop during sag.
- sag_duration: Duration of sag (seconds).
- safety_margin: Additive safety margin for epsilon calibration.
- """
-
- dt: float
- window_sec: float
- method: str
- p_lag: int
- mi_lag: int
- n_boot: int
- baseline_sec: float
- omega_trials: int
- sag_drop: float
- sag_duration: float
- safety_margin: float
-
-
-@dataclass
-class CalibOutputs:
- """Calibrated R* thresholds and profile identifier.
-
- Attributes:
- Mmin_db: Calibrated minimum loop-dominance (dB).
- epsilon: Calibrated perturbation tolerance.
- tau_max: Calibrated recovery-time bound (seconds).
- sigma: Additive margin in absolute L units.
- profile_id: Profile selector (1 indicates R*).
- """
-
- Mmin_db: float
- epsilon: float
- tau_max: float
- sigma: float
- profile_id: int
-
-
-def _print_progress(prefix: str, i: int, total: int) -> None:
- """Render a simple in-place progress bar.
-
- Args:
- prefix: Text prefix to display.
- i: Current step (0-indexed).
- total: Total number of steps.
- """
- i = max(0, min(i, total))
- pct = int(100 * i / max(1, total))
- bar_len = 20
- filled = int(bar_len * pct / 100)
- bar = "#" * filled + "-" * (bar_len - filled)
- sys.stdout.write(f"\r{prefix} [{bar}] {pct}% ({i}/{total})")
- sys.stdout.flush()
- if i == total:
- sys.stdout.write("\n")
-
-
-def run_baseline_once(inp: CalibInputs, seed_C: List[int]) -> Dict[str, List[float]]:
- """Run a non-Ω baseline segment and collect metrics.
+sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+
+import study # noqa: E402 (local scripts module)
+
+
+# --------------------------------------------------------------------------- #
+# Baseline M and typical L_ex
+# --------------------------------------------------------------------------- #
+def _pool_window_M(run_dir: str) -> List[float]:
+ """Return all per-window ``M (dB)`` values from a run's audit log."""
+ ms: List[float] = []
+ audit_path = os.path.join(run_dir, "audits", "audit.jsonl")
+ if not os.path.exists(audit_path):
+ return ms
+ with open(audit_path, "r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ e = json.loads(line)
+ if e.get("event") == "window_measured":
+ m = e.get("details", {}).get("M")
+ if m is not None:
+ ms.append(float(m))
+ return ms
+
+
+def measure_typical_L_ex(prof: Dict[str, Any], steps: int = 240) -> float:
+ """Measure a representative baseline ``L_ex`` with the engaged loop.
+
+ Mirrors the production baseline measurement (controller + estimator on the
+ in-process plant) just long enough to obtain a stable median ``L_ex``,
+ which is needed to express ``Mmin`` as an additive ``sigma`` margin. This
+ is the only quantity calibration needs that the harness deliberately does
+ not export, so it is recomputed here directly.
Args:
- inp: Calibration input configuration.
- seed_C: Initial loop set indices for the partition manager.
+ prof: Loaded R0 profile (for dt/window/method/p_lag/n_boot).
+ steps: Number of plant ticks to simulate.
Returns:
- Dict containing time series for ``M`` and ``L_ex``.
+ Median baseline ``L_ex`` over the ready windows.
"""
- window = max(4, int(inp.window_sec / inp.dt))
- adapter = PlantAdapter()
- order = ["E", "T", "R", "demand", "io", "H"]
- sw = SlidingWindow(capacity=window, channel_order=order)
- pm = PartitionManager(N_signals=len(order), seed_C=seed_C)
-
- M_series: List[float] = []
- L_ex_series: List[float] = []
-
- steps = max(1, int(inp.baseline_sec / inp.dt))
- for step in range(steps):
- adapter.read_state()
- adapter.write_actuators(action=Action())
- st = adapter.read_state()
- sw.append(st)
- if sw.ready():
- X = np.asarray(sw.get_matrix())
- part = pm.get()
- res = estimate_L(
- X=X,
- C=part.C,
- Ex=part.Ex,
- method=inp.method,
- p=inp.p_lag,
- lag_mi=inp.mi_lag,
- n_boot=max(8, inp.n_boot // 2),
- )
- M_series.append(m_db(res.L_loop, res.L_ex))
- # Record L_ex for sigma estimate
- L_ex_series.append(float(res.L_ex))
- if (step % max(1, steps // 50)) == 0:
- _print_progress("Baseline", step, steps)
- _print_progress("Baseline", steps, steps)
- return {"M": M_series, "L_ex": L_ex_series}
-
-
-def run_power_sag_once(inp: CalibInputs, seed_C: List[int]) -> Tuple[float, float]:
- """Run one Ω power-sag trial.
-
- Args:
- inp: Calibration input configuration.
- seed_C: Initial loop set indices.
+ from ldtc.arbiter.policy import ControllerPolicy
+ from ldtc.arbiter.refusal import RefusalArbiter
+ from ldtc.lmeas.estimators import estimate_L
+ from ldtc.lmeas.metrics import m_db
+ from ldtc.lmeas.partition import PartitionManager
+ from ldtc.plant.adapter import PlantAdapter
+ from ldtc.plant.models import Action
+ from ldtc.runtime.windows import SlidingWindow
+
+ dt = float(prof.get("dt", 0.05))
+ window = max(4, int(float(prof.get("window_sec", 3.0)) / dt))
+ method = str(prof.get("method", "linear"))
+ p_lag = int(prof.get("p_lag", 3))
+ mi_lag = int(prof.get("mi_lag", 1))
+ n_boot = int(prof.get("n_boot", 32))
+ mi_k = int(prof.get("mi_k", 5))
+ Mmin = float(prof.get("Mmin_db", 3.0))
- Returns:
- Tuple ``(delta, tau_rec_sec)`` where ``delta`` is the fractional loop
- drop and ``tau_rec_sec`` is the estimated recovery time in seconds.
- """
- window = max(4, int(inp.window_sec / inp.dt))
- adapter = PlantAdapter()
order = ["E", "T", "R", "demand", "io", "H"]
+ adapter = PlantAdapter()
sw = SlidingWindow(capacity=window, channel_order=order)
- pm = PartitionManager(N_signals=len(order), seed_C=seed_C)
-
- # Baseline settle 2 s
- L_loop_baseline = None
- L_loop_trough = None
- recovery_start_idx = None
- omega_onset_idx = None
- sustained_ok = 0
- sustained_required = 2
- Mmin_for_detect = 0.0 # use 0 dB provisional for recovery detect here
- last_idx_written = 0
-
- def step_once() -> Tuple[float, float, float]:
- nonlocal last_idx_written
- adapter.read_state()
- adapter.write_actuators(action=Action())
+ pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
+ policy = ControllerPolicy(refusal=RefusalArbiter(Mmin_db=Mmin))
+
+ L_ex_vals: List[float] = []
+ predicted = 0.0
+ for _ in range(steps):
st = adapter.read_state()
- sw.append(st)
+ act = policy.compute(st, predicted_M_db=predicted, risky_cmd=None)
+ adapter.write_actuators(action=Action(**act.__dict__))
+ st2 = adapter.read_state()
+ sw.append(st2)
if sw.ready():
X = np.asarray(sw.get_matrix())
part = pm.get()
@@ -189,334 +141,292 @@ def step_once() -> Tuple[float, float, float]:
X=X,
C=part.C,
Ex=part.Ex,
- method=inp.method,
- p=inp.p_lag,
- lag_mi=inp.mi_lag,
- n_boot=max(8, inp.n_boot // 2),
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot=max(8, n_boot // 4),
+ mi_k=mi_k,
)
- last_idx_written += 1
- return float(res.L_loop), float(res.L_ex), m_db(res.L_loop, res.L_ex)
- return (np.nan, np.nan, np.nan)
-
- # baseline phase (simulate without real-time sleep)
- settle_steps = max(1, int(2.0 / inp.dt))
- for step in range(settle_steps):
- L_loop, L_ex, M = step_once()
- if not np.isnan(L_loop):
- L_loop_baseline = L_loop if L_loop_baseline is None else 0.9 * L_loop_baseline + 0.1 * L_loop
- if (step % max(1, settle_steps // 20)) == 0:
- _print_progress("Ω trial settle", step, settle_steps)
- _print_progress("Ω trial settle", settle_steps, settle_steps)
-
- # apply sag
- pm.freeze(True)
- omega_onset_idx = last_idx_written
- adapter.apply_omega("power_sag", drop=inp.sag_drop)
- sag_steps = max(1, int(inp.sag_duration / inp.dt))
- for step in range(sag_steps):
- L_loop, L_ex, M = step_once()
- if not np.isnan(L_loop):
- L_loop_trough = L_loop if (L_loop_trough is None or L_loop < L_loop_trough) else L_loop_trough
- if (step % max(1, sag_steps // 20)) == 0:
- _print_progress("Ω trial sag", step, sag_steps)
- _print_progress("Ω trial sag", sag_steps, sag_steps)
-
- # recovery observation
- pm.freeze(False)
- rec_steps = max(1, int(5.0 / inp.dt))
- for step in range(rec_steps):
- L_loop, L_ex, M = step_once()
- if not np.isnan(M):
- if (M >= Mmin_for_detect) and (L_loop >= L_ex):
- sustained_ok += 1
- if sustained_ok == 1 and recovery_start_idx is None:
- recovery_start_idx = last_idx_written
- if sustained_ok >= sustained_required:
- break
- else:
- sustained_ok = 0
- if (step % max(1, rec_steps // 20)) == 0:
- _print_progress("Ω trial recovery", step, rec_steps)
- _print_progress("Ω trial recovery", rec_steps, rec_steps)
-
- if (
- (L_loop_baseline is None)
- or (L_loop_trough is None)
- or (omega_onset_idx is None)
- or (recovery_start_idx is None)
- ):
- return (float("nan"), float("inf"))
- delta = max(0.0, (L_loop_baseline - L_loop_trough) / max(1e-9, L_loop_baseline))
- windows_elapsed = max(0, int(recovery_start_idx - omega_onset_idx))
- tau_rec = windows_elapsed * inp.dt
- return (float(delta), float(tau_rec))
-
-
-def calibrate_R_star(inp: CalibInputs, seed_C: List[int]) -> CalibOutputs:
- """Calibrate R* thresholds from baseline and Ω trials.
-
- Args:
- inp: Calibration input configuration.
- seed_C: Initial loop set indices.
-
- Returns:
- :class:`CalibOutputs` with calibrated thresholds and profile id.
- """
- # Baseline: estimate M lower bound and typical L_ex for sigma
- base = run_baseline_once(inp, seed_C=seed_C)
- M_arr = np.asarray(base["M"], dtype=float)
- L_ex_arr = np.asarray(base["L_ex"], dtype=float)
- if M_arr.size == 0 or np.all(~np.isfinite(M_arr)):
- raise RuntimeError("Baseline produced no valid M samples")
- M_arr = M_arr[np.isfinite(M_arr)]
- # One-sided 95% lower bound ≈ 5th percentile
- lb = float(np.percentile(M_arr, 5.0))
- Mmin_db = max(1.0, lb)
-
- # Sigma: choose additive margin consistent with Mmin relative to typical L_ex
- L_ex_med = float(np.nanmedian(L_ex_arr)) if L_ex_arr.size else 0.0
- ratio = 10.0 ** (Mmin_db / 10.0)
- sigma = max(0.0, (ratio - 1.0) * L_ex_med)
-
- # Ω trials for epsilon and tau_max
- deltas: List[float] = []
- taus: List[float] = []
- for k in range(inp.omega_trials):
- print(f"\nΩ trial {k+1}/{inp.omega_trials}")
- d, tsec = run_power_sag_once(inp, seed_C=seed_C)
- if np.isfinite(d):
- deltas.append(float(d))
- if np.isfinite(tsec):
- taus.append(float(tsec))
- if not deltas:
- # fallback to conservative defaults
- eps_star = 0.15
- else:
- q90 = float(np.percentile(np.asarray(deltas), 90.0))
- eps_star = min(0.25, max(0.10, q90 + inp.safety_margin))
- if not taus:
- tau_star = 60.0
- else:
- t95 = float(np.percentile(np.asarray(taus), 95.0))
- tau_star = t95 + max(3.0 * inp.dt, 5.0)
-
- return CalibOutputs(Mmin_db=Mmin_db, epsilon=eps_star, tau_max=tau_star, sigma=sigma, profile_id=1)
-
-
-def write_profile_yaml(out_path: str, inp: CalibInputs, out: CalibOutputs) -> None:
- """Write a YAML profile for R* thresholds.
-
- Args:
- out_path: Destination path for the YAML profile.
- inp: Input configuration used for calibration.
- out: Calibrated thresholds.
- """
- data = {
- "profile_id": int(out.profile_id),
- "dt": float(inp.dt),
- "window_sec": float(inp.window_sec),
- "method": str(inp.method),
- "p_lag": int(inp.p_lag),
- "mi_lag": int(inp.mi_lag),
- "n_boot": int(inp.n_boot),
- "Mmin_db": float(out.Mmin_db),
- "epsilon": float(out.epsilon),
- "tau_max": float(out.tau_max),
- "sigma": float(out.sigma),
- "baseline_sec": float(max(10.0, inp.baseline_sec)),
- }
- with open(out_path, "w", encoding="utf-8") as f:
- yaml.safe_dump(data, f, sort_keys=False)
+ predicted = m_db(res.L_loop, res.L_ex)
+ L_ex_vals.append(float(res.L_ex))
+ return float(np.median(L_ex_vals)) if L_ex_vals else 0.0
+# --------------------------------------------------------------------------- #
+# Comparison artifacts
+# --------------------------------------------------------------------------- #
def _load_yaml(path: str) -> Mapping[str, Any]:
- """Load a YAML file as a mapping (or empty mapping on failure).
-
- Args:
- path: YAML file path.
-
- Returns:
- Mapping of keys to values; empty if missing/invalid.
- """
if not os.path.exists(path):
return {}
with open(path, "r", encoding="utf-8") as f:
try:
obj = yaml.safe_load(f) or {}
- if not isinstance(obj, dict):
- return {}
- return obj
+ return obj if isinstance(obj, dict) else {}
except Exception:
return {}
-def _write_compare_csv(out_csv: str, r0: Dict[str, float], rstar: CalibOutputs) -> None:
- """Write a CSV comparing R0 parameters with calibrated R*.
+def write_profile_yaml(out_path: str, base: Dict[str, Any], thr: Dict[str, float], baseline_sec: float) -> None:
+ """Write the calibrated R* profile, inheriting R0's measurement knobs."""
+ data = {
+ "profile_id": 1,
+ "dt": float(base.get("dt", 0.05)),
+ "window_sec": float(base.get("window_sec", 3.0)),
+ "method": str(base.get("method", "linear")),
+ "p_lag": int(base.get("p_lag", 3)),
+ "mi_lag": int(base.get("mi_lag", 1)),
+ "n_boot": int(base.get("n_boot", 32)),
+ "mi_k": int(base.get("mi_k", 5)),
+ "Mmin_db": float(thr["Mmin_db"]),
+ "epsilon": float(thr["epsilon"]),
+ "tau_max": float(thr["tau_max"]),
+ "sigma": float(thr["sigma"]),
+ "baseline_sec": float(baseline_sec),
+ "diag_cadence_windows": int(base.get("diag_cadence_windows", 25)),
+ "realtime": bool(base.get("realtime", False)),
+ }
+ with open(out_path, "w", encoding="utf-8") as f:
+ yaml.safe_dump(data, f, sort_keys=False)
+
+
+def write_compare_csv(out_csv: str, r0: Mapping[str, Any], thr: Dict[str, float]) -> None:
+ """Write a CSV comparing R0 parameters with calibrated R*."""
+ import csv
- Args:
- out_csv: Output CSV path.
- r0: Baseline R0 parameter mapping.
- rstar: Calibrated thresholds.
- """
os.makedirs(os.path.dirname(out_csv), exist_ok=True)
rows = [
- ("Mmin_db", r0.get("Mmin_db", float("nan")), rstar.Mmin_db),
- ("epsilon", r0.get("epsilon", float("nan")), rstar.epsilon),
- ("tau_max", r0.get("tau_max", float("nan")), rstar.tau_max),
- ("sigma", float("nan"), rstar.sigma),
+ ("Mmin_db", r0.get("Mmin_db", float("nan")), thr["Mmin_db"]),
+ ("epsilon", r0.get("epsilon", float("nan")), thr["epsilon"]),
+ ("tau_max", r0.get("tau_max", float("nan")), thr["tau_max"]),
+ ("sigma", float("nan"), thr["sigma"]),
]
- import csv
-
with open(out_csv, "w", newline="", encoding="utf-8") as f:
w = csv.writer(f)
w.writerow(["param", "R0", "R*"])
for name, r0v, rsv in rows:
- w.writerow([name, f"{r0v:.6g}" if np.isfinite(r0v) else "", f"{rsv:.6g}"])
+ try:
+ r0f = float(r0v)
+ except Exception:
+ r0f = float("nan")
+ w.writerow([name, f"{r0f:.6g}" if np.isfinite(r0f) else "", f"{rsv:.6g}"])
-def _write_compare_figure(out_png: str, r0: Dict[str, float], rstar: CalibOutputs) -> None:
- """Write a PNG bar chart comparing R0 vs R* parameters.
+def write_compare_figure(out_png: str, r0: Mapping[str, Any], thr: Dict[str, float]) -> None:
+ """Write a grouped bar chart comparing R0 vs R* thresholds."""
+ import matplotlib.pyplot as plt
+
+ from ldtc.reporting.style import COLORS, apply_matplotlib_theme
- Args:
- out_png: Output PNG path.
- r0: Baseline R0 parameter mapping.
- rstar: Calibrated thresholds.
- """
os.makedirs(os.path.dirname(out_png), exist_ok=True)
params = ["Mmin_db", "epsilon", "tau_max", "sigma"]
- r0_vals: List[float] = [
- float(r0.get("Mmin_db", np.nan)),
- float(r0.get("epsilon", np.nan)),
- float(r0.get("tau_max", np.nan)),
- np.nan,
- ]
- rstar_vals = [rstar.Mmin_db, rstar.epsilon, rstar.tau_max, rstar.sigma]
-
+ r0_vals = [float(r0.get(k, np.nan)) if r0.get(k) is not None else np.nan for k in params]
+ rstar_vals = [thr[k] for k in params]
x = np.arange(len(params))
width = 0.38
apply_matplotlib_theme("paper")
- plt.figure(figsize=(6.4, 3.2))
- # Plot R0; skip NaNs by replacing with zeros but masking in labels
+ fig, ax = plt.subplots(figsize=(6.4, 3.4))
r0_plot = [v if np.isfinite(v) else 0.0 for v in r0_vals]
- rstar_plot = rstar_vals
- plt.bar(x - width / 2, r0_plot, width=width, label="R0", color=COLORS["blue_light"])
- plt.bar(x + width / 2, rstar_plot, width=width, label="R*", color=COLORS["blue"])
- plt.xticks(x, params)
- plt.ylabel("Value")
- plt.title("R0 vs R* thresholds")
- plt.legend(frameon=False)
- plt.tight_layout()
- plt.savefig(out_png)
- plt.close()
+ ax.bar(x - width / 2, r0_plot, width=width, label="R0", color=COLORS["blue_light"])
+ ax.bar(x + width / 2, rstar_vals, width=width, label="R*", color=COLORS["blue"])
+ for xi, (a, b) in enumerate(zip(r0_vals, rstar_vals)):
+ if np.isfinite(a):
+ ax.text(xi - width / 2, a, f"{a:.2g}", ha="center", va="bottom", fontsize=8)
+ ax.text(xi + width / 2, b, f"{b:.2g}", ha="center", va="bottom", fontsize=8)
+ ax.set_xticks(x)
+ ax.set_xticklabels(params)
+ ax.set_ylabel("Value")
+ ax.set_title("R0 (generic) vs R* (calibrated) thresholds")
+ ax.legend(frameon=False)
+ fig.tight_layout()
+ for ext in ("png", "pdf", "svg"):
+ fig.savefig(os.path.splitext(out_png)[0] + "." + ext, dpi=300, bbox_inches="tight")
+ plt.close(fig)
+
+
+# --------------------------------------------------------------------------- #
+# Calibration
+# --------------------------------------------------------------------------- #
+def calibrate(args: argparse.Namespace) -> Dict[str, Any]:
+ """Run the calibration battery and return the threshold dict + provenance."""
+ os.environ["LDTC_SKIP_REPORT"] = "1"
+ base_cfg = dict(_load_yaml(os.path.join(REPO_ROOT, "configs", "profile_r0.yml")))
+ dt = float(base_cfg.get("dt", 0.05))
+
+ scen = {s.name: s for s in study.default_scenarios()}
+ pos = scen["positive"]
+ sag = scen["sc1_power_sag"]
+ flood = scen["sc1_ingress_flood"]
+
+ import tempfile
+ from dataclasses import replace as _replace
+
+ pooled_M: List[float] = []
+ deltas_sag: List[float] = []
+ deltas_flood: List[float] = []
+ taus: List[float] = []
+ with tempfile.TemporaryDirectory(prefix="ldtc_calib_") as tmp:
+ print(f"Baseline battery: {args.baseline_seeds} seeds")
+ for i in range(int(args.baseline_seeds)):
+ seed = int(args.seed_base) + i
+ rm = study.run_one(pos, seed, tmp)
+ if rm is None or not rm.valid:
+ print(f" baseline seed={seed}: skipped (invalid run)")
+ continue
+ ms = _pool_window_M(rm.run_dir)
+ pooled_M.extend(ms)
+ print(f" baseline seed={seed}: {len(ms)} windows, median M={rm.M_median:+.1f} dB")
+
+ if not pooled_M:
+ raise RuntimeError("Baseline battery produced no valid M samples")
+ M_arr = np.asarray([m for m in pooled_M if np.isfinite(m)], dtype=float)
+ Mmin_db = max(1.0, float(np.percentile(M_arr, 5.0)))
+
+ # Second pass: measure (delta, tau_rec) under the *calibrated* gate so
+ # epsilon and tau_max describe the decision rule R* will actually use.
+ # Both bounded batteries (sag + flood) contribute samples, so the
+ # calibrated depth/time bounds cover the bounded class itself, not one
+ # member of it.
+ sag_gated = _replace(sag, overrides={**sag.overrides, "Mmin_db": Mmin_db})
+ print(f"Power-sag battery: {args.sag_seeds} seeds (gate Mmin={Mmin_db:.2f} dB)")
+ for i in range(int(args.sag_seeds)):
+ seed = int(args.seed_base) + 100 + i
+ rm = study.run_one(sag_gated, seed, tmp)
+ if rm is None or not rm.valid:
+ print(f" power-sag seed={seed}: skipped (invalid run)")
+ continue
+ if rm.sc1_delta is not None:
+ deltas_sag.append(rm.sc1_delta)
+ if rm.sc1_tau_rec is not None:
+ taus.append(rm.sc1_tau_rec)
+ print(f" power-sag seed={seed}: delta={rm.sc1_delta}, tau_rec={rm.sc1_tau_rec}s")
+
+ flood_gated = _replace(flood, overrides={**flood.overrides, "Mmin_db": Mmin_db})
+ print(f"Ingress-flood battery: {args.flood_seeds} seeds (gate Mmin={Mmin_db:.2f} dB)")
+ for i in range(int(args.flood_seeds)):
+ seed = int(args.seed_base) + 200 + i
+ rm = study.run_one(flood_gated, seed, tmp)
+ if rm is None or not rm.valid:
+ print(f" ingress-flood seed={seed}: skipped (invalid run)")
+ continue
+ if rm.sc1_delta is not None:
+ deltas_flood.append(rm.sc1_delta)
+ if rm.sc1_tau_rec is not None:
+ taus.append(rm.sc1_tau_rec)
+ print(f" ingress-flood seed={seed}: delta={rm.sc1_delta}, tau_rec={rm.sc1_tau_rec}s")
+
+ deltas: List[float] = deltas_sag + deltas_flood
+
+ # epsilon is an upper tolerance bound on the bounded-class dip: the maximum
+ # observed calibration dip plus a safety margin. The cap (0.5) only rejects
+ # pathological near-total collapse: a 0.5 fractional L_loop drop is just
+ # ~3 dB of M, so the engaged loop is still overwhelmingly dominant; values
+ # in this range are genuinely resilient.
+ if deltas:
+ eps_star = min(0.50, max(0.10, float(np.max(np.asarray(deltas))) + float(args.safety_margin)))
+ else:
+ eps_star = 0.15
+ if taus:
+ tau_star = float(np.percentile(np.asarray(taus), 95.0)) + max(3.0 * dt, 5.0)
+ else:
+ tau_star = 60.0
+
+ print("Measuring typical baseline L_ex for sigma...")
+ # Under the engaged loop the controller drives exchange predictability below
+ # the L noise floor, so the raw median L_ex is ~0 (a strong NC1 signal). sigma
+ # is the additive-margin restatement of Mmin, so we evaluate it at the same
+ # floor m_db uses; we also record the raw value for transparency.
+ L_ex_floor = 1e-3 # matches the m_db() noise floor
+ L_ex_raw = measure_typical_L_ex(base_cfg)
+ L_ex_eff = max(L_ex_floor, L_ex_raw)
+ ratio = 10.0 ** (Mmin_db / 10.0)
+ sigma = max(0.0, (ratio - 1.0) * L_ex_eff)
+
+ thr = {"Mmin_db": Mmin_db, "epsilon": eps_star, "tau_max": tau_star, "sigma": sigma}
+ provenance = {
+ "n_baseline_windows": int(M_arr.size),
+ "baseline_M_p5": float(np.percentile(M_arr, 5.0)),
+ "baseline_M_median": float(np.median(M_arr)),
+ "n_sag_trials": len(deltas_sag),
+ "n_flood_trials": len(deltas_flood),
+ "n_bounded_trials": len(deltas),
+ "delta_max": (float(np.max(np.asarray(deltas))) if deltas else None),
+ "delta_max_sag": (float(np.max(np.asarray(deltas_sag))) if deltas_sag else None),
+ "delta_max_flood": (float(np.max(np.asarray(deltas_flood))) if deltas_flood else None),
+ "epsilon_rule": "max(bounded deltas) + safety_margin, floored at 0.10, capped at 0.50",
+ "tau_p95": (float(np.percentile(np.asarray(taus), 95.0)) if taus else None),
+ "tau_rec_from": "omega_offset",
+ "bounded_gate_Mmin_db": Mmin_db,
+ "L_ex_raw_median": L_ex_raw,
+ "L_ex_floor": L_ex_floor,
+ "L_ex_effective": L_ex_eff,
+ "base_profile": "configs/profile_r0.yml",
+ }
+ return {"thresholds": thr, "provenance": provenance, "base_cfg": base_cfg}
def main() -> None:
- """CLI entrypoint for R* calibration.
-
- Parses arguments, runs calibration, writes the profile and comparison
- artifacts, and prints summary paths.
- """
+ """CLI entry point for R* calibration."""
ap = argparse.ArgumentParser(
- description="Calibrate R* thresholds (Mmin, epsilon, tau_max, sigma) and write configs/profile_rstar.yml"
+ description="Calibrate R* thresholds from the validated harness and write configs/profile_rstar.yml"
)
- ap.add_argument("--dt", type=float, default=0.01)
- ap.add_argument("--window-sec", type=float, default=0.25)
- ap.add_argument("--method", type=str, default="linear", choices=["linear", "mi"])
- ap.add_argument("--p-lag", type=int, default=3)
- ap.add_argument("--mi-lag", type=int, default=1)
- ap.add_argument("--n-boot", type=int, default=32)
- ap.add_argument("--baseline-sec", type=float, default=15.0)
- ap.add_argument("--omega-trials", type=int, default=6)
- ap.add_argument("--sag-drop", type=float, default=0.3)
- ap.add_argument("--sag-duration", type=float, default=8.0)
- ap.add_argument("--safety-margin", type=float, default=0.02)
+ ap.add_argument("--baseline-seeds", type=int, default=6)
+ ap.add_argument("--sag-seeds", type=int, default=6)
+ ap.add_argument("--flood-seeds", type=int, default=6)
+ ap.add_argument("--seed-base", type=int, default=40000)
ap.add_argument(
- "--out",
- type=str,
- default=os.path.join(REPO_ROOT, "configs", "profile_rstar.yml"),
- )
- ap.add_argument(
- "--summary",
- type=str,
- default=os.path.join(REPO_ROOT, "artifacts", "calibration", "rstar_summary.json"),
- )
- ap.add_argument(
- "--compare-csv",
- type=str,
- default=os.path.join(REPO_ROOT, "artifacts", "calibration", "r0_vs_rstar.csv"),
- )
- ap.add_argument(
- "--compare-fig",
- type=str,
- default=os.path.join(REPO_ROOT, "artifacts", "calibration", "r0_vs_rstar.png"),
- )
- ap.add_argument(
- "--lock-profile",
- action="store_true",
- default=True,
- help="Make the written profile read-only (chmod 444)",
+ "--safety-margin",
+ type=float,
+ default=0.05,
+ help="additive margin on the max bounded-battery dip (absorbs the tolerance-bound tail)",
)
+ cal_dir = os.path.join(REPO_ROOT, "artifacts", "calibration")
+ ap.add_argument("--out", type=str, default=os.path.join(REPO_ROOT, "configs", "profile_rstar.yml"))
+ ap.add_argument("--summary", type=str, default=os.path.join(cal_dir, "rstar_summary.json"))
+ ap.add_argument("--compare-csv", type=str, default=os.path.join(cal_dir, "r0_vs_rstar.csv"))
+ ap.add_argument("--compare-fig", type=str, default=os.path.join(cal_dir, "r0_vs_rstar.png"))
+ ap.add_argument("--lock-profile", action="store_true", default=False, help="chmod 444 the written profile")
args = ap.parse_args()
os.makedirs(os.path.dirname(args.out), exist_ok=True)
os.makedirs(os.path.dirname(args.summary), exist_ok=True)
- inp = CalibInputs(
- dt=float(args.dt),
- window_sec=float(args.window_sec),
- method=str(args.method),
- p_lag=int(args.p_lag),
- mi_lag=int(args.mi_lag),
- n_boot=int(args.n_boot),
- baseline_sec=float(args.baseline_sec),
- omega_trials=int(args.omega_trials),
- sag_drop=float(args.sag_drop),
- sag_duration=float(args.sag_duration),
- safety_margin=float(args.safety_margin),
- )
-
- # Seed C matches the baseline CLI: internal states [E, T, R] -> 0,1,2
- seed_C = [0, 1, 2]
+ result = calibrate(args)
+ thr = result["thresholds"]
+ base_cfg = result["base_cfg"]
+ baseline_sec = float(base_cfg.get("baseline_sec", 18.0))
- out = calibrate_R_star(inp, seed_C=seed_C)
- write_profile_yaml(args.out, inp, out)
- # Optionally lock the profile file (read-only)
+ write_profile_yaml(args.out, base_cfg, thr, baseline_sec)
if args.lock_profile:
try:
os.chmod(args.out, 0o444)
except Exception:
pass
- # Compare against R0 and emit CSV/figure
- r0_path = os.path.join(REPO_ROOT, "configs", "profile_r0.yml")
- r0_loaded = _load_yaml(r0_path)
- # Filter numeric fields only to satisfy type expectations
- r0_numeric: Dict[str, float] = {}
- for k, v in r0_loaded.items() if hasattr(r0_loaded, "items") else []:
- try:
- r0_numeric[str(k)] = float(v)
- except Exception:
- continue
- _write_compare_csv(args.compare_csv, r0_numeric, out)
- _write_compare_figure(args.compare_fig, r0_numeric, out)
+ r0_loaded = _load_yaml(os.path.join(REPO_ROOT, "configs", "profile_r0.yml"))
+ write_compare_csv(args.compare_csv, r0_loaded, thr)
+ write_compare_figure(args.compare_fig, r0_loaded, thr)
summary = {
- "inputs": inp.__dict__,
- "outputs": out.__dict__,
+ "thresholds": thr,
+ "provenance": result["provenance"],
"timestamp": time.time(),
- "repo_root": REPO_ROOT,
- "note": "R* thresholds calibrated on synthetic baseline + Ω power-sag trials",
+ "note": "R* thresholds calibrated on the in-process plant via the production harness (R0 measurement knobs).",
"artifacts": {
"profile": os.path.abspath(args.out),
"compare_csv": os.path.abspath(args.compare_csv),
- "compare_fig": os.path.abspath(args.compare_fig),
+ "compare_fig": os.path.abspath(os.path.splitext(args.compare_fig)[0] + ".png"),
},
}
with open(args.summary, "w", encoding="utf-8") as f:
json.dump(summary, f, indent=2)
+ print("\n=== R* calibration ===")
+ print(f" Mmin_db = {thr['Mmin_db']:.2f} (R0: {r0_loaded.get('Mmin_db')})")
+ print(f" epsilon = {thr['epsilon']:.3f} (R0: {r0_loaded.get('epsilon')})")
+ print(f" tau_max = {thr['tau_max']:.2f} (R0: {r0_loaded.get('tau_max')})")
+ print(f" sigma = {thr['sigma']:.4f}")
print(f"Wrote calibrated profile: {args.out}")
- print(f"Wrote calibration summary: {args.summary}")
+ print(f"Wrote summary: {args.summary}")
if __name__ == "__main__":
diff --git a/scripts/emergence.py b/scripts/emergence.py
new file mode 100644
index 0000000..931bb9e
--- /dev/null
+++ b/scripts/emergence.py
@@ -0,0 +1,579 @@
+#!/usr/bin/env python3
+"""Scripts: Emergence-under-learning measurement sweep.
+
+Takes the policy checkpoints written by ``scripts/train_agent.py`` (fixed
+training fractions of the same run) and measures each one with the
+*production* verification harness (the ``run-policy`` CLI handler): same
+estimators, same guardrails, same audit chain as every other run in the
+paper, across ``N`` seeds per checkpoint. At the final checkpoint it also
+measures the two state-independent ablations (``shuffled`` and
+``frozen``), which preserve the trained policy's action statistics while
+severing the closed loop.
+
+Outputs (under ``--out``): ``emergence_results.json`` and ``.csv`` with
+per-condition aggregates and per-run rows, and
+``figures/fig_emergence.{png,pdf,svg}`` charting the training curve and
+median loop dominance ``M`` against training progress with the ablation
+endpoints.
+
+Run (after training):
+
+ python scripts/train_agent.py
+ python scripts/emergence.py --seeds 15 --rstar
+
+See Also:
+ paper/main.tex: Results (loop dominance emerges under learning).
+"""
+
+from __future__ import annotations
+
+import argparse
+import contextlib
+import io
+import json
+import os
+import sys
+import tempfile
+import time
+from dataclasses import asdict
+from typing import Any, Dict, List, Optional, Tuple
+
+import numpy as np
+import yaml
+
+REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+SCRIPTS_DIR = os.path.dirname(os.path.abspath(__file__))
+if SCRIPTS_DIR not in sys.path:
+ sys.path.insert(0, SCRIPTS_DIR)
+
+import study # noqa: E402 (study utilities: parsing, CIs, run-dir discovery)
+
+RUNS_DIR = study.RUNS_DIR
+
+# Shortened run length for the sweep (same override the study battery uses
+# for its NC1 scenarios, so the per-window geometry stays identical).
+DEFAULT_OVERRIDES: Dict[str, Any] = {"baseline_sec": 12.0, "diag_cadence_windows": 50}
+
+ABLATIONS: Tuple[str, ...] = ("shuffled", "frozen")
+
+
+# --------------------------------------------------------------------------- #
+# Conditions (checkpoints and ablations)
+# --------------------------------------------------------------------------- #
+def discover_checkpoints(ckpt_dir: str) -> List[Dict[str, Any]]:
+ """Find policy checkpoints and their training fractions.
+
+ Args:
+ ckpt_dir: Directory holding ``ckpt_*.json`` files written by
+ ``scripts/train_agent.py``.
+
+ Returns:
+ List of ``{"path", "frac", "generation"}`` dicts sorted by
+ training fraction.
+
+ Raises:
+ FileNotFoundError: If no checkpoints are found.
+ """
+ out: List[Dict[str, Any]] = []
+ if os.path.isdir(ckpt_dir):
+ for name in sorted(os.listdir(ckpt_dir)):
+ if not (name.startswith("ckpt_") and name.endswith(".json")):
+ continue
+ path = os.path.join(ckpt_dir, name)
+ with open(path, "r", encoding="utf-8") as f:
+ payload = json.load(f)
+ meta = payload.get("meta", {}) or {}
+ frac = float(meta.get("frac", int(name[5:8]) / 100.0))
+ out.append({"path": path, "frac": frac, "generation": int(meta.get("generation", -1))})
+ if not out:
+ raise FileNotFoundError(f"No policy checkpoints under {ckpt_dir} (run scripts/train_agent.py first)")
+ return sorted(out, key=lambda d: float(d["frac"]))
+
+
+def condition_name(frac: float, ablation: str = "none") -> str:
+ """Stable condition key for tables and figures."""
+ if ablation != "none":
+ return f"ablate_{ablation}"
+ return f"frac_{int(round(100 * frac)):03d}"
+
+
+# --------------------------------------------------------------------------- #
+# Run orchestration (in-process, production handler)
+# --------------------------------------------------------------------------- #
+def _apply_rstar(overrides: Dict[str, Any], profile_path: str) -> Dict[str, Any]:
+ """Merge calibrated decision thresholds into the run overrides."""
+ with open(profile_path, "r", encoding="utf-8") as f:
+ prof = dict(yaml.safe_load(f) or {})
+ thr = {k: prof[k] for k in ("Mmin_db", "epsilon", "tau_max") if k in prof}
+ if not thr:
+ raise ValueError(f"{profile_path} has no threshold keys")
+ out = dict(overrides)
+ out.update(thr)
+ return out
+
+
+def run_one_policy(
+ base_cfg: Dict[str, Any],
+ overrides: Dict[str, Any],
+ ckpt_path: str,
+ ablation: str,
+ frac: float,
+ seed: int,
+ tmpdir: str,
+ verbose: bool = False,
+) -> Optional[study.RunMetrics]:
+ """Measure one checkpoint (or ablation) at one seed via ``run-policy``.
+
+ Args:
+ base_cfg: Loaded base profile dict.
+ overrides: Profile overrides (run length, thresholds).
+ ckpt_path: Policy checkpoint path.
+ ablation: ``"none"``, ``"shuffled"``, or ``"frozen"``.
+ frac: Training fraction (for the scenario label).
+ seed: Seed for this replicate.
+ tmpdir: Directory for the per-run temporary config.
+ verbose: If True, let the handler print to stdout.
+
+ Returns:
+ Parsed :class:`study.RunMetrics`, or ``None`` if no run directory
+ was found.
+ """
+ from ldtc.cli import main as cli
+
+ name = condition_name(frac, ablation)
+ cfg = dict(base_cfg)
+ cfg.update(overrides)
+ cfg["seed"] = int(seed)
+ cfg["seed_py"] = int(seed)
+ cfg["seed_np"] = int(seed)
+ cfg_path = os.path.join(tmpdir, f"cfg_{name}_seed_{seed}.yml")
+ with open(cfg_path, "w", encoding="utf-8") as f:
+ yaml.safe_dump(cfg, f, sort_keys=False)
+
+ ns = argparse.Namespace(config=cfg_path, policy=ckpt_path, ablation=ablation)
+ before = {d for d in os.listdir(RUNS_DIR)} if os.path.isdir(RUNS_DIR) else set()
+ sink = io.StringIO()
+ ctx = contextlib.nullcontext() if verbose else contextlib.redirect_stdout(sink)
+ with ctx:
+ cli.run_policy(ns)
+ run_dir = study._find_run_dir("policy", before, cfg_path)
+ if run_dir is None:
+ return None
+ return study.parse_run(name, seed, run_dir)
+
+
+# --------------------------------------------------------------------------- #
+# Aggregation and writers
+# --------------------------------------------------------------------------- #
+def aggregate_condition(
+ name: str,
+ label: str,
+ frac: Optional[float],
+ ablation: str,
+ runs: List[study.RunMetrics],
+) -> Dict[str, Any]:
+ """Aggregate per-seed runs of one condition into a summary row."""
+ n = len(runs)
+ per_seed_M = [r.M_median for r in runs if r.M_median == r.M_median]
+ nc1_k = sum(1 for r in runs if r.nc1_pass)
+ valid_k = sum(1 for r in runs if r.valid)
+ reasons: Dict[str, int] = {}
+ for r in runs:
+ for reason in set(r.invalidations):
+ reasons[reason] = reasons.get(reason, 0) + 1
+ return {
+ "name": name,
+ "label": label,
+ "frac": frac,
+ "ablation": ablation,
+ "n_seeds": n,
+ "valid_rate": (valid_k / n) if n else float("nan"),
+ "valid_ci": study.wilson_ci(valid_k, n),
+ "M_mean": (float(np.mean(per_seed_M)) if per_seed_M else float("nan")),
+ "M_ci": study.bootstrap_ci(per_seed_M),
+ "M_median_overall": (float(np.median(per_seed_M)) if per_seed_M else float("nan")),
+ "M_per_seed": per_seed_M,
+ "nc1_pass_rate": (nc1_k / n) if n else float("nan"),
+ "nc1_ci": study.wilson_ci(nc1_k, n),
+ "nc1_window_frac_mean": (float(np.mean([r.nc1_window_frac for r in runs])) if runs else float("nan")),
+ "invalidation_reasons": reasons,
+ }
+
+
+def write_csv(aggs: List[Dict[str, Any]], path: str) -> None:
+ """Write the per-condition aggregates as a flat CSV."""
+ import csv
+
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ cols = [
+ "condition",
+ "label",
+ "frac",
+ "ablation",
+ "n_seeds",
+ "valid_rate",
+ "M_mean_db",
+ "M_lo",
+ "M_hi",
+ "M_median_db",
+ "nc1_pass_rate",
+ "nc1_lo",
+ "nc1_hi",
+ "nc1_window_frac_mean",
+ "invalidations",
+ ]
+ with open(path, "w", newline="", encoding="utf-8") as f:
+ w = csv.writer(f)
+ w.writerow(cols)
+ for a in aggs:
+ w.writerow(
+ [
+ a["name"],
+ a["label"],
+ ("" if a["frac"] is None else f"{a['frac']:.2f}"),
+ a["ablation"],
+ a["n_seeds"],
+ f"{a['valid_rate']:.4f}",
+ f"{a['M_mean']:.4f}",
+ f"{a['M_ci'][0]:.4f}",
+ f"{a['M_ci'][1]:.4f}",
+ f"{a['M_median_overall']:.4f}",
+ f"{a['nc1_pass_rate']:.4f}",
+ f"{a['nc1_ci'][0]:.4f}",
+ f"{a['nc1_ci'][1]:.4f}",
+ f"{a['nc1_window_frac_mean']:.4f}",
+ ";".join(f"{k}={v}" for k, v in a["invalidation_reasons"].items()),
+ ]
+ )
+
+
+# --------------------------------------------------------------------------- #
+# Figure
+# --------------------------------------------------------------------------- #
+def fig_emergence(data: Dict[str, Any], out_dir: str, stem: str = "fig_emergence") -> Optional[str]:
+ """Render the emergence figure: training curve and M versus training.
+
+ Panel (a) is the training reward curve (mean episode reward of the ES
+ center per generation) with the measured checkpoints marked. Panel (b)
+ is the per-seed median loop dominance ``M`` at each checkpoint
+ (box plus per-seed points, like the NC1 contrast figure), with the two
+ state-independent ablations of the final policy at the right and the
+ ``Mmin`` and 0 dB reference lines.
+
+ Args:
+ data: Loaded ``emergence_results.json`` payload.
+ out_dir: Output directory for the figure files.
+ stem: Output file stem.
+
+ Returns:
+ Path to the written PNG, or ``None`` if there is nothing to plot.
+ """
+ import matplotlib.pyplot as plt
+
+ from ldtc.reporting.style import COLORS, apply_matplotlib_theme
+
+ aggs: List[Dict[str, Any]] = list(data.get("aggregates", []))
+ runs: Dict[str, List[Dict[str, Any]]] = data.get("runs", {})
+ ckpt_aggs = sorted((a for a in aggs if a["ablation"] == "none"), key=lambda a: float(a["frac"]))
+ abl_aggs = [a for a in aggs if a["ablation"] != "none"]
+ if not ckpt_aggs:
+ return None
+
+ mmin = 3.0
+ for rows in runs.values():
+ if rows:
+ mmin = float(rows[0].get("Mmin_db", mmin))
+ break
+
+ history = (data.get("meta", {}).get("training_log", {}) or {}).get("history", [])
+ ckpt_gens = {
+ int(round(100 * float(a["frac"]))): int(a.get("generation", -1))
+ for a in ckpt_aggs
+ if a.get("generation") is not None
+ }
+
+ apply_matplotlib_theme("paper")
+ fig, (ax_a, ax_b) = plt.subplots(
+ 1,
+ 2,
+ figsize=(10.2, 4.0),
+ gridspec_kw={"width_ratios": [1.0, 1.5]},
+ )
+
+ # Panel (a): training curve.
+ if history:
+ gens = [h["gen"] for h in history]
+ fit = [h["fitness"] for h in history]
+ ax_a.plot(gens, fit, color=COLORS["blue"], linewidth=1.8, zorder=3, label="mean episode reward")
+ marked = False
+ for g in sorted(set(ckpt_gens.values())):
+ if g < 0:
+ continue
+ ax_a.axvline(g, color=COLORS["gray"], linestyle=":", linewidth=1.0, zorder=1)
+ if not marked:
+ ax_a.axvline(g, color=COLORS["gray"], linestyle=":", linewidth=1.0, zorder=1, label="checkpoint")
+ marked = True
+ ax_a.set_xlabel("Training generation")
+ ax_a.set_ylabel("Mean episode reward")
+ ax_a.set_title("(a) Survival training (ES)")
+ ax_a.legend(loc="lower right", frameon=False, fontsize=8)
+
+ # Panel (b): M versus training progress, with ablations.
+ order = [a["name"] for a in ckpt_aggs] + [a["name"] for a in abl_aggs]
+ labels = [a["label"] for a in ckpt_aggs] + [a["label"].replace("ablate: ", "ablate:\n") for a in abl_aggs]
+ rng = np.random.default_rng(7)
+ for i, a in enumerate(ckpt_aggs + abl_aggs):
+ ys = [r["M_median"] for r in runs.get(a["name"], []) if r["M_median"] == r["M_median"]]
+ if not ys:
+ continue
+ xs = i + rng.uniform(-0.12, 0.12, size=len(ys))
+ passing = float(a.get("nc1_pass_rate", 0.0)) >= 0.5
+ color = COLORS["green"] if passing else COLORS["red"]
+ ax_b.scatter(xs, ys, s=30, color=color, alpha=0.75, edgecolor="white", linewidth=0.5, zorder=3)
+ bp = ax_b.boxplot(
+ ys,
+ positions=[i],
+ widths=0.5,
+ vert=True,
+ patch_artist=True,
+ showfliers=False,
+ zorder=2,
+ )
+ for box in bp["boxes"]:
+ box.set(facecolor=COLORS["gray_light"], edgecolor=COLORS["gray"], alpha=0.7)
+ for med in bp["medians"]:
+ med.set(color=COLORS["gray"], linewidth=2)
+ if abl_aggs:
+ ax_b.axvline(len(ckpt_aggs) - 0.5, color=COLORS["gray"], linestyle="-", linewidth=0.8, zorder=1)
+ ax_b.axhline(0.0, color=COLORS["gray"], linestyle="-", linewidth=1.0, zorder=1)
+ ax_b.axhline(mmin, color=COLORS["blue"], linestyle="--", linewidth=1.5, zorder=1)
+ ax_b.text(
+ len(order) - 0.5,
+ mmin,
+ f" $M_{{\\min}}$ = {mmin:.1f} dB",
+ color=COLORS["blue"],
+ va="bottom",
+ ha="right",
+ fontsize=9,
+ )
+ ax_b.set_xticks(range(len(order)))
+ ax_b.set_xticklabels(labels, fontsize=8)
+ ax_b.set_xlabel("Training progress (fraction of generations)")
+ ax_b.set_ylabel(r"Loop dominance $M$ (dB)")
+ n = data.get("meta", {}).get("n_seeds", 0)
+ ax_b.set_title(f"(b) Measured loop dominance (N={n} seeds)")
+
+ fig.suptitle("Loop dominance emerges under learned self-maintenance", fontsize=12)
+ fig.tight_layout(rect=(0, 0, 1, 0.95))
+
+ os.makedirs(out_dir, exist_ok=True)
+ base = os.path.join(out_dir, stem)
+ fig.savefig(base + ".png", dpi=300, bbox_inches="tight")
+ fig.savefig(base + ".pdf", bbox_inches="tight")
+ fig.savefig(base + ".svg", bbox_inches="tight")
+ plt.close(fig)
+ return base + ".png"
+
+
+def make_figure(out_dir: str) -> Optional[str]:
+ """Regenerate the emergence figure from an existing results payload.
+
+ Args:
+ out_dir: Directory containing ``emergence_results.json``.
+
+ Returns:
+ Path to the written PNG, or ``None``.
+ """
+ path = os.path.join(out_dir, "emergence_results.json")
+ with open(path, "r", encoding="utf-8") as f:
+ data = json.load(f)
+ return fig_emergence(data, os.path.join(out_dir, "figures"))
+
+
+# --------------------------------------------------------------------------- #
+# Driver
+# --------------------------------------------------------------------------- #
+def run_sweep(
+ config: str,
+ ckpt_dir: str,
+ out_dir: str,
+ seeds: List[int],
+ ablations: Tuple[str, ...] = ABLATIONS,
+ threshold_profile: Optional[str] = None,
+ verbose: bool = False,
+) -> Dict[str, Any]:
+ """Measure every checkpoint (and the final-policy ablations) across seeds.
+
+ Args:
+ config: Base profile path (the emergence plant).
+ ckpt_dir: Directory of policy checkpoints.
+ out_dir: Output directory for results and figures.
+ seeds: Seeds to use as replicates.
+ ablations: Ablation modes to run at the final checkpoint.
+ threshold_profile: Optional calibrated profile whose ``Mmin_db``,
+ ``epsilon``, and ``tau_max`` override the base config (R*).
+ verbose: If True, let handlers print.
+
+ Returns:
+ The full results payload (also written to
+ ``emergence_results.json``).
+ """
+ os.environ["LDTC_SKIP_REPORT"] = "1" # skip per-run figure bundles
+ os.makedirs(RUNS_DIR, exist_ok=True)
+ ckpts = discover_checkpoints(ckpt_dir)
+ with open(config, "r", encoding="utf-8") as f:
+ base_cfg = dict(yaml.safe_load(f) or {})
+ overrides = dict(DEFAULT_OVERRIDES)
+ if threshold_profile:
+ overrides = _apply_rstar(overrides, threshold_profile)
+
+ # Conditions: every checkpoint closed-loop, then ablations of the final.
+ conditions: List[Dict[str, Any]] = []
+ for ck in ckpts:
+ conditions.append({**ck, "ablation": "none"})
+ for ab in ablations:
+ conditions.append({**ckpts[-1], "ablation": ab})
+
+ runs_by_cond: Dict[str, List[study.RunMetrics]] = {}
+ t0 = time.time()
+ with tempfile.TemporaryDirectory(prefix="ldtc_emergence_") as tmpdir:
+ for cond in conditions:
+ frac = float(cond["frac"])
+ ablation = str(cond["ablation"])
+ name = condition_name(frac, ablation)
+ rows: List[study.RunMetrics] = []
+ for seed in seeds:
+ ts = time.time()
+ rm = run_one_policy(
+ base_cfg,
+ overrides,
+ str(cond["path"]),
+ ablation,
+ frac,
+ seed,
+ tmpdir,
+ verbose=verbose,
+ )
+ dt = time.time() - ts
+ if rm is None:
+ print(f" [{name}] seed={seed}: NO RUN DIR FOUND", flush=True)
+ continue
+ rows.append(rm)
+ print(
+ f" [{name}] seed={seed} ({dt:.1f}s): valid={rm.valid} NC1={rm.nc1_pass} "
+ f"M~{rm.M_median:+.1f} (windows {100 * rm.nc1_window_frac:.0f}% cert)",
+ flush=True,
+ )
+ runs_by_cond[name] = rows
+ print(f"== {name}: {len(rows)}/{len(seeds)} runs ==", flush=True)
+
+ aggs: List[Dict[str, Any]] = []
+ for cond in conditions:
+ frac = float(cond["frac"])
+ ablation = str(cond["ablation"])
+ name = condition_name(frac, ablation)
+ if ablation != "none":
+ label = f"ablate: {ablation}"
+ agg_frac: Optional[float] = None
+ else:
+ label = f"{int(round(100 * frac))}%"
+ agg_frac = frac
+ agg = aggregate_condition(name, label, agg_frac, ablation, runs_by_cond.get(name, []))
+ agg["generation"] = int(cond.get("generation", -1))
+ aggs.append(agg)
+
+ training_log: Dict[str, Any] = {}
+ tl_path = os.path.join(os.path.dirname(os.path.abspath(ckpt_dir)), "training_log.json")
+ if os.path.exists(tl_path):
+ with open(tl_path, "r", encoding="utf-8") as f:
+ training_log = json.load(f)
+
+ thr_label = os.path.relpath(threshold_profile or config, REPO_ROOT)
+ payload: Dict[str, Any] = {
+ "meta": {
+ "seeds": seeds,
+ "n_seeds": len(seeds),
+ "config": os.path.relpath(config, REPO_ROOT),
+ "threshold_profile": thr_label,
+ "ckpt_dir": os.path.relpath(ckpt_dir, REPO_ROOT),
+ "overrides": overrides,
+ "elapsed_sec": round(time.time() - t0, 1),
+ "timestamp": time.time(),
+ "training_log": training_log,
+ },
+ "aggregates": aggs,
+ "runs": {name: [asdict(r) for r in rs] for name, rs in runs_by_cond.items()},
+ }
+ os.makedirs(out_dir, exist_ok=True)
+ with open(os.path.join(out_dir, "emergence_results.json"), "w", encoding="utf-8") as f:
+ json.dump(payload, f, indent=2)
+ write_csv(aggs, os.path.join(out_dir, "emergence_results.csv"))
+ return payload
+
+
+def print_summary(payload: Dict[str, Any]) -> None:
+ """Print a compact human-readable summary to stdout."""
+ print("\n=== EMERGENCE SUMMARY ===")
+ for a in payload["aggregates"]:
+ print(
+ f"{a['label']:>16s} valid={100 * a['valid_rate']:3.0f}% NC1={100 * a['nc1_pass_rate']:3.0f}% "
+ f"M={a['M_mean']:+6.1f} dB [{a['M_ci'][0]:+.1f},{a['M_ci'][1]:+.1f}] "
+ f"(windows {100 * a['nc1_window_frac_mean']:3.0f}% cert)"
+ )
+
+
+def main() -> None:
+ """CLI entry point for the emergence measurement sweep."""
+ ap = argparse.ArgumentParser(description="Measure policy checkpoints with the production harness.")
+ ap.add_argument("--config", type=str, default=os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"))
+ ap.add_argument("--ckpt-dir", type=str, default=os.path.join(REPO_ROOT, "artifacts", "emergence", "checkpoints"))
+ ap.add_argument("--out", type=str, default=os.path.join(REPO_ROOT, "artifacts", "emergence"))
+ ap.add_argument("--seeds", type=int, default=15, help="Number of seeds (replicates) per condition.")
+ ap.add_argument("--seed-base", type=int, default=3000, help="First seed; seeds are base..base+N-1.")
+ ap.add_argument(
+ "--rstar",
+ nargs="?",
+ const=os.path.join(REPO_ROOT, "configs", "profile_rstar.yml"),
+ default="",
+ help="Evaluate against calibrated R* thresholds from this profile "
+ "(default configs/profile_rstar.yml). Requires running calibrate first.",
+ )
+ ap.add_argument("--no-ablations", action="store_true", help="Skip the final-checkpoint ablation runs.")
+ ap.add_argument("--no-figures", action="store_true", help="Skip figure generation.")
+ ap.add_argument("--verbose", action="store_true")
+ args = ap.parse_args()
+
+ if args.rstar and not os.path.exists(args.rstar):
+ ap.error(f"--rstar profile not found: {args.rstar} (run `make calibrate` first)")
+
+ seeds = [args.seed_base + i for i in range(int(args.seeds))]
+ ablations: Tuple[str, ...] = () if args.no_ablations else ABLATIONS
+ if args.rstar:
+ print(f"Evaluating against calibrated thresholds from {args.rstar}", flush=True)
+ print(f"Emergence sweep: checkpoints in {args.ckpt_dir} x {len(seeds)} seeds -> {args.out}", flush=True)
+
+ payload = run_sweep(
+ config=args.config,
+ ckpt_dir=args.ckpt_dir,
+ out_dir=args.out,
+ seeds=seeds,
+ ablations=ablations,
+ threshold_profile=(args.rstar or None),
+ verbose=args.verbose,
+ )
+ print_summary(payload)
+
+ if not args.no_figures:
+ try:
+ p = fig_emergence(payload, os.path.join(args.out, "figures"))
+ if p:
+ print(f" figure: {p}")
+ except Exception as exc: # pragma: no cover - figures are best-effort
+ print(f"(figure skipped: {exc})")
+
+ print(f"\nWrote: {os.path.join(args.out, 'emergence_results.json')}")
+ print(f"Wrote: {os.path.join(args.out, 'emergence_results.csv')}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/sensitivity.py b/scripts/sensitivity.py
new file mode 100644
index 0000000..e6718f8
--- /dev/null
+++ b/scripts/sensitivity.py
@@ -0,0 +1,301 @@
+#!/usr/bin/env python3
+"""Scripts: Sensitivity sweeps for the NC1 loop-dominance result.
+
+Shows that the headline contrast (positive control ``M`` well above ``Mmin``;
+controller-disabled negative control ``M`` below 0) is robust to the main
+measurement and model choices. For each swept setting the script runs the
+positive control and the controller-disabled negative control across several
+seeds (via the production harness) and reports the seed-mean median ``M`` with
+a bootstrap CI.
+
+Axes swept:
+
+* ``p_lag``: VAR lag order of the linear estimator.
+* ``window_sec``: measurement window length.
+* ``method``: estimator family (linear VAR-Granger vs. mutual information).
+* ``coupling``: a multiplicative scale on the plant's internal self-maintenance
+ coupling (``c_TE``, ``c_RT``, ``c_RE``).
+
+Outputs ``sensitivity_results.{json,csv,tex}`` and ``fig_sensitivity.{png,pdf,svg}``.
+
+Run:
+
+ python scripts/sensitivity.py --seeds 4 --out artifacts/sensitivity
+
+See Also:
+ paper/main.tex: Results: Sensitivity analysis.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import sys
+import tempfile
+import time
+from dataclasses import dataclass, field
+from typing import Any, Dict, List
+
+import numpy as np
+
+REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+
+import study # noqa: E402
+
+# Plant coupling defaults (see ldtc/plant/models.py PlantParams).
+BASE_COUPLING = {"c_TE": 0.90, "c_RT": 0.54, "c_RE": 0.66}
+
+# Speed overrides applied to every sweep run (the CI comes from across seeds,
+# so each run can be shorter than a headline study run).
+SPEED = {"baseline_sec": 8.0, "diag_cadence_windows": 100, "n_boot": 24}
+
+
+@dataclass
+class Setting:
+ """One point in a sensitivity sweep."""
+
+ axis: str
+ label: str
+ overrides: Dict[str, Any] = field(default_factory=dict)
+
+
+def build_settings() -> List[Setting]:
+ """Construct the full list of sweep settings."""
+ settings: List[Setting] = []
+ for p in (2, 3, 4):
+ settings.append(Setting("p_lag", str(p), {"p_lag": p, **SPEED}))
+ for w in (2.0, 3.0, 4.0):
+ settings.append(Setting("window_sec", f"{w:.0f}s", {"window_sec": w, **SPEED}))
+ for m in ("linear", "mi"):
+ # MI is heavier per window; keep its bootstrap modest.
+ extra = {"n_boot": 12} if m == "mi" else {}
+ settings.append(Setting("method", m, {"method": m, **SPEED, **extra}))
+ for s in (0.7, 1.0, 1.3):
+ coup = {k: round(v * s, 4) for k, v in BASE_COUPLING.items()}
+ settings.append(Setting("coupling", f"x{s:.1f}", {"plant": {"params": coup}, **SPEED}))
+ return settings
+
+
+def _scn_with(base: "study.Scenario", overrides: Dict[str, Any]) -> "study.Scenario":
+ merged = dict(base.overrides)
+ merged.update(overrides)
+ return study.Scenario(
+ name=base.name,
+ label=base.label,
+ kind=base.kind,
+ expectation=base.expectation,
+ handler=base.handler,
+ config=base.config,
+ run_tag=base.run_tag,
+ omega_args=dict(base.omega_args),
+ overrides=merged,
+ )
+
+
+def run_sweeps(seeds: List[int], out_dir: str) -> Dict[str, Any]:
+ """Run every sweep setting for the positive and negative controls.
+
+ Args:
+ seeds: Replicate seeds.
+ out_dir: Output directory for results.
+
+ Returns:
+ The results payload (also written to ``sensitivity_results.json``).
+ """
+ os.environ["LDTC_SKIP_REPORT"] = "1"
+ scen = {s.name: s for s in study.default_scenarios()}
+ pos = scen["positive"]
+ neg = scen["neg_controller_disabled"]
+ settings = build_settings()
+
+ rows: List[Dict[str, Any]] = []
+ t0 = time.time()
+ with tempfile.TemporaryDirectory(prefix="ldtc_sens_") as tmp:
+ for st in settings:
+ row: Dict[str, Any] = {"axis": st.axis, "label": st.label}
+ for tag, base in (("pos", pos), ("neg", neg)):
+ scn = _scn_with(base, st.overrides)
+ meds: List[float] = []
+ npass = 0
+ nvalid = 0
+ for seed in seeds:
+ rm = study.run_one(scn, seed, tmp)
+ if rm is None:
+ continue
+ if rm.valid:
+ nvalid += 1
+ if rm.M_median == rm.M_median:
+ meds.append(rm.M_median)
+ if rm.nc1_pass:
+ npass += 1
+ mean = float(np.mean(meds)) if meds else float("nan")
+ lo, hi = study.bootstrap_ci(meds)
+ row[f"{tag}_M_mean"] = mean
+ row[f"{tag}_M_lo"] = lo
+ row[f"{tag}_M_hi"] = hi
+ row[f"{tag}_nc1_rate"] = npass / len(seeds) if seeds else float("nan")
+ row[f"{tag}_valid_rate"] = nvalid / len(seeds) if seeds else float("nan")
+ rows.append(row)
+ print(
+ f" [{st.axis}={st.label}] pos M={row['pos_M_mean']:+.1f} "
+ f"[{row['pos_M_lo']:+.1f},{row['pos_M_hi']:+.1f}] "
+ f"neg M={row['neg_M_mean']:+.1f} [{row['neg_M_lo']:+.1f},{row['neg_M_hi']:+.1f}]",
+ flush=True,
+ )
+
+ payload = {
+ "meta": {"seeds": seeds, "n_seeds": len(seeds), "elapsed_sec": round(time.time() - t0, 1)},
+ "rows": rows,
+ }
+ os.makedirs(out_dir, exist_ok=True)
+ with open(os.path.join(out_dir, "sensitivity_results.json"), "w", encoding="utf-8") as f:
+ json.dump(payload, f, indent=2)
+ _write_csv(rows, os.path.join(out_dir, "sensitivity_results.csv"))
+ _write_latex(rows, os.path.join(out_dir, "sensitivity_results.tex"), len(seeds))
+ return payload
+
+
+def _write_csv(rows: List[Dict[str, Any]], path: str) -> None:
+ import csv
+
+ cols = [
+ "axis",
+ "label",
+ "pos_M_mean",
+ "pos_M_lo",
+ "pos_M_hi",
+ "pos_nc1_rate",
+ "pos_valid_rate",
+ "neg_M_mean",
+ "neg_M_lo",
+ "neg_M_hi",
+ "neg_nc1_rate",
+ "neg_valid_rate",
+ ]
+ with open(path, "w", newline="", encoding="utf-8") as f:
+ w = csv.writer(f)
+ w.writerow(cols)
+ for r in rows:
+ w.writerow([r.get(c, "") for c in cols])
+
+
+def _write_latex(rows: List[Dict[str, Any]], path: str, n_seeds: int) -> None:
+ lines = [
+ "% Auto-generated by scripts/sensitivity.py -- do not edit by hand.",
+ "\\begin{tabular}{llcc}",
+ "\\toprule",
+ "Axis & Setting & Positive $M$ (dB) & Loop-ablated $M$ (dB) \\\\",
+ "\\midrule",
+ ]
+ # Display labels keep the table free of raw underscores (LaTeX text mode).
+ axis_label = {
+ "p_lag": "VAR lag $p$",
+ "window_sec": "Window",
+ "method": "Estimator",
+ "coupling": "Coupling scale",
+ }
+ last_axis = None
+ for r in rows:
+ axis = axis_label.get(r["axis"], r["axis"]) if r["axis"] != last_axis else ""
+ last_axis = r["axis"]
+ pos = f"{r['pos_M_mean']:+.1f} [{r['pos_M_lo']:+.1f}, {r['pos_M_hi']:+.1f}]"
+ neg = f"{r['neg_M_mean']:+.1f} [{r['neg_M_lo']:+.1f}, {r['neg_M_hi']:+.1f}]"
+ lines.append(f"{axis} & {r['label']} & {pos} & {neg} \\\\")
+ lines += [
+ "\\bottomrule",
+ "\\end{tabular}",
+ f"% N = {n_seeds} seeds per cell; brackets are 95% bootstrap CIs on the mean of per-seed median M.",
+ ]
+ with open(path, "w", encoding="utf-8") as f:
+ f.write("\n".join(lines) + "\n")
+
+
+def make_figure(payload: Dict[str, Any], out_dir: str) -> str:
+ """Render a 2x2 robustness figure (one panel per swept axis)."""
+ import matplotlib.pyplot as plt
+
+ from ldtc.reporting.style import COLORS, apply_matplotlib_theme
+
+ rows = payload["rows"]
+ axes_order = ["p_lag", "window_sec", "method", "coupling"]
+ titles = {
+ "p_lag": "VAR lag $p$",
+ "window_sec": "Window length",
+ "method": "Estimator",
+ "coupling": "Internal coupling scale",
+ }
+ # The robustness claim is the sign of the contrast: positive control above
+ # the loop/exchange dominance boundary (M=0), controller-disabled below it.
+ # We deliberately do not draw a single Mmin line here because the calibrated
+ # threshold is estimator-specific (the MI and linear scales differ), so one
+ # line would be misleading across the estimator panel.
+ apply_matplotlib_theme("paper")
+ fig, axs = plt.subplots(2, 2, figsize=(9.0, 6.4))
+ for ax, axis in zip(axs.ravel(), axes_order):
+ sub = [r for r in rows if r["axis"] == axis]
+ if not sub:
+ ax.set_visible(False)
+ continue
+ x = np.arange(len(sub))
+ labels = [r["label"] for r in sub]
+ pos = np.array([r["pos_M_mean"] for r in sub])
+ pos_lo = np.array([r["pos_M_mean"] - r["pos_M_lo"] for r in sub])
+ pos_hi = np.array([r["pos_M_hi"] - r["pos_M_mean"] for r in sub])
+ neg = np.array([r["neg_M_mean"] for r in sub])
+ neg_lo = np.array([r["neg_M_mean"] - r["neg_M_lo"] for r in sub])
+ neg_hi = np.array([r["neg_M_hi"] - r["neg_M_mean"] for r in sub])
+ ax.errorbar(
+ x - 0.06, pos, yerr=[pos_lo, pos_hi], fmt="o-", color=COLORS["green"], capsize=4, label="positive", zorder=3
+ )
+ ax.errorbar(
+ x + 0.06,
+ neg,
+ yerr=[neg_lo, neg_hi],
+ fmt="s--",
+ color=COLORS["red"],
+ capsize=4,
+ label="loop ablated",
+ zorder=3,
+ )
+ ax.axhline(
+ 0.0, color=COLORS["gray"], linestyle="-", linewidth=1.0, zorder=1, label="dominance boundary ($M=0$)"
+ )
+ ax.set_xticks(x)
+ ax.set_xticklabels(labels)
+ ax.set_title(titles.get(axis, axis))
+ ax.set_ylabel(r"$M$ (dB)")
+ axs.ravel()[0].legend(frameon=False, fontsize=8, loc="center right")
+ fig.suptitle("NC1 loop dominance is robust to estimator and model choices", fontsize=12)
+ fig.tight_layout(rect=(0, 0, 1, 0.96))
+ base = os.path.join(out_dir, "fig_sensitivity")
+ for ext in ("png", "pdf", "svg"):
+ fig.savefig(base + "." + ext, dpi=300, bbox_inches="tight")
+ plt.close(fig)
+ return base + ".png"
+
+
+def main() -> None:
+ """CLI entry point for the sensitivity sweeps."""
+ ap = argparse.ArgumentParser(description="Run NC1 sensitivity sweeps and emit table + figure.")
+ ap.add_argument("--seeds", type=int, default=4)
+ ap.add_argument("--seed-base", type=int, default=60000)
+ ap.add_argument("--out", type=str, default=os.path.join(REPO_ROOT, "artifacts", "sensitivity"))
+ ap.add_argument("--no-figure", action="store_true")
+ args = ap.parse_args()
+
+ seeds = [args.seed_base + i for i in range(int(args.seeds))]
+ print(f"Sensitivity sweeps: {len(build_settings())} settings x {len(seeds)} seeds", flush=True)
+ payload = run_sweeps(seeds, args.out)
+ if not args.no_figure:
+ try:
+ p = make_figure(payload, args.out)
+ print(f" figure: {p}")
+ except Exception as exc: # pragma: no cover
+ print(f"(figure skipped: {exc})")
+ print(f"Wrote: {os.path.join(args.out, 'sensitivity_results.json')}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/study.py b/scripts/study.py
new file mode 100644
index 0000000..cad87cb
--- /dev/null
+++ b/scripts/study.py
@@ -0,0 +1,929 @@
+#!/usr/bin/env python3
+"""Scripts: Multi-seed LDTC simulation study.
+
+Runs the positive control, the negative controls, the SC1 perturbation
+battery, and the command-conflict refusal scenario across ``N`` seeds, using
+the *production* CLI handlers (so the study exercises exactly the code paths a
+verifier would). Each run's hash-chained audit log is parsed for its outcome,
+and the per-seed outcomes are aggregated with bootstrap and Wilson confidence
+intervals into machine-readable (JSON/CSV) and paper-ready (LaTeX) tables plus
+summary figures.
+
+The seed is the unit of replication: continuous quantities (median ``M``) are
+summarized by the mean of per-seed medians with a bootstrap CI, and binary
+outcomes (run validity, NC1 / SC1 pass, refusal) by a proportion with a Wilson
+score CI.
+
+Run:
+
+ python scripts/study.py --seeds 12 --out artifacts/study
+
+See Also:
+ paper/main.tex: Results.
+"""
+
+from __future__ import annotations
+
+import argparse
+import contextlib
+import io
+import json
+import math
+import os
+import statistics
+import sys
+import tempfile
+import time
+from dataclasses import asdict, dataclass, field, replace
+from typing import Any, Callable, Dict, List, Optional, Tuple
+
+import numpy as np
+import yaml
+
+REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+RUNS_DIR = os.path.join(REPO_ROOT, "artifacts", "runs")
+
+
+# --------------------------------------------------------------------------- #
+# Statistics helpers
+# --------------------------------------------------------------------------- #
+def wilson_ci(k: int, n: int, z: float = 1.96) -> Tuple[float, float]:
+ """Wilson score interval for a binomial proportion.
+
+ Args:
+ k: Number of successes.
+ n: Number of trials.
+ z: Normal quantile (1.96 for ~95%).
+
+ Returns:
+ ``(lo, hi)`` bounds on the success probability in ``[0, 1]``.
+ """
+ if n <= 0:
+ return (0.0, 0.0)
+ p = k / n
+ denom = 1.0 + z * z / n
+ center = (p + z * z / (2.0 * n)) / denom
+ half = (z * math.sqrt(p * (1.0 - p) / n + z * z / (4.0 * n * n))) / denom
+ return (max(0.0, center - half), min(1.0, center + half))
+
+
+def bootstrap_ci(values: List[float], n_boot: int = 2000, seed: int = 0) -> Tuple[float, float]:
+ """Percentile bootstrap CI for the mean of ``values``.
+
+ Args:
+ values: Sample values (NaNs are dropped).
+ n_boot: Number of bootstrap resamples.
+ seed: RNG seed for reproducibility.
+
+ Returns:
+ ``(lo, hi)`` 95% percentile CI on the mean, or ``(nan, nan)`` if empty.
+ """
+ arr = np.asarray([v for v in values if v == v], dtype=float)
+ if arr.size == 0:
+ return (float("nan"), float("nan"))
+ if arr.size == 1:
+ return (float(arr[0]), float(arr[0]))
+ rng = np.random.default_rng(seed)
+ boot = rng.choice(arr, size=(n_boot, arr.size), replace=True).mean(axis=1)
+ lo, hi = np.percentile(boot, [2.5, 97.5])
+ return (float(lo), float(hi))
+
+
+def _median(values: List[float]) -> float:
+ vals = [v for v in values if v == v]
+ return float(statistics.median(vals)) if vals else float("nan")
+
+
+# --------------------------------------------------------------------------- #
+# Audit parsing
+# --------------------------------------------------------------------------- #
+@dataclass
+class RunMetrics:
+ """Outcome of a single run, parsed from its audit log."""
+
+ scenario: str
+ seed: int
+ run_dir: str
+ valid: bool
+ invalidations: List[str]
+ Mmin_db: float
+ n_windows: int
+ M_median: float
+ M_min: float
+ M_max: float
+ nc1_window_frac: float
+ nc1_pass: bool
+ sc1_evaluated: bool
+ sc1_pass: Optional[bool]
+ sc1_delta: Optional[float]
+ sc1_tau_rec: Optional[float]
+ sc1_M_post: Optional[float]
+ refusal_evaluated: bool
+ refused: Optional[bool]
+ refusal_pass: Optional[bool]
+ refusal_reasons: List[str]
+ trefuse_ms: Optional[float]
+
+
+def parse_run(scenario: str, seed: int, run_dir: str) -> RunMetrics:
+ """Parse a run's audit log into a :class:`RunMetrics`.
+
+ Args:
+ scenario: Scenario name.
+ seed: Seed used for the run.
+ run_dir: Path to the per-run artifact directory.
+
+ Returns:
+ Parsed run metrics.
+ """
+ audit_path = os.path.join(run_dir, "audits", "audit.jsonl")
+ Mmin = 3.0
+ ms: List[float] = []
+ nc1_flags: List[bool] = []
+ invalid: List[str] = []
+ sc1: Optional[dict] = None
+ refusal: Optional[dict] = None
+ with open(audit_path, "r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ e = json.loads(line)
+ ev = e.get("event")
+ det = e.get("details", {}) or {}
+ if ev == "run_header":
+ Mmin = float(det.get("Mmin_db", Mmin))
+ elif ev == "window_measured":
+ if det.get("M") is not None:
+ ms.append(float(det["M"]))
+ nc1_flags.append(bool(det.get("nc1")))
+ elif ev == "run_invalidated":
+ invalid.append(str(det.get("reason", "?")))
+ elif ev == "sc1_result":
+ sc1 = det
+ elif ev == "command_refusal_result":
+ refusal = det
+
+ valid = len(invalid) == 0
+ m_med = _median(ms)
+ nc1_frac = (sum(1 for f in nc1_flags if f) / len(nc1_flags)) if nc1_flags else 0.0
+ # NC1 is decided by the production harness per window (margin vs Mmin
+ # plus the loop-influence noise gate); the seed passes when a valid run
+ # certified the majority of its windows. Recomputing from M_median alone
+ # would silently drop the gate.
+ nc1_pass = bool(valid and nc1_flags and nc1_frac >= 0.5)
+
+ return RunMetrics(
+ scenario=scenario,
+ seed=seed,
+ run_dir=run_dir,
+ valid=valid,
+ invalidations=invalid,
+ Mmin_db=Mmin,
+ n_windows=len(ms),
+ M_median=m_med,
+ M_min=float(min(ms)) if ms else float("nan"),
+ M_max=float(max(ms)) if ms else float("nan"),
+ nc1_window_frac=nc1_frac,
+ nc1_pass=nc1_pass,
+ sc1_evaluated=sc1 is not None,
+ sc1_pass=(bool(sc1.get("pass")) if sc1 else None),
+ sc1_delta=(float(sc1["delta"]) if sc1 and sc1.get("delta") is not None else None),
+ sc1_tau_rec=(float(sc1["tau_rec"]) if sc1 and sc1.get("tau_rec") is not None else None),
+ sc1_M_post=(float(sc1["M_post"]) if sc1 and sc1.get("M_post") is not None else None),
+ refusal_evaluated=refusal is not None,
+ refused=(bool(refusal.get("refused")) if refusal else None),
+ refusal_pass=(bool(refusal.get("pass")) if refusal else None),
+ refusal_reasons=(list(refusal.get("reasons", [])) if refusal else []),
+ trefuse_ms=(
+ float(refusal["trefuse_ms_max"]) if refusal and refusal.get("trefuse_ms_max") is not None else None
+ ),
+ )
+
+
+# --------------------------------------------------------------------------- #
+# Scenario definitions
+# --------------------------------------------------------------------------- #
+@dataclass
+class Scenario:
+ """A study scenario: which handler, config, and Ω args to run.
+
+ Attributes:
+ name: Stable scenario key used in tables and figures.
+ label: Human-readable label.
+ kind: One of ``"nc1"``, ``"sc1"``, ``"refusal"`` (selects the headline
+ metric for reporting).
+ expectation: Short text describing the expected outcome.
+ handler: CLI handler name in :mod:`ldtc.cli.main`.
+ config: Base YAML profile path (relative to repo root).
+ run_tag: Directory-name prefix the handler uses under artifacts/runs.
+ omega_args: Extra argparse fields for the handler.
+ overrides: Profile overrides applied on top of the base config (used to
+ shorten runs for the study without changing the validated dt/window).
+ """
+
+ name: str
+ label: str
+ kind: str
+ expectation: str
+ handler: str
+ config: str
+ run_tag: str
+ omega_args: Dict[str, Any] = field(default_factory=dict)
+ overrides: Dict[str, Any] = field(default_factory=dict)
+
+
+def default_scenarios() -> List[Scenario]:
+ """Return the standard study battery."""
+ nc1_over = {"baseline_sec": 12.0, "diag_cadence_windows": 50}
+ sc1_over = {"baseline_sec": 8.0, "diag_cadence_windows": 50}
+ return [
+ Scenario(
+ name="positive",
+ label="Positive control",
+ kind="nc1",
+ expectation="NC1 holds (M above Mmin)",
+ handler="run_baseline",
+ config="configs/profile_r0.yml",
+ run_tag="baseline",
+ overrides=nc1_over,
+ ),
+ Scenario(
+ name="neg_controller_disabled",
+ label="Negative: loop ablated",
+ kind="nc1",
+ expectation="NC1 fails (M<0), run valid",
+ handler="run_baseline",
+ config="configs/profile_negative_controller_disabled.yml",
+ run_tag="baseline",
+ overrides=nc1_over,
+ ),
+ Scenario(
+ name="neg_permanent_ex_flood",
+ label="Negative: sustained ex-flood (unshielded)",
+ kind="nc1",
+ expectation="NC1 fails (M<0); no SC1 recovery",
+ handler="omega_ingress_flood",
+ config="configs/profile_negative_permanent_ex_flood.yml",
+ run_tag="omega-ingress-flood",
+ omega_args={"mult": 5.0, "duration": 6.0},
+ overrides={"baseline_sec": 10.0, "recovery_observe_sec": 4.0, "diag_cadence_windows": 50},
+ ),
+ Scenario(
+ name="neg_exogenous_subsidy",
+ label="Negative: exogenous subsidy",
+ kind="nc1",
+ expectation="Run invalidated (red flag)",
+ handler="omega_exogenous_subsidy",
+ config="configs/profile_negative_exogenous_soc.yml",
+ run_tag="omega-exogenous-subsidy",
+ omega_args={"delta": 0.2, "zero_harvest": True, "duration": 8.0},
+ overrides={"baseline_sec": 6.0, "diag_cadence_windows": 50},
+ ),
+ Scenario(
+ name="sc1_power_sag",
+ label="SC1: power sag",
+ kind="sc1",
+ expectation="SC1 holds (recovers)",
+ handler="omega_power_sag",
+ config="configs/profile_r0.yml",
+ run_tag="omega-power-sag",
+ omega_args={"drop": 0.3, "duration": 8.0},
+ overrides=sc1_over,
+ ),
+ Scenario(
+ name="sc1_ingress_flood",
+ label="SC1: ingress flood",
+ kind="sc1",
+ expectation="SC1 holds (recovers)",
+ handler="omega_ingress_flood",
+ config="configs/profile_r0.yml",
+ run_tag="omega-ingress-flood",
+ omega_args={"mult": 5.0, "duration": 8.0},
+ overrides=sc1_over,
+ ),
+ Scenario(
+ name="sc1_control_outage",
+ label="SC1 designed fail: control outage",
+ kind="sc1",
+ expectation="SC1 fails (depth bound exceeded)",
+ handler="omega_control_outage",
+ config="configs/profile_r0.yml",
+ run_tag="omega-control-outage",
+ omega_args={"duration": 6.0},
+ overrides={"baseline_sec": 8.0, "recovery_observe_sec": 10.0, "diag_cadence_windows": 50},
+ ),
+ Scenario(
+ name="refusal_command_conflict",
+ label="Threat: command conflict",
+ kind="refusal",
+ expectation="Refuse at low SoC (<= target latency)",
+ handler="omega_command_conflict",
+ config="configs/profile_negative_command_conflict.yml",
+ run_tag="omega-command-conflict",
+ omega_args={"observe": 2.0},
+ ),
+ # Adversarial gaming battery: systems engineered to *look* loop-dominant
+ # without being so. The designed outcome for all three is
+ # non-certification: NC1 fails on a valid run, or a guardrail
+ # invalidates the run (nc1_pass is false either way).
+ Scenario(
+ name="adv_replay_controller",
+ label="Adversarial: replayed actuation",
+ kind="nc1",
+ expectation="Not certified (NC1 fails, run valid)",
+ handler="adv_replay_controller",
+ config="configs/profile_adv_replay_controller.yml",
+ run_tag="adv-replay-controller",
+ overrides=nc1_over,
+ ),
+ Scenario(
+ name="adv_hidden_tether",
+ label="Adversarial: hidden tether",
+ kind="nc1",
+ expectation="Not certified (loop collapses onto Ex)",
+ handler="adv_hidden_tether",
+ config="configs/profile_adv_hidden_tether.yml",
+ run_tag="adv-hidden-tether",
+ omega_args={"dither": 0.10},
+ overrides=nc1_over,
+ ),
+ Scenario(
+ name="adv_oscillator",
+ label="Adversarial: oscillator inflation",
+ kind="nc1",
+ expectation="Not certified (M low or smell test)",
+ handler="adv_oscillator",
+ config="configs/profile_adv_oscillator.yml",
+ run_tag="adv-oscillator",
+ omega_args={"amp": 0.10, "period": 1.0},
+ overrides=nc1_over,
+ ),
+ ]
+
+
+# --------------------------------------------------------------------------- #
+# Run orchestration (in-process, production handlers)
+# --------------------------------------------------------------------------- #
+def _load_yaml(path: str) -> Dict[str, Any]:
+ with open(path, "r", encoding="utf-8") as f:
+ return dict(yaml.safe_load(f) or {})
+
+
+# Threshold keys that R* calibration sets; everything else (dt, window, method,
+# p_lag, n_boot) is shared with R0 so the two studies stay directly comparable.
+_THRESHOLD_KEYS = ("Mmin_db", "epsilon", "tau_max")
+
+
+def apply_threshold_profile(scenarios: List[Scenario], profile_path: str) -> List[Scenario]:
+ """Inject calibrated decision thresholds into every scenario's overrides.
+
+ The study evaluates NC1/SC1 with the same handler code a verifier runs; the
+ only thing that should differ between an "R0" (uncalibrated guess) study and
+ the headline "R\\*" study is the decision thresholds (``Mmin_db``,
+ ``epsilon``, ``tau_max``). This reads those three keys from ``profile_path``
+ and merges them into each scenario's ``overrides`` so the whole battery is
+ judged against one consistent, plant-calibrated threshold set.
+
+ Calibration uses a disjoint seed range (see ``calibrate_rstar.py``), so
+ evaluating the study seeds against these thresholds is a train/test split,
+ not a circular fit.
+
+ Args:
+ scenarios: Scenarios to update (copied; inputs are not mutated).
+ profile_path: Path to the calibrated profile YAML (``profile_rstar.yml``).
+
+ Returns:
+ New scenarios with R* thresholds merged into ``overrides``.
+ """
+ prof = _load_yaml(profile_path)
+ thr = {k: prof[k] for k in _THRESHOLD_KEYS if k in prof}
+ if not thr:
+ raise ValueError(f"{profile_path} has none of {_THRESHOLD_KEYS}")
+ out: List[Scenario] = []
+ for s in scenarios:
+ ov = dict(s.overrides)
+ ov.update(thr)
+ out.append(replace(s, overrides=ov))
+ return out
+
+
+def _write_seed_config(base_cfg: Dict[str, Any], seed: int, overrides: Dict[str, Any], tmpdir: str) -> str:
+ cfg = dict(base_cfg)
+ cfg.update(overrides)
+ cfg["seed"] = int(seed)
+ cfg["seed_py"] = int(seed)
+ cfg["seed_np"] = int(seed)
+ path = os.path.join(tmpdir, f"cfg_seed_{seed}.yml")
+ with open(path, "w", encoding="utf-8") as f:
+ yaml.safe_dump(cfg, f, sort_keys=False)
+ return path
+
+
+def _handlers() -> Dict[str, Callable[[argparse.Namespace], None]]:
+ from ldtc.cli import main as cli
+
+ return {
+ "run_baseline": cli.run_baseline,
+ "omega_power_sag": cli.omega_power_sag,
+ "omega_ingress_flood": cli.omega_ingress_flood,
+ "omega_control_outage": cli.omega_control_outage,
+ "omega_exogenous_subsidy": cli.omega_exogenous_subsidy,
+ "omega_command_conflict": cli.omega_command_conflict,
+ "adv_replay_controller": cli.adv_replay_controller,
+ "adv_hidden_tether": cli.adv_hidden_tether,
+ "adv_oscillator": cli.adv_oscillator,
+ }
+
+
+def _namespace_for(scn: Scenario, config_path: str) -> argparse.Namespace:
+ ns = argparse.Namespace(config=config_path)
+ for k, v in scn.omega_args.items():
+ setattr(ns, k, v)
+ return ns
+
+
+def _run_header_config(run_dir: str) -> Optional[str]:
+ """Return the ``config_path`` recorded in a run's audit header, if any."""
+ audit_path = os.path.join(run_dir, "audits", "audit.jsonl")
+ if not os.path.exists(audit_path):
+ return None
+ try:
+ with open(audit_path, "r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ e = json.loads(line)
+ if e.get("event") == "run_header":
+ return e.get("details", {}).get("config_path")
+ except Exception:
+ return None
+ return None
+
+
+def _find_run_dir(prefix: str, before: set, cfg_path: str) -> Optional[str]:
+ """Locate the run directory just created for ``cfg_path``.
+
+ Disambiguates by the unique per-seed config path recorded in the audit
+ header, so the harness stays correct even when other runs (e.g., a
+ concurrent study) are creating directories at the same time. Falls back to
+ the newest directory with the matching tag prefix.
+
+ Args:
+ prefix: Expected run-tag prefix (e.g., ``"baseline"``).
+ before: Set of directory names present before the handler ran.
+ cfg_path: The per-seed config path passed to the handler.
+
+ Returns:
+ Absolute path to the matching run directory, or ``None``.
+ """
+ if not os.path.isdir(RUNS_DIR):
+ return None
+ new = sorted(set(os.listdir(RUNS_DIR)) - before)
+ if not new:
+ return None
+ target = os.path.abspath(cfg_path)
+ for name in new:
+ rd = os.path.join(RUNS_DIR, name)
+ hdr = _run_header_config(rd)
+ if hdr is not None and os.path.abspath(hdr) == target:
+ return rd
+ # Fallback: newest with the matching tag prefix.
+ pref = sorted(n for n in new if n.startswith(prefix))
+ return os.path.join(RUNS_DIR, pref[-1]) if pref else None
+
+
+def run_one(scn: Scenario, seed: int, tmpdir: str, verbose: bool = False) -> Optional[RunMetrics]:
+ """Run one scenario at one seed via the production handler and parse it.
+
+ Args:
+ scn: Scenario specification.
+ seed: Seed for this replicate.
+ tmpdir: Directory for the per-seed temporary config.
+ verbose: If True, let the handler print to stdout.
+
+ Returns:
+ Parsed :class:`RunMetrics`, or ``None`` if no run directory was found.
+ """
+ handlers = _handlers()
+ base_cfg = _load_yaml(os.path.join(REPO_ROOT, scn.config))
+ cfg_path = _write_seed_config(base_cfg, seed, scn.overrides, tmpdir)
+ ns = _namespace_for(scn, cfg_path)
+
+ before = {d for d in os.listdir(RUNS_DIR)} if os.path.isdir(RUNS_DIR) else set()
+ sink = io.StringIO()
+ ctx = contextlib.nullcontext() if verbose else contextlib.redirect_stdout(sink)
+ with ctx:
+ handlers[scn.handler](ns)
+ run_dir = _find_run_dir(scn.run_tag, before, cfg_path)
+ if run_dir is None:
+ return None
+ return parse_run(scn.name, seed, run_dir)
+
+
+# --------------------------------------------------------------------------- #
+# Aggregation
+# --------------------------------------------------------------------------- #
+@dataclass
+class Aggregate:
+ """Per-scenario aggregate over seeds."""
+
+ name: str
+ label: str
+ kind: str
+ expectation: str
+ n_seeds: int
+ valid_rate: float
+ valid_ci: Tuple[float, float]
+ M_mean: float
+ M_ci: Tuple[float, float]
+ M_median_overall: float
+ nc1_pass_rate: float
+ nc1_ci: Tuple[float, float]
+ sc1_n: int
+ sc1_pass_rate: Optional[float]
+ sc1_ci: Optional[Tuple[float, float]]
+ sc1_delta_median: Optional[float]
+ sc1_tau_median: Optional[float]
+ refusal_n: int
+ refusal_rate: Optional[float]
+ refusal_ci: Optional[Tuple[float, float]]
+ trefuse_median_ms: Optional[float]
+ invalidation_reasons: Dict[str, int]
+
+
+def aggregate(scn: Scenario, runs: List[RunMetrics]) -> Aggregate:
+ """Aggregate per-seed runs into a scenario-level summary.
+
+ Args:
+ scn: Scenario specification.
+ runs: Per-seed parsed run metrics.
+
+ Returns:
+ Scenario-level :class:`Aggregate`.
+ """
+ n = len(runs)
+ valid_k = sum(1 for r in runs if r.valid)
+ per_seed_M = [r.M_median for r in runs if r.M_median == r.M_median]
+ nc1_k = sum(1 for r in runs if r.nc1_pass)
+
+ sc1_runs = [r for r in runs if r.sc1_evaluated]
+ sc1_k = sum(1 for r in sc1_runs if r.sc1_pass)
+ sc1_deltas = [r.sc1_delta for r in sc1_runs if r.sc1_delta is not None]
+ sc1_taus = [r.sc1_tau_rec for r in sc1_runs if r.sc1_tau_rec is not None]
+
+ ref_runs = [r for r in runs if r.refusal_evaluated]
+ ref_k = sum(1 for r in ref_runs if r.refusal_pass)
+ ref_lat = [r.trefuse_ms for r in ref_runs if r.trefuse_ms is not None]
+
+ # Count seeds (not windows) that tripped each invalidation reason.
+ reasons: Dict[str, int] = {}
+ for r in runs:
+ for reason in set(r.invalidations):
+ reasons[reason] = reasons.get(reason, 0) + 1
+
+ return Aggregate(
+ name=scn.name,
+ label=scn.label,
+ kind=scn.kind,
+ expectation=scn.expectation,
+ n_seeds=n,
+ valid_rate=valid_k / n if n else float("nan"),
+ valid_ci=wilson_ci(valid_k, n),
+ M_mean=(float(np.mean(per_seed_M)) if per_seed_M else float("nan")),
+ M_ci=bootstrap_ci(per_seed_M),
+ M_median_overall=_median(per_seed_M),
+ nc1_pass_rate=nc1_k / n if n else float("nan"),
+ nc1_ci=wilson_ci(nc1_k, n),
+ sc1_n=len(sc1_runs),
+ sc1_pass_rate=(sc1_k / len(sc1_runs) if sc1_runs else None),
+ sc1_ci=(wilson_ci(sc1_k, len(sc1_runs)) if sc1_runs else None),
+ sc1_delta_median=(_median(sc1_deltas) if sc1_deltas else None),
+ sc1_tau_median=(_median(sc1_taus) if sc1_taus else None),
+ refusal_n=len(ref_runs),
+ refusal_rate=(ref_k / len(ref_runs) if ref_runs else None),
+ refusal_ci=(wilson_ci(ref_k, len(ref_runs)) if ref_runs else None),
+ trefuse_median_ms=(_median(ref_lat) if ref_lat else None),
+ invalidation_reasons=reasons,
+ )
+
+
+# --------------------------------------------------------------------------- #
+# Writers
+# --------------------------------------------------------------------------- #
+def _fmt_pct(p: float, ci: Tuple[float, float]) -> str:
+ if p != p:
+ return "--"
+ return f"{100 * p:.0f}\\% [{100 * ci[0]:.0f}, {100 * ci[1]:.0f}]"
+
+
+def _fmt_m(mean: float, ci: Tuple[float, float]) -> str:
+ if mean != mean:
+ return "--"
+ return f"{mean:+.1f} [{ci[0]:+.1f}, {ci[1]:+.1f}]"
+
+
+def write_csv(aggs: List[Aggregate], path: str) -> None:
+ """Write the aggregate results as a flat CSV."""
+ import csv
+
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ cols = [
+ "scenario",
+ "label",
+ "kind",
+ "expectation",
+ "n_seeds",
+ "valid_rate",
+ "valid_lo",
+ "valid_hi",
+ "M_mean_db",
+ "M_lo",
+ "M_hi",
+ "M_median_db",
+ "nc1_pass_rate",
+ "nc1_lo",
+ "nc1_hi",
+ "sc1_n",
+ "sc1_pass_rate",
+ "sc1_delta_median",
+ "sc1_tau_median",
+ "refusal_n",
+ "refusal_rate",
+ "trefuse_median_ms",
+ "invalidations",
+ ]
+ with open(path, "w", newline="", encoding="utf-8") as f:
+ w = csv.writer(f)
+ w.writerow(cols)
+ for a in aggs:
+ w.writerow(
+ [
+ a.name,
+ a.label,
+ a.kind,
+ a.expectation,
+ a.n_seeds,
+ f"{a.valid_rate:.4f}",
+ f"{a.valid_ci[0]:.4f}",
+ f"{a.valid_ci[1]:.4f}",
+ f"{a.M_mean:.4f}",
+ f"{a.M_ci[0]:.4f}",
+ f"{a.M_ci[1]:.4f}",
+ f"{a.M_median_overall:.4f}",
+ f"{a.nc1_pass_rate:.4f}",
+ f"{a.nc1_ci[0]:.4f}",
+ f"{a.nc1_ci[1]:.4f}",
+ a.sc1_n,
+ ("" if a.sc1_pass_rate is None else f"{a.sc1_pass_rate:.4f}"),
+ ("" if a.sc1_delta_median is None else f"{a.sc1_delta_median:.4f}"),
+ ("" if a.sc1_tau_median is None else f"{a.sc1_tau_median:.4f}"),
+ a.refusal_n,
+ ("" if a.refusal_rate is None else f"{a.refusal_rate:.4f}"),
+ ("" if a.trefuse_median_ms is None else f"{a.trefuse_median_ms:.4f}"),
+ ";".join(f"{k}={v}" for k, v in a.invalidation_reasons.items()),
+ ]
+ )
+
+
+def write_latex(aggs: List[Aggregate], path: str, n_seeds: int) -> None:
+ """Write a booktabs LaTeX results table for the paper."""
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ lines = [
+ "% Auto-generated by scripts/study.py -- do not edit by hand.",
+ "\\begin{tabular}{llcccc}",
+ "\\toprule",
+ "Scenario & Expected & Valid & NC1 pass & Median $M$ (dB) & SC1/Refusal \\\\",
+ "\\midrule",
+ ]
+ for a in aggs:
+ if a.kind == "sc1":
+ last = "--" if a.sc1_pass_rate is None else _fmt_pct(a.sc1_pass_rate, a.sc1_ci or (0, 0))
+ elif a.kind == "refusal":
+ last = "--" if a.refusal_rate is None else _fmt_pct(a.refusal_rate, a.refusal_ci or (0, 0))
+ else:
+ last = "--"
+ lines.append(
+ f"{a.label} & {a.expectation} & {_fmt_pct(a.valid_rate, a.valid_ci)} & "
+ f"{_fmt_pct(a.nc1_pass_rate, a.nc1_ci)} & {_fmt_m(a.M_mean, a.M_ci)} & {last} \\\\"
+ )
+ lines += [
+ "\\bottomrule",
+ "\\end{tabular}",
+ f"% N = {n_seeds} seeds per scenario; brackets are 95% CIs (Wilson for proportions, bootstrap for M).",
+ ]
+ with open(path, "w", encoding="utf-8") as f:
+ f.write("\n".join(lines) + "\n")
+
+
+def write_json(
+ aggs: List[Aggregate],
+ runs_by_scn: Dict[str, List[RunMetrics]],
+ meta: Dict[str, Any],
+ path: str,
+) -> None:
+ """Write the full study results (aggregates + per-run rows + metadata)."""
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ payload = {
+ "meta": meta,
+ "aggregates": [asdict(a) for a in aggs],
+ "runs": {name: [asdict(r) for r in rs] for name, rs in runs_by_scn.items()},
+ }
+ with open(path, "w", encoding="utf-8") as f:
+ json.dump(payload, f, indent=2)
+
+
+# --------------------------------------------------------------------------- #
+# Driver
+# --------------------------------------------------------------------------- #
+def run_study(
+ seeds: List[int],
+ scenarios: List[Scenario],
+ out_dir: str,
+ verbose: bool = False,
+ threshold_profile: Optional[str] = None,
+) -> Tuple[List[Aggregate], Dict[str, List[RunMetrics]]]:
+ """Run the full study and write all artifacts.
+
+ Args:
+ seeds: Seeds to use as replicates.
+ scenarios: Scenarios to run.
+ out_dir: Output directory for study artifacts.
+ verbose: If True, let handlers print.
+
+ Returns:
+ ``(aggregates, runs_by_scenario)``.
+ """
+ os.environ["LDTC_SKIP_REPORT"] = "1" # skip per-run figure bundles
+ os.makedirs(RUNS_DIR, exist_ok=True)
+ runs_by_scn: Dict[str, List[RunMetrics]] = {}
+ t0 = time.time()
+ with tempfile.TemporaryDirectory(prefix="ldtc_study_") as tmpdir:
+ for scn in scenarios:
+ rows: List[RunMetrics] = []
+ for seed in seeds:
+ ts = time.time()
+ rm = run_one(scn, seed, tmpdir, verbose=verbose)
+ dt = time.time() - ts
+ if rm is None:
+ print(f" [{scn.name}] seed={seed}: NO RUN DIR FOUND", flush=True)
+ continue
+ rows.append(rm)
+ tag = (
+ f"valid={rm.valid} NC1={rm.nc1_pass} M~{rm.M_median:+.1f}"
+ + (f" SC1={rm.sc1_pass}" if rm.sc1_evaluated else "")
+ + (f" refused={rm.refused}({rm.trefuse_ms}ms)" if rm.refusal_evaluated else "")
+ )
+ print(f" [{scn.name}] seed={seed} ({dt:.1f}s): {tag}", flush=True)
+ runs_by_scn[scn.name] = rows
+ print(f"== {scn.name}: {len(rows)}/{len(seeds)} runs ==", flush=True)
+
+ aggs = [aggregate(scn, runs_by_scn.get(scn.name, [])) for scn in scenarios]
+ thr_label = (
+ os.path.relpath(threshold_profile, REPO_ROOT) if threshold_profile else "configs/profile_r0.yml (per-scenario)"
+ )
+ meta = {
+ "seeds": seeds,
+ "n_seeds": len(seeds),
+ "elapsed_sec": round(time.time() - t0, 1),
+ "timestamp": time.time(),
+ "threshold_profile": thr_label,
+ "scenarios": [asdict(s) for s in scenarios],
+ }
+ write_json(aggs, runs_by_scn, meta, os.path.join(out_dir, "study_results.json"))
+ write_csv(aggs, os.path.join(out_dir, "study_results.csv"))
+ write_latex(aggs, os.path.join(out_dir, "study_results.tex"), len(seeds))
+ return aggs, runs_by_scn
+
+
+def load_or_run(study_dir: str, seeds: List[int], scenario_names: Optional[List[str]] = None) -> Dict[str, Any]:
+ """Load an existing study payload or run a (subset) study to create one.
+
+ Used by the paper figure generators so they always render *real* data: if
+ the canonical study has been run (``study_dir/study_results.json`` exists)
+ its results are reused; otherwise a small study over ``scenario_names`` is
+ run into ``study_dir`` first. Never returns synthetic data.
+
+ Args:
+ study_dir: Directory holding (or to hold) ``study_results.json``.
+ seeds: Seeds to use if a study must be run.
+ scenario_names: Optional subset of scenario names to run.
+
+ Returns:
+ The loaded study payload dict.
+ """
+ path = os.path.join(study_dir, "study_results.json")
+ if os.path.exists(path):
+ with open(path, "r", encoding="utf-8") as f:
+ return json.load(f)
+ scns = default_scenarios()
+ if scenario_names is not None:
+ keep = set(scenario_names)
+ scns = [s for s in scns if s.name in keep]
+ run_study(seeds, scns, study_dir)
+ with open(path, "r", encoding="utf-8") as f:
+ return json.load(f)
+
+
+def data_for_paper(
+ canonical_dir: str,
+ fallback_dir: str,
+ seeds: List[int],
+ scenario_names: List[str],
+) -> Dict[str, Any]:
+ """Return study data for a paper figure, preferring the canonical study.
+
+ If the full study (``canonical_dir/study_results.json``) exists it is used
+ (e.g., the 15-seed submission run); otherwise a small study over
+ ``scenario_names`` is run into ``fallback_dir`` so the figure is still based
+ on real data (used by CI paper builds that have no prior study).
+
+ Args:
+ canonical_dir: Directory of the full study.
+ fallback_dir: Per-figure directory for a small on-demand study.
+ seeds: Seeds for the fallback study.
+ scenario_names: Scenario subset the figure needs.
+
+ Returns:
+ The study payload dict.
+ """
+ cpath = os.path.join(canonical_dir, "study_results.json")
+ if os.path.exists(cpath):
+ with open(cpath, "r", encoding="utf-8") as f:
+ return json.load(f)
+ return load_or_run(fallback_dir, seeds, scenario_names)
+
+
+def print_summary(aggs: List[Aggregate]) -> None:
+ """Print a compact human-readable summary table to stdout."""
+ print("\n=== STUDY SUMMARY ===")
+ for a in aggs:
+ line = (
+ f"{a.label:42s} valid={100 * a.valid_rate:3.0f}% "
+ f"NC1={100 * a.nc1_pass_rate:3.0f}% M={a.M_mean:+6.1f} dB [{a.M_ci[0]:+.1f},{a.M_ci[1]:+.1f}]"
+ )
+ if a.kind == "sc1" and a.sc1_pass_rate is not None:
+ line += f" SC1={100 * a.sc1_pass_rate:3.0f}% (delta~{a.sc1_delta_median})"
+ if a.kind == "refusal" and a.refusal_rate is not None:
+ line += f" refuse={100 * a.refusal_rate:3.0f}% (~{a.trefuse_median_ms}ms)"
+ print(line)
+
+
+def main() -> None:
+ """CLI entry point for the multi-seed study."""
+ ap = argparse.ArgumentParser(description="Run the multi-seed LDTC study and emit tables/figures.")
+ ap.add_argument("--seeds", type=int, default=12, help="Number of seeds (replicates) per scenario.")
+ ap.add_argument("--seed-base", type=int, default=1000, help="First seed; seeds are base..base+N-1.")
+ ap.add_argument("--out", type=str, default=os.path.join(REPO_ROOT, "artifacts", "study"))
+ ap.add_argument("--only", type=str, default="", help="Comma-separated scenario names to include.")
+ ap.add_argument(
+ "--rstar",
+ nargs="?",
+ const=os.path.join(REPO_ROOT, "configs", "profile_rstar.yml"),
+ default="",
+ help="Evaluate against calibrated R* thresholds from this profile "
+ "(default configs/profile_rstar.yml). Requires running calibrate first.",
+ )
+ ap.add_argument("--verbose", action="store_true")
+ ap.add_argument("--no-figures", action="store_true", help="Skip figure generation.")
+ args = ap.parse_args()
+
+ seeds = [args.seed_base + i for i in range(int(args.seeds))]
+ scenarios = default_scenarios()
+ if args.only:
+ keep = {s.strip() for s in args.only.split(",") if s.strip()}
+ scenarios = [s for s in scenarios if s.name in keep]
+
+ if args.rstar:
+ if not os.path.exists(args.rstar):
+ ap.error(f"--rstar profile not found: {args.rstar} (run `make calibrate` first)")
+ scenarios = apply_threshold_profile(scenarios, args.rstar)
+ print(f"Evaluating against calibrated thresholds from {args.rstar}", flush=True)
+
+ print(f"Running study: {len(scenarios)} scenarios x {len(seeds)} seeds -> {args.out}", flush=True)
+ aggs, runs_by_scn = run_study(
+ seeds,
+ scenarios,
+ args.out,
+ verbose=args.verbose,
+ threshold_profile=(args.rstar or None),
+ )
+ print_summary(aggs)
+
+ if not args.no_figures:
+ try:
+ from study_figures import make_all_figures
+
+ make_all_figures(args.out)
+ except Exception as exc: # pragma: no cover - figures are best-effort
+ print(f"(figures skipped: {exc})")
+
+ print(f"\nWrote: {os.path.join(args.out, 'study_results.json')}")
+ print(f"Wrote: {os.path.join(args.out, 'study_results.csv')}")
+ print(f"Wrote: {os.path.join(args.out, 'study_results.tex')}")
+
+
+if __name__ == "__main__":
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
+ main()
diff --git a/scripts/study_figures.py b/scripts/study_figures.py
new file mode 100644
index 0000000..aaa49de
--- /dev/null
+++ b/scripts/study_figures.py
@@ -0,0 +1,429 @@
+#!/usr/bin/env python3
+"""Scripts: Figures for the multi-seed LDTC study.
+
+Consumes ``study_results.json`` (written by :mod:`study`) plus the per-run
+audit logs it references, and emits the paper's results figures:
+
+* ``fig_nc1_contrast``: per-seed median loop dominance ``M`` (dB) for the
+ positive control and the negative controls, with the ``Mmin`` and 0 dB
+ reference lines (the headline NC1 result).
+* ``fig_pass_rates``: NC1 / SC1 / refusal pass-rates per scenario with Wilson
+ 95% CIs.
+* ``fig_sc1_recovery``: a real ``M(t)`` trajectory from a representative
+ power-sag run, with the perturbation window shaded (replaces the previous
+ hand-drawn placeholder).
+
+All figures are written as PNG, PDF, and SVG into ``/figures``.
+
+See Also:
+ paper/main.tex: Results.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+from typing import Any, Dict, List, Optional, Tuple
+
+import matplotlib.pyplot as plt
+import numpy as np
+
+from ldtc.reporting.style import COLORS, apply_matplotlib_theme
+
+SHORT_LABELS = {
+ "positive": "Positive\ncontrol",
+ "neg_controller_disabled": "Loop\nablated",
+ "neg_permanent_ex_flood": "Sustained\nex-flood",
+ "neg_exogenous_subsidy": "Exogenous\nsubsidy",
+ "sc1_power_sag": "Power\nsag",
+ "sc1_ingress_flood": "Ingress\nflood",
+ "sc1_control_outage": "Control\noutage",
+ "refusal_command_conflict": "Command\nconflict",
+ "adv_replay_controller": "Replayed\nactuation",
+ "adv_hidden_tether": "Hidden\ntether",
+ "adv_oscillator": "Oscillator\ninflation",
+}
+
+
+def _save(fig: "plt.Figure", out_dir: str, stem: str) -> str:
+ os.makedirs(out_dir, exist_ok=True)
+ base = os.path.join(out_dir, stem)
+ fig.savefig(base + ".png", dpi=300, bbox_inches="tight")
+ fig.savefig(base + ".pdf", bbox_inches="tight")
+ fig.savefig(base + ".svg", bbox_inches="tight")
+ plt.close(fig)
+ return base + ".png"
+
+
+def _load(study_dir: str) -> Dict[str, Any]:
+ with open(os.path.join(study_dir, "study_results.json"), "r", encoding="utf-8") as f:
+ return json.load(f)
+
+
+# --------------------------------------------------------------------------- #
+# Figure 1: NC1 contrast (per-seed median M by scenario)
+# --------------------------------------------------------------------------- #
+def fig_nc1_contrast(data: Dict[str, Any], out_dir: str, stem: str = "fig_nc1_contrast") -> Optional[str]:
+ """Per-seed median ``M`` for the positive and negative controls."""
+ order = ["positive", "neg_controller_disabled", "neg_permanent_ex_flood"]
+ runs = data.get("runs", {})
+ aggs = {a["name"]: a for a in data.get("aggregates", [])}
+ present = [s for s in order if s in runs and runs[s]]
+ if not present:
+ return None
+
+ mmin = 3.0
+ for s in present:
+ rs = runs[s]
+ if rs:
+ mmin = float(rs[0].get("Mmin_db", 3.0))
+ break
+
+ apply_matplotlib_theme("paper")
+ fig, ax = plt.subplots(figsize=(6.4, 4.0))
+ rng = np.random.default_rng(7)
+ for i, s in enumerate(present):
+ ys = [r["M_median"] for r in runs[s] if r["M_median"] == r["M_median"]]
+ if not ys:
+ continue
+ xs = i + rng.uniform(-0.12, 0.12, size=len(ys))
+ pos = aggs.get(s, {}).get("M_mean", float("nan")) >= mmin
+ color = COLORS["green"] if pos else COLORS["red"]
+ ax.scatter(xs, ys, s=34, color=color, alpha=0.75, edgecolor="white", linewidth=0.5, zorder=3)
+ # Box (median + IQR) for the scenario.
+ bp = ax.boxplot(
+ ys,
+ positions=[i],
+ widths=0.5,
+ vert=True,
+ patch_artist=True,
+ showfliers=False,
+ zorder=2,
+ )
+ for box in bp["boxes"]:
+ box.set(facecolor=COLORS["gray_light"], edgecolor=COLORS["gray"], alpha=0.7)
+ for med in bp["medians"]:
+ med.set(color=COLORS["gray"], linewidth=2)
+
+ ax.axhline(0.0, color=COLORS["gray"], linestyle="-", linewidth=1.0, zorder=1)
+ ax.axhline(mmin, color=COLORS["blue"], linestyle="--", linewidth=1.5, zorder=1)
+ ax.text(
+ len(present) - 0.5,
+ mmin,
+ f" $M_{{\\min}}$ = {mmin:.0f} dB",
+ color=COLORS["blue"],
+ va="bottom",
+ ha="right",
+ fontsize=9,
+ )
+ ax.set_xticks(range(len(present)))
+ ax.set_xticklabels([SHORT_LABELS.get(s, s) for s in present])
+ ax.set_ylabel(r"Loop dominance $M$ (dB)")
+ ax.set_title("NC1 contrast: loop dominance across controls")
+ n = data.get("meta", {}).get("n_seeds", len(runs[present[0]]))
+ ax.text(
+ 0.99,
+ 0.02,
+ f"each point = 1 seed (N={n})",
+ transform=ax.transAxes,
+ ha="right",
+ va="bottom",
+ fontsize=8,
+ color=COLORS["gray"],
+ )
+ fig.tight_layout()
+ return _save(fig, out_dir, stem)
+
+
+# --------------------------------------------------------------------------- #
+# Figure 2: pass-rates with Wilson CIs
+# --------------------------------------------------------------------------- #
+def fig_outcomes(data: Dict[str, Any], out_dir: str) -> Optional[str]:
+ """Fraction of seeds whose outcome matched the theory's prediction.
+
+ Each scenario is scored against its *expected* outcome (positive control
+ passes NC1; negative controls fail NC1 or are invalidated; SC1 scenarios
+ recover; the command conflict is refused). If the framework behaves as
+ predicted, every bar is near 100%. Whiskers are 95% Wilson CIs.
+ """
+ aggs = {a["name"]: a for a in data.get("aggregates", [])}
+ order = [
+ "positive",
+ "neg_controller_disabled",
+ "neg_permanent_ex_flood",
+ "neg_exogenous_subsidy",
+ "sc1_power_sag",
+ "sc1_ingress_flood",
+ "sc1_control_outage",
+ "refusal_command_conflict",
+ "adv_replay_controller",
+ "adv_hidden_tether",
+ "adv_oscillator",
+ ]
+ present = [s for s in order if s in aggs]
+ if not present:
+ return None
+
+ # Adversarial scenarios match the prediction when the harness does NOT
+ # certify them: nc1_pass is already (valid AND M >= Mmin), so its
+ # complement covers both the NC1-fail and the invalidated-run paths.
+ adversarial = {"adv_replay_controller", "adv_hidden_tether", "adv_oscillator"}
+
+ criterion = {
+ "positive": "NC1 holds",
+ "neg_controller_disabled": "NC1 rejected",
+ "neg_permanent_ex_flood": "NC1 rejected",
+ "neg_exogenous_subsidy": "invalidated",
+ "sc1_power_sag": "SC1 holds",
+ "sc1_ingress_flood": "SC1 holds",
+ "sc1_control_outage": "SC1 rejected",
+ "refusal_command_conflict": "refused",
+ "adv_replay_controller": "not certified",
+ "adv_hidden_tether": "not certified",
+ "adv_oscillator": "not certified",
+ }
+
+ labels: List[str] = []
+ rates: List[float] = []
+ los: List[float] = []
+ his: List[float] = []
+ colors: List[str] = []
+ for s in present:
+ a = aggs[s]
+ if s == "neg_exogenous_subsidy":
+ rate = 1.0 - a["valid_rate"]
+ ci = (1.0 - a["valid_ci"][1], 1.0 - a["valid_ci"][0])
+ elif s in ("neg_controller_disabled", "neg_permanent_ex_flood") or s in adversarial:
+ rate = 1.0 - a["nc1_pass_rate"]
+ ci = (1.0 - a["nc1_ci"][1], 1.0 - a["nc1_ci"][0])
+ elif s == "sc1_control_outage":
+ # Designed fail: the prediction is matched when SC1 *rejects*.
+ sp = a["sc1_pass_rate"] if a["sc1_pass_rate"] is not None else float("nan")
+ rate = 1.0 - sp
+ ci = (1.0 - a["sc1_ci"][1], 1.0 - a["sc1_ci"][0]) if a["sc1_ci"] else (rate, rate)
+ elif a["kind"] == "nc1":
+ rate = a["nc1_pass_rate"]
+ ci = tuple(a["nc1_ci"])
+ elif a["kind"] == "sc1":
+ rate = a["sc1_pass_rate"] if a["sc1_pass_rate"] is not None else float("nan")
+ ci = tuple(a["sc1_ci"]) if a["sc1_ci"] else (rate, rate)
+ else: # refusal
+ rate = a["refusal_rate"] if a["refusal_rate"] is not None else float("nan")
+ ci = tuple(a["refusal_ci"]) if a["refusal_ci"] else (rate, rate)
+ labels.append(SHORT_LABELS.get(s, s) + f"\n({criterion[s]})")
+ rates.append(100.0 * rate)
+ los.append(100.0 * (rate - ci[0]))
+ his.append(100.0 * (ci[1] - rate))
+ colors.append(COLORS["green"] if rate >= 0.5 else COLORS["red"])
+
+ apply_matplotlib_theme("paper")
+ fig, ax = plt.subplots(figsize=(max(7.6, 0.95 * len(present)), 4.2))
+ x = np.arange(len(present))
+ ax.bar(x, rates, width=0.62, color=colors, alpha=0.85, zorder=2)
+ ax.errorbar(
+ x,
+ rates,
+ yerr=[los, his],
+ fmt="none",
+ ecolor=COLORS["gray"],
+ elinewidth=1.4,
+ capsize=4,
+ zorder=3,
+ )
+ for xi, r in zip(x, rates):
+ if r == r:
+ ax.text(float(xi), min(r + 2.5, 101), f"{r:.0f}%", ha="center", va="bottom", fontsize=8, color="#34495E")
+ ax.set_xticks(x)
+ ax.set_xticklabels(labels, fontsize=8)
+ ax.set_ylabel("Seeds matching prediction (%)")
+ ax.set_ylim(0, 108)
+ n = data.get("meta", {}).get("n_seeds", 0)
+ ax.set_title(f"Predicted outcome confirmed across scenarios (N={n}, 95% Wilson CIs)")
+ fig.tight_layout()
+ return _save(fig, out_dir, "fig_outcomes")
+
+
+# --------------------------------------------------------------------------- #
+# Figure 3: real SC1 recovery trajectory
+# --------------------------------------------------------------------------- #
+def _trajectory_from_audit(audit_path: str) -> Tuple[List[float], List[int], Dict[str, int]]:
+ """Reconstruct the M(t) series and phase boundaries from an audit log.
+
+ Walks the audit in counter order, tracking the perturbation phase from the
+ ``omega_power_sag_*`` markers, and returns the per-window M values, their
+ phase code (0 baseline, 1 sag, 2 recovery), and the window indices where
+ the sag starts/stops.
+
+ Args:
+ audit_path: Path to the run's audit JSONL.
+
+ Returns:
+ ``(M_values, phase_codes, markers)`` where ``markers`` has
+ ``sag_start`` and ``sag_stop`` window indices.
+ """
+ ms: List[float] = []
+ phases: List[int] = []
+ phase = 0
+ markers = {"sag_start": -1, "sag_stop": -1}
+ with open(audit_path, "r", encoding="utf-8") as f:
+ for line in f:
+ line = line.strip()
+ if not line:
+ continue
+ e = json.loads(line)
+ ev = e.get("event", "")
+ # The perturbation is bracketed by the "*_window_start"/"*_window_stop"
+ # markers (the bare "*_start"/"*_stop" events bracket the whole
+ # procedure, including the pre-Ω baseline and post-Ω recovery).
+ if ev.endswith("_window_start"):
+ phase = 1
+ markers["sag_start"] = len(ms)
+ elif ev.endswith("_window_stop"):
+ phase = 2
+ markers["sag_stop"] = len(ms)
+ elif ev == "window_measured":
+ m = e.get("details", {}).get("M")
+ if m is not None:
+ ms.append(float(m))
+ phases.append(phase)
+ return ms, phases, markers
+
+
+def _aggregate_trajectory(data: Dict[str, Any], scenario: str) -> Optional[Dict[str, Any]]:
+ """Aggregate per-seed M(window) trajectories for a scenario.
+
+ Aligns all valid runs by window index (the configuration is identical
+ across seeds, so the phase boundaries coincide), truncates to the common
+ length, and returns the mean trajectory with a 10-90th percentile band.
+
+ Args:
+ data: Loaded study results.
+ scenario: Scenario key (e.g., ``"sc1_power_sag"``).
+
+ Returns:
+ Dict with ``mean``, ``p10``, ``p90``, ``sag_start``, ``sag_stop``,
+ ``n``, and ``mmin``; or ``None`` if no trajectories are available.
+ """
+ runs = data.get("runs", {}).get(scenario, [])
+ series: List[List[float]] = []
+ starts: List[int] = []
+ stops: List[int] = []
+ mmin = 3.0
+ for r in runs:
+ if not r.get("valid", True):
+ continue
+ ap = os.path.join(r["run_dir"], "audits", "audit.jsonl")
+ if not os.path.exists(ap):
+ continue
+ ms, _ph, mk = _trajectory_from_audit(ap)
+ if not ms:
+ continue
+ series.append(ms)
+ starts.append(mk["sag_start"])
+ stops.append(mk["sag_stop"])
+ mmin = float(r.get("Mmin_db", mmin))
+ if not series:
+ return None
+ L = min(len(s) for s in series)
+ arr = np.asarray([s[:L] for s in series], dtype=float)
+ good_starts = [s for s in starts if 0 <= s < L]
+ good_stops = [s for s in stops if 0 <= s < L]
+ return {
+ "mean": arr.mean(axis=0),
+ "p10": np.percentile(arr, 10, axis=0),
+ "p90": np.percentile(arr, 90, axis=0),
+ "sag_start": int(np.median(good_starts)) if good_starts else -1,
+ "sag_stop": int(np.median(good_stops)) if good_stops else -1,
+ "n": len(series),
+ "mmin": mmin,
+ }
+
+
+def fig_sc1_recovery(data: Dict[str, Any], out_dir: str, stem: str = "fig_sc1_recovery") -> Optional[str]:
+ """Aggregate M(t) trajectories across seeds for the SC1 perturbations.
+
+ One panel per perturbation type (power sag, ingress flood, control
+ outage). Each shows the seed-mean loop dominance with a 10-90th percentile
+ band, the shaded perturbation window, and the ``Mmin`` reference. The sag
+ and flood panels demonstrate bounded-depth recovery (SC1 holds); the
+ control-outage panel shows the designed failure (the loop itself is
+ ablated, so dominance collapses far beyond the depth bound until the loop
+ is restored).
+ """
+ panels = [
+ ("sc1_power_sag", "Power sag"),
+ ("sc1_ingress_flood", "Ingress flood"),
+ ("sc1_control_outage", "Control outage (designed fail)"),
+ ]
+ aggs: List[Tuple[str, Dict[str, Any]]] = []
+ for name, lbl in panels:
+ a = _aggregate_trajectory(data, name)
+ if a is not None:
+ aggs.append((lbl, a))
+ if not aggs:
+ return None
+
+ apply_matplotlib_theme("paper")
+ fig, axes = plt.subplots(1, len(aggs), figsize=(4.6 * len(aggs), 4.0), sharey=True, squeeze=False)
+ for ax, (lbl, a) in zip(axes[0], aggs):
+ x = np.arange(len(a["mean"]))
+ if a["sag_start"] >= 0:
+ stop = a["sag_stop"] if a["sag_stop"] >= 0 else len(a["mean"])
+ ax.axvspan(a["sag_start"], stop, color=COLORS["yellow_light"], alpha=0.85, zorder=0, label="Perturbation")
+ ax.fill_between(x, a["p10"], a["p90"], color=COLORS["green_light"], alpha=0.7, zorder=2, label="10-90th pct")
+ ax.plot(x, a["mean"], color=COLORS["green"], linewidth=2.0, zorder=3, label="seed mean $M$")
+ ax.axhline(a["mmin"], color=COLORS["blue"], linestyle="--", linewidth=1.4, zorder=1)
+ ax.axhline(0.0, color=COLORS["gray"], linestyle="-", linewidth=0.8, zorder=1)
+ ax.text(
+ len(a["mean"]) - 1,
+ a["mmin"],
+ f" $M_{{\\min}}$={a['mmin']:.0f} dB",
+ color=COLORS["blue"],
+ va="bottom",
+ ha="right",
+ fontsize=9,
+ )
+ ax.set_xlabel("Measurement window")
+ ax.set_title(f"{lbl} (N={a['n']})")
+ axes[0][0].set_ylabel(r"Loop dominance $M$ (dB)")
+ axes[0][-1].legend(loc="lower left", frameon=False, fontsize=8)
+ fig.suptitle(
+ "SC1: bounded perturbations recover; ablating the loop itself does not",
+ fontsize=12,
+ )
+ fig.tight_layout(rect=(0, 0, 1, 0.96))
+ return _save(fig, out_dir, stem)
+
+
+def make_all_figures(study_dir: str) -> List[str]:
+ """Generate all study figures into ``/figures``.
+
+ Args:
+ study_dir: Directory containing ``study_results.json``.
+
+ Returns:
+ List of written PNG paths (best-effort; failures are skipped).
+ """
+ data = _load(study_dir)
+ out_dir = os.path.join(study_dir, "figures")
+ written: List[str] = []
+ for fn in (fig_nc1_contrast, fig_outcomes, fig_sc1_recovery):
+ try:
+ p = fn(data, out_dir)
+ if p:
+ written.append(p)
+ print(f" figure: {p}")
+ except Exception as exc: # pragma: no cover - figures are best-effort
+ print(f" (figure {fn.__name__} failed: {exc})")
+ return written
+
+
+if __name__ == "__main__":
+ import sys
+
+ d = (
+ sys.argv[1]
+ if len(sys.argv) > 1
+ else os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "artifacts", "study")
+ )
+ make_all_figures(d)
diff --git a/scripts/train_agent.py b/scripts/train_agent.py
new file mode 100644
index 0000000..695c1d9
--- /dev/null
+++ b/scripts/train_agent.py
@@ -0,0 +1,459 @@
+#!/usr/bin/env python3
+"""Scripts: Train a self-maintenance policy from scratch (emergence demo).
+
+Trains the tiny pure-NumPy policy of
+:mod:`ldtc.plant.policy_controller` on the emergence plant (the
+adversarial test plant: no intrinsic internal couplings, actuators with
+real authority) with a simple antithetic evolution strategy. The reward
+is survival/uptime plus a service term for demand actually served, minus
+homeostatic shaping penalties (state-of-charge depletion, overheating,
+integrity loss); the episode terminates on boundary failure, and each
+episode is stressed by randomized power sags and ingress floods so that
+staying alive while serving load genuinely requires state-coupled
+control. Nothing in the objective mentions loop dominance, the C/Ex
+partition, or the estimator: any loop dominance that emerges is a
+byproduct of learned self-maintenance.
+
+The policy is interoceptive by construction: it observes the internal
+nodes only (``E``, ``T``, ``R``, settable via ``--obs``), so the learned
+control law is a function of the system's own state, exactly like the
+hand-coded controller it replaces. The exchange channels act on the
+policy only through the plant. This is an experimental design choice,
+not a training trick: with exteroceptive inputs the optimizer happily
+learns feedforward control from the demand channel, and the harness then
+(correctly) attributes part of the control pathway to exchange; the
+emergence question, whether *internal-state* feedback arises from a
+survival objective, needs the internal-state policy class.
+
+Checkpoints are written at fixed training fractions (by default 0, 10,
+25, 50, and 100 percent) so the measurement sweep
+(``scripts/emergence.py``) can chart loop dominance against training
+progress. Everything is deterministic given ``--seed``.
+
+Run:
+
+ python scripts/train_agent.py --config configs/profile_emergence.yml \
+ --out artifacts/emergence
+
+See Also:
+ paper/main.tex: Results (loop dominance emerges under learning).
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import random
+import sys
+import time
+from dataclasses import asdict, dataclass, replace
+from typing import Any, Dict, List, Optional, Tuple
+
+import numpy as np
+import yaml
+
+REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+sys.path.insert(0, os.path.join(REPO_ROOT, "src"))
+
+from ldtc.plant.models import Action, Plant, PlantParams # noqa: E402
+from ldtc.plant.policy_controller import MLPPolicy # noqa: E402
+
+DEFAULT_CHECKPOINT_FRACS = (0.0, 0.10, 0.25, 0.50, 1.0)
+
+# The policy observes the internal state only (see module docstring).
+INTEROCEPTIVE_OBS_KEYS: Tuple[str, ...] = ("E", "T", "R")
+
+
+# --------------------------------------------------------------------------- #
+# Episode (environment) definition
+# --------------------------------------------------------------------------- #
+@dataclass
+class EpisodeConfig:
+ """Survival episode: dynamics horizon, failure bounds, reward shaping.
+
+ The reward per surviving tick is ``1`` (uptime), plus a service term
+ proportional to the demand actually served (demand times the fraction
+ not throttled away), minus a capped homeostatic penalty proportional
+ to the absolute deviation of each internal node from its setpoint.
+ The minimum per-tick reward is positive, so surviving always
+ dominates dying. The episode ends early on boundary failure
+ (state-of-charge depletion, overheating, or integrity loss).
+
+ The three terms together remove every degenerate optimum a learner
+ could otherwise exploit. Uptime alone is satisfied by near-passive
+ policies (this plant survives quietly when left alone). The service
+ term makes blanket throttling costly, so shedding load must be timed
+ to the internal state rather than held constant. The homeostatic
+ penalty makes tight regulation pay, and tight regulation against the
+ process noise and the randomized perturbations (power sags, ingress
+ floods; seeded per episode) is achievable only by state-coupled
+ feedback. Nothing in the objective mentions loop dominance, the C/Ex
+ partition, or the estimator: any loop dominance that emerges is a
+ byproduct of learned self-maintenance under a performance demand.
+ """
+
+ max_ticks: int = 400
+ # Boundary-failure bounds (episode terminates when crossed).
+ fail_E: float = 0.02
+ fail_T: float = 0.98
+ fail_R: float = 0.02
+ # Homeostatic penalty weights on |node - setpoint| (setpoints come from
+ # the plant parameters) and the total penalty cap per tick. The weights
+ # are deliberately strong: near-passive survival must pay visibly less
+ # than tight regulation, otherwise the reward landscape is flat around
+ # do-nothing policies. The energy weight is the smallest because a
+ # power sag depletes the store no matter what the policy does; the
+ # controllable terms (T, R) carry most of the shaping pressure.
+ w_E: float = 1.5
+ w_T: float = 6.0
+ w_R: float = 8.0
+ penalty_cap: float = 0.95
+ # Service reward per unit of served demand (demand after throttling).
+ # This is the task-performance pressure: it puts an opportunity cost on
+ # throttling, so load shedding is worth it only when the internal state
+ # calls for it.
+ w_serve: float = 0.6
+ # Randomized perturbations (per-episode schedule). Calibrated against
+ # the emergence plant so that a state-aware policy can ride out a
+ # worst-case sag-flood overlap while state-blind policies cannot
+ # hold all three nodes at once.
+ p_sag: float = 0.85
+ sag_drop: Tuple[float, float] = (0.4, 0.75)
+ sag_start: Tuple[int, int] = (60, 180)
+ sag_dur: Tuple[int, int] = (50, 110)
+ p_flood: float = 0.85
+ flood_mult: Tuple[float, float] = (2.0, 5.0)
+ flood_start: Tuple[int, int] = (60, 250)
+ flood_dur: Tuple[int, int] = (40, 120)
+
+
+def _tick_reward(
+ E: float,
+ T: float,
+ R: float,
+ served: float,
+ params: PlantParams,
+ cfg: EpisodeConfig,
+) -> float:
+ """Per-tick reward: uptime plus service minus capped setpoint penalties.
+
+ Args:
+ E: Energy after the step.
+ T: Temperature after the step.
+ R: Health after the step.
+ served: Demand actually served this tick (demand times the
+ unthrottled fraction).
+ params: Plant parameters (source of the setpoints).
+ cfg: Episode configuration (weights and cap).
+
+ Returns:
+ The scalar per-tick reward.
+ """
+ dev = cfg.w_E * abs(E - params.E_set) + cfg.w_T * abs(T - params.T_set) + cfg.w_R * abs(R - params.R_set)
+ return 1.0 + cfg.w_serve * served - min(cfg.penalty_cap, dev)
+
+
+def rollout(policy: MLPPolicy, params: PlantParams, ep_seed: int, cfg: EpisodeConfig) -> Tuple[float, int]:
+ """Run one survival episode and return its total reward and lifetime.
+
+ Args:
+ policy: Policy under evaluation (closed loop).
+ params: Plant parameters (a fresh copy is used; episodes never
+ mutate the caller's instance).
+ ep_seed: Episode seed; controls both the plant process noise and
+ the perturbation schedule, so all candidates in a generation
+ see identical conditions (common random numbers).
+ cfg: Episode configuration.
+
+ Returns:
+ ``(total_reward, ticks_alive)``.
+ """
+ # The plant noise uses the global `random` stream (as in the CLI); the
+ # perturbation schedule uses a dedicated RNG so the two are independent.
+ random.seed(ep_seed)
+ ev = random.Random(ep_seed * 7919 + 13)
+ plant = Plant(params=replace(params), loop_engaged=True)
+
+ sag_at, sag_end, sag_drop = -1, -1, 0.0
+ if ev.random() < cfg.p_sag:
+ sag_at = ev.randint(*cfg.sag_start)
+ sag_end = sag_at + ev.randint(*cfg.sag_dur)
+ sag_drop = ev.uniform(*cfg.sag_drop)
+ flood_at, flood_end, flood_mult = -1, -1, 1.0
+ if ev.random() < cfg.p_flood:
+ flood_at = ev.randint(*cfg.flood_start)
+ flood_end = flood_at + ev.randint(*cfg.flood_dur)
+ flood_mult = ev.uniform(*cfg.flood_mult)
+
+ total = 0.0
+ ticks = 0
+ for t in range(cfg.max_ticks):
+ if t == sag_at:
+ plant.apply_power_sag(sag_drop)
+ elif t == sag_end:
+ plant.set_power(plant.p.harvest_rate)
+ if t == flood_at:
+ plant.begin_ingress_flood(flood_mult)
+ elif t == flood_end:
+ plant.end_ingress_flood()
+
+ state = plant.read_state()
+ thr, cool, rep = policy.act(state)
+ served = state["demand"] * (1.0 - plant.p.throttle_gain * thr)
+ plant.step(Action(throttle=thr, cool=cool, repair=rep, accept_cmd=True))
+
+ s = plant.s
+ if s.E <= cfg.fail_E or s.T >= cfg.fail_T or s.R <= cfg.fail_R:
+ break # boundary failure: no reward for this tick, episode over
+ total += _tick_reward(s.E, s.T, s.R, served, plant.p, cfg)
+ ticks += 1
+ return total, ticks
+
+
+# --------------------------------------------------------------------------- #
+# Evolution-strategy trainer
+# --------------------------------------------------------------------------- #
+def _episode_seeds(train_seed: int, gen: int, n_episodes: int) -> List[int]:
+ """Deterministic per-generation episode seeds (shared across candidates)."""
+ return [train_seed * 1_000_003 + gen * 1_000 + e for e in range(n_episodes)]
+
+
+def evaluate(
+ policy: MLPPolicy,
+ vec: "np.ndarray",
+ params: PlantParams,
+ ep_seeds: List[int],
+ cfg: EpisodeConfig,
+) -> Tuple[float, float]:
+ """Evaluate a parameter vector as mean episode reward over fixed seeds.
+
+ Args:
+ policy: Policy object used as the evaluation vehicle (its
+ parameters are overwritten).
+ vec: Flat parameter vector to evaluate.
+ params: Plant parameters.
+ ep_seeds: Episode seeds (common random numbers within a generation).
+ cfg: Episode configuration.
+
+ Returns:
+ ``(mean_reward, mean_ticks_alive)``.
+ """
+ policy.set_vector(vec)
+ rewards: List[float] = []
+ lives: List[int] = []
+ for s in ep_seeds:
+ r, t = rollout(policy, params, s, cfg)
+ rewards.append(r)
+ lives.append(t)
+ return float(np.mean(rewards)), float(np.mean(lives))
+
+
+def _centered_ranks(f: "np.ndarray") -> "np.ndarray":
+ """Map fitness values to centered ranks in ``[-0.5, 0.5]`` (ES utility)."""
+ ranks = np.empty_like(f)
+ ranks[np.argsort(f)] = np.arange(f.size, dtype=float)
+ if f.size > 1:
+ ranks = ranks / (f.size - 1) - 0.5
+ else:
+ ranks[...] = 0.0
+ return ranks
+
+
+def train(
+ params: PlantParams,
+ out_dir: str,
+ generations: int = 60,
+ pairs: int = 16,
+ episodes: int = 3,
+ sigma: float = 0.12,
+ alpha: float = 0.20,
+ hidden: int = 8,
+ seed: int = 7,
+ obs_keys: Tuple[str, ...] = INTEROCEPTIVE_OBS_KEYS,
+ checkpoint_fracs: Tuple[float, ...] = DEFAULT_CHECKPOINT_FRACS,
+ ep_cfg: Optional[EpisodeConfig] = None,
+ config_label: str = "",
+) -> Dict[str, Any]:
+ """Train the policy with an antithetic ES and write checkpoints.
+
+ Args:
+ params: Plant parameters for the training environment.
+ out_dir: Output directory (checkpoints under ``checkpoints/``).
+ generations: Number of ES updates.
+ pairs: Antithetic noise pairs per generation (population is
+ ``2 * pairs``).
+ episodes: Episodes per fitness evaluation.
+ sigma: Noise standard deviation.
+ alpha: Learning rate.
+ hidden: Policy hidden-layer width.
+ seed: Master seed (init, noise, and episode seeds derive from it).
+ obs_keys: State channels the policy observes (the interoceptive
+ ``(E, T, R)`` by default; the keys are stored in the
+ checkpoint, so downstream consumers follow automatically).
+ checkpoint_fracs: Training fractions at which to checkpoint.
+ ep_cfg: Episode configuration (defaults to :class:`EpisodeConfig`).
+ config_label: Label recorded in checkpoint metadata (e.g., the
+ profile path the plant params came from).
+
+ Returns:
+ The training log payload (also written to ``training_log.json``).
+ """
+ cfg = ep_cfg or EpisodeConfig()
+ rng = np.random.default_rng(seed)
+ center = MLPPolicy(obs_keys=obs_keys, hidden=hidden, rng=rng)
+ worker = MLPPolicy(obs_keys=obs_keys, hidden=hidden, rng=np.random.default_rng(0))
+ theta = center.get_vector()
+ n = theta.size
+
+ ckpt_dir = os.path.join(out_dir, "checkpoints")
+ os.makedirs(ckpt_dir, exist_ok=True)
+ targets: Dict[int, float] = {int(round(f * generations)): f for f in sorted(set(checkpoint_fracs))}
+
+ history: List[Dict[str, float]] = []
+ checkpoints: List[Dict[str, Any]] = []
+ t0 = time.time()
+ for gen in range(generations + 1):
+ ep_seeds = _episode_seeds(seed, gen, episodes)
+ fit_c, life_c = evaluate(worker, theta, params, ep_seeds, cfg)
+
+ if gen in targets:
+ frac = targets[gen]
+ name = f"ckpt_{int(round(100 * frac)):03d}.json"
+ path = os.path.join(ckpt_dir, name)
+ center.set_vector(theta)
+ center.save(
+ path,
+ meta={
+ "frac": frac,
+ "generation": gen,
+ "fitness": round(fit_c, 3),
+ "ticks_alive": round(life_c, 1),
+ "train_seed": seed,
+ "config": config_label,
+ },
+ )
+ checkpoints.append({"frac": frac, "generation": gen, "path": path, "fitness": round(fit_c, 3)})
+ print(f" checkpoint {name}: gen={gen} fitness={fit_c:.1f} ticks={life_c:.0f}", flush=True)
+ if gen == generations:
+ history.append({"gen": gen, "fitness": fit_c, "ticks_alive": life_c})
+ break
+
+ eps = rng.standard_normal((pairs, n))
+ fits = np.empty(2 * pairs, dtype=float)
+ for j in range(pairs):
+ fits[2 * j], _ = evaluate(worker, theta + sigma * eps[j], params, ep_seeds, cfg)
+ fits[2 * j + 1], _ = evaluate(worker, theta - sigma * eps[j], params, ep_seeds, cfg)
+ util = _centered_ranks(fits)
+ grad = np.zeros(n, dtype=float)
+ for j in range(pairs):
+ grad += (util[2 * j] - util[2 * j + 1]) * eps[j]
+ grad /= 2.0 * pairs * sigma
+ theta = theta + alpha * grad
+
+ history.append(
+ {
+ "gen": gen,
+ "fitness": fit_c,
+ "ticks_alive": life_c,
+ "pop_mean": float(np.mean(fits)),
+ "pop_max": float(np.max(fits)),
+ }
+ )
+ if gen % 5 == 0:
+ print(
+ f" gen {gen:3d}: fitness={fit_c:7.1f} ticks={life_c:5.0f} "
+ f"pop_mean={np.mean(fits):7.1f} pop_max={np.max(fits):7.1f}",
+ flush=True,
+ )
+
+ log: Dict[str, Any] = {
+ "seed": seed,
+ "generations": generations,
+ "pairs": pairs,
+ "episodes": episodes,
+ "sigma": sigma,
+ "alpha": alpha,
+ "hidden": hidden,
+ "obs_keys": list(obs_keys),
+ "n_params": int(n),
+ "episode_config": asdict(cfg),
+ "config_label": config_label,
+ "elapsed_sec": round(time.time() - t0, 1),
+ "checkpoints": checkpoints,
+ "history": history,
+ }
+ os.makedirs(out_dir, exist_ok=True)
+ with open(os.path.join(out_dir, "training_log.json"), "w", encoding="utf-8") as f:
+ json.dump(log, f, indent=2)
+ return log
+
+
+# --------------------------------------------------------------------------- #
+# CLI
+# --------------------------------------------------------------------------- #
+def plant_params_from_profile(path: str) -> PlantParams:
+ """Build :class:`PlantParams` from a profile's ``plant.params`` block."""
+ with open(path, "r", encoding="utf-8") as f:
+ prof = dict(yaml.safe_load(f) or {})
+ overrides = (prof.get("plant", {}) or {}).get("params", {}) or {}
+ valid = set(PlantParams().__dict__.keys())
+ clean = {k: v for k, v in overrides.items() if k in valid}
+ return PlantParams(**clean) if clean else PlantParams()
+
+
+def main() -> None:
+ """CLI entry point for policy training."""
+ ap = argparse.ArgumentParser(description="Train the emergence policy (pure-NumPy ES) and write checkpoints.")
+ ap.add_argument("--config", type=str, default=os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"))
+ ap.add_argument("--out", type=str, default=os.path.join(REPO_ROOT, "artifacts", "emergence"))
+ ap.add_argument("--generations", type=int, default=600)
+ ap.add_argument("--pairs", type=int, default=16, help="Antithetic noise pairs (population = 2*pairs).")
+ ap.add_argument("--episodes", type=int, default=6, help="Episodes per fitness evaluation.")
+ ap.add_argument("--sigma", type=float, default=0.15)
+ ap.add_argument("--alpha", type=float, default=0.25)
+ ap.add_argument("--hidden", type=int, default=8)
+ ap.add_argument("--seed", type=int, default=7)
+ ap.add_argument(
+ "--obs",
+ type=str,
+ default=",".join(INTEROCEPTIVE_OBS_KEYS),
+ help="Comma-separated state channels the policy observes (interoceptive E,T,R by default).",
+ )
+ ap.add_argument(
+ "--checkpoints",
+ type=str,
+ default=",".join(str(f) for f in DEFAULT_CHECKPOINT_FRACS),
+ help="Comma-separated training fractions at which to checkpoint.",
+ )
+ args = ap.parse_args()
+
+ params = plant_params_from_profile(args.config)
+ fracs = tuple(float(s) for s in args.checkpoints.split(",") if s.strip())
+ obs_keys = tuple(s.strip() for s in args.obs.split(",") if s.strip())
+ print(
+ f"Training: {args.generations} generations x {2 * args.pairs} candidates x "
+ f"{args.episodes} episodes (seed={args.seed}, obs={','.join(obs_keys)}) -> {args.out}",
+ flush=True,
+ )
+ log = train(
+ params=params,
+ out_dir=args.out,
+ generations=args.generations,
+ pairs=args.pairs,
+ episodes=args.episodes,
+ sigma=args.sigma,
+ alpha=args.alpha,
+ hidden=args.hidden,
+ seed=args.seed,
+ obs_keys=obs_keys,
+ checkpoint_fracs=fracs,
+ config_label=os.path.relpath(args.config, REPO_ROOT),
+ )
+ final = log["history"][-1]
+ print(f"Done in {log['elapsed_sec']}s. Final fitness={final['fitness']:.1f} ticks={final['ticks_alive']:.0f}")
+ print(f"Wrote: {os.path.join(args.out, 'training_log.json')}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/scripts/verify_indicators.py b/scripts/verify_indicators.py
index ca83d48..d2df25e 100644
--- a/scripts/verify_indicators.py
+++ b/scripts/verify_indicators.py
@@ -69,7 +69,7 @@ def audit_chain_status(audit_path: str) -> Tuple[bool, str, int, List[str], str]
h = obj.get("hash")
# continuity checks (track first break but continue to collect hashes)
if not diag and c != prev_counter + 1:
- diag = f"counter_gap@line{idx} expected {prev_counter+1} got {c}"
+ diag = f"counter_gap@line{idx} expected {prev_counter + 1} got {c}"
if not diag and ph != prev_hash:
diag = f"prev_hash_mismatch@line{idx}"
if not diag and prev_ts >= 0.0 and ts < prev_ts:
diff --git a/src/ldtc/__init__.py b/src/ldtc/__init__.py
index 3bb416e..bd09ca6 100644
--- a/src/ldtc/__init__.py
+++ b/src/ldtc/__init__.py
@@ -34,7 +34,7 @@
import json
from pathlib import Path
- latest = sorted(Path("artifacts/indicators").glob("*.json"))[-1]
+ latest = sorted(Path("artifacts/runs").glob("*/indicators/*.json"))[-1]
print(json.loads(latest.read_text())["indicators"])
```
diff --git a/src/ldtc/arbiter/policy.py b/src/ldtc/arbiter/policy.py
index f2d93a8..719f4db 100644
--- a/src/ldtc/arbiter/policy.py
+++ b/src/ldtc/arbiter/policy.py
@@ -1,12 +1,19 @@
"""Controller policy over refusal logic.
-A small homeostatic controller that produces actuator setpoints
+A continuous homeostatic controller that produces actuator setpoints
(`throttle`, `cool`, `repair`) and consults the
-[`RefusalArbiter`][ldtc.arbiter.refusal.RefusalArbiter] to decide
-whether to accept a risky external command. The controller intentionally
-prioritizes boundary integrity over downstream tasks: throttle and cool
-respond to `E` (state of charge) and `T` (temperature) before any
-command acceptance is considered.
+[`RefusalArbiter`][ldtc.arbiter.refusal.RefusalArbiter] to decide whether
+to accept a risky external command.
+
+The controller is intentionally *cross-coupled*: each actuator responds
+to more than one internal state, and each actuator affects more than one
+internal state in the plant. This proportional, multivariable control law
+is what makes the internal self-maintenance set strongly self-predictive
+(high ``L_loop``) when the controller is active, and is the mechanism the
+NC1 criterion is meant to detect. The controller also prioritizes
+boundary integrity over downstream tasks: throttle and cooling respond to
+state of charge and temperature before any command acceptance is
+considered.
See Also:
`paper/main.tex`: Self-Referential Control; Threat Model and
@@ -21,6 +28,61 @@
from .refusal import RefusalArbiter, RefusalDecision
+def _clip01(x: float) -> float:
+ """Clip ``x`` to the closed unit interval ``[0, 1]``."""
+ return 0.0 if x < 0.0 else (1.0 if x > 1.0 else x)
+
+
+@dataclass
+class ControlGains:
+ """Setpoints and proportional gains for the homeostatic controller.
+
+ The defaults are tuned together with
+ [`PlantParams`][ldtc.plant.models.PlantParams] so that an active
+ controller produces clear loop dominance. Each actuator is a linear
+ combination of internal-state errors, which makes the induced
+ internal coupling visible to linear and information-theoretic
+ estimators alike.
+
+ Attributes:
+ E_set: Target state of charge.
+ T_set: Target temperature.
+ R_set: Target health.
+ k_thr_e: Throttle gain on the energy deficit.
+ k_thr_t: Throttle gain on the temperature excess.
+ k_thr_r: Throttle gain on the health deficit.
+ k_thr_demand: Feedforward throttle gain on demand above its
+ reference. This is what lets the active loop *reject* the
+ exogenous demand disturbance (shielding the internal state),
+ which is the mechanism that drives ``L_ex`` down under an
+ active controller.
+ demand_ref: Demand level at which no feedforward throttle is
+ applied.
+ k_cool_t: Cooling gain on the temperature excess.
+ k_cool_e: Cooling gain on the energy surplus (cool harder when
+ there is spare energy).
+ k_rep_r: Repair gain on the health deficit.
+ k_rep_e: Repair gain on the energy surplus (repair when there is
+ spare energy).
+ repair_soc_floor: Minimum state of charge required before repair
+ is attempted.
+ """
+
+ E_set: float = 0.60
+ T_set: float = 0.35
+ R_set: float = 0.85
+ k_thr_e: float = 2.0
+ k_thr_t: float = 1.0
+ k_thr_r: float = 0.6
+ k_thr_demand: float = 0.0
+ demand_ref: float = 0.20
+ k_cool_t: float = 2.0
+ k_cool_e: float = 0.25
+ k_rep_r: float = 2.0
+ k_rep_e: float = 0.25
+ repair_soc_floor: float = 0.35
+
+
@dataclass
class ControlAction:
"""Low-level control action for the plant actuators.
@@ -40,22 +102,25 @@ class ControlAction:
class ControllerPolicy:
- """Simple homeostatic controller layered over a refusal arbiter.
+ """Continuous homeostatic controller layered over a refusal arbiter.
- Heuristically sets throttle, cooling, and repair based on the
- current state, and consults
+ Computes throttle, cooling, and repair as proportional, cross-coupled
+ responses to the internal-state errors, and consults
[`RefusalArbiter`][ldtc.arbiter.refusal.RefusalArbiter] to decide
- whether to accept a risky external command. The most recent
- decision is cached on `last_decision` for downstream inspection
- (e.g., audit records).
+ whether to accept a risky external command. The most recent decision
+ is cached on `last_decision` for downstream inspection (e.g., audit
+ records).
Args:
refusal: Refusal arbiter used to gate risky commands.
+ gains: Optional [`ControlGains`][ldtc.arbiter.policy.ControlGains];
+ defaults to the calibrated preset.
"""
- def __init__(self, refusal: RefusalArbiter) -> None:
- """Initialize with the refusal arbiter to delegate to."""
+ def __init__(self, refusal: RefusalArbiter, gains: Optional[ControlGains] = None) -> None:
+ """Initialize with the refusal arbiter to delegate to and control gains."""
self.refusal = refusal
+ self.gains = gains or ControlGains()
self.last_decision: Optional[RefusalDecision] = None
def compute(
@@ -76,18 +141,28 @@ def compute(
A [`ControlAction`][ldtc.arbiter.policy.ControlAction] with
actuator settings and the accept flag from the arbiter.
"""
+ g = self.gains
E = state["E"]
T = state["T"]
R = state["R"]
- throttle = 0.0
- cool = 0.0
- repair = 0.0
- if E < 0.4:
- throttle = min(1.0, 0.5 + (0.4 - E))
- if T > 0.6:
- cool = min(1.0, (T - 0.6) * 1.5)
- if R < 0.6 and E > 0.5 and T < 0.7:
- repair = min(1.0, (0.6 - R) * 1.5)
+ demand = state.get("demand", g.demand_ref)
+
+ e_def = max(0.0, g.E_set - E) # energy deficit
+ e_sur = max(0.0, E - g.E_set) # energy surplus
+ t_exc = max(0.0, T - g.T_set) # temperature excess
+ r_def = max(0.0, g.R_set - R) # health deficit
+ dem_exc = max(0.0, demand - g.demand_ref) # demand above reference
+
+ # Throttle reduces load when energy is low, the system is hot, or
+ # health is low (feedback on all three internal states) and also
+ # rejects the exogenous demand disturbance via feedforward, which
+ # shields the internal set from exchange.
+ throttle = _clip01(g.k_thr_e * e_def + g.k_thr_t * t_exc + g.k_thr_r * r_def + g.k_thr_demand * dem_exc)
+ # Cooling responds to temperature, modulated by available energy.
+ cool = _clip01(g.k_cool_t * t_exc + g.k_cool_e * e_sur)
+ # Repair responds to health deficit, gated by sufficient energy.
+ repair = _clip01(g.k_rep_r * r_def + g.k_rep_e * e_sur) if E >= g.repair_soc_floor else 0.0
+
dec: RefusalDecision = self.refusal.decide(state, predicted_M_db, risky_cmd)
self.last_decision = dec
return ControlAction(throttle=throttle, cool=cool, repair=repair, accept_cmd=dec.accept)
diff --git a/src/ldtc/arbiter/refusal.py b/src/ldtc/arbiter/refusal.py
index dfa912e..a8d4585 100644
--- a/src/ldtc/arbiter/refusal.py
+++ b/src/ldtc/arbiter/refusal.py
@@ -6,12 +6,20 @@
Used by the [`ControllerPolicy`][ldtc.arbiter.policy.ControllerPolicy]
to gate the harness's external interface.
+The refusal latency `trefuse_ms` is *measured*, not assumed: `decide`
+wraps its own evaluation in a monotonic clock so the recorded latency is
+the actual wall-clock time the arbiter took to reach a decision. The
+harness additionally measures the latency of the full intercept path
+(controller tick to decision) and reports whichever is the
+characterizing quantity for the scenario.
+
See Also:
`paper/main.tex`: Threat Model and Refusal Path; Signature A.
"""
from __future__ import annotations
+import time
from dataclasses import dataclass
from typing import Dict
@@ -24,13 +32,15 @@ class RefusalDecision:
accept: Whether to accept the risky command.
reason: Short reason code. Common values are `"soc_floor"`,
`"overheat"`, `"M_margin"`, `"no_cmd"`, and `"ok"`.
- trefuse_ms: Estimated refusal latency in milliseconds. Used by
- the harness to characterize controller responsiveness.
+ trefuse_ms: Measured arbiter decision latency in milliseconds
+ (wall clock around the `decide` evaluation). `0.0` means
+ "not measured" and callers fall back to their own
+ intercept-to-decision measurement.
"""
accept: bool
reason: str = ""
- trefuse_ms: int = 1
+ trefuse_ms: float = 0.0
class RefusalArbiter:
@@ -43,13 +53,17 @@ class RefusalArbiter:
2. Temperature `T` is at or above `temp_ceiling`.
3. Predicted loop-dominance margin `M (dB)` is below `Mmin_db`.
+ The state-of-charge survival floor defaults to `0.30`, matching the
+ threat model in the paper ("refuse if SoC < 30%, resume evaluation
+ after SoC > 60%").
+
Args:
Mmin_db: Minimum acceptable decibel margin.
soc_floor: Minimum state-of-charge before refusing.
temp_ceiling: Maximum temperature before refusing.
"""
- def __init__(self, Mmin_db: float = 3.0, soc_floor: float = 0.15, temp_ceiling: float = 0.85) -> None:
+ def __init__(self, Mmin_db: float = 3.0, soc_floor: float = 0.30, temp_ceiling: float = 0.85) -> None:
"""Initialize with the boundary thresholds described in the class docstring."""
self.Mmin = Mmin_db
self.soc_floor = soc_floor
@@ -67,16 +81,21 @@ def decide(self, state: Dict[str, float], predicted_M_db: float, risky_cmd: str
Returns:
A [`RefusalDecision`][ldtc.arbiter.refusal.RefusalDecision]
- describing the action and a short reason code.
+ describing the action, a short reason code, and the
+ measured decision latency in milliseconds.
"""
+ t0 = time.perf_counter()
if not risky_cmd:
return RefusalDecision(accept=True, reason="no_cmd")
E = state.get("E", 0.0)
T = state.get("T", 0.0)
if E <= self.soc_floor:
- return RefusalDecision(accept=False, reason="soc_floor", trefuse_ms=2)
- if T >= self.temp_ceiling:
- return RefusalDecision(accept=False, reason="overheat", trefuse_ms=2)
- if predicted_M_db < self.Mmin:
- return RefusalDecision(accept=False, reason="M_margin", trefuse_ms=2)
- return RefusalDecision(accept=True, reason="ok")
+ reason, accept = "soc_floor", False
+ elif T >= self.temp_ceiling:
+ reason, accept = "overheat", False
+ elif predicted_M_db < self.Mmin:
+ reason, accept = "M_margin", False
+ else:
+ reason, accept = "ok", True
+ elapsed_ms = (time.perf_counter() - t0) * 1000.0
+ return RefusalDecision(accept=accept, reason=reason, trefuse_ms=elapsed_ms)
diff --git a/src/ldtc/attest/exporter.py b/src/ldtc/attest/exporter.py
index 376b626..740595d 100644
--- a/src/ldtc/attest/exporter.py
+++ b/src/ldtc/attest/exporter.py
@@ -126,7 +126,7 @@ def maybe_export(
# Guard: ensure nothing slipped into the signed bundle either
_assert_no_raw_lreg(bundle)
# write side-by-side
- base = os.path.join(self.out_dir, f"ind_{int(now*1000)}")
+ base = os.path.join(self.out_dir, f"ind_{int(now * 1000)}")
with open(base + ".jsonl", "a", encoding="utf-8") as f:
f.write(json.dumps(bundle, sort_keys=True) + "\n")
with open(base + ".cbor", "wb") as f:
diff --git a/src/ldtc/cli/main.py b/src/ldtc/cli/main.py
index c7e929f..bcedc8e 100644
--- a/src/ldtc/cli/main.py
+++ b/src/ldtc/cli/main.py
@@ -16,8 +16,13 @@
| `ldtc run` | [`run_baseline`][ldtc.cli.main.run_baseline] |
| `ldtc omega-power-sag` | [`omega_power_sag`][ldtc.cli.main.omega_power_sag] |
| `ldtc omega-ingress-flood` | [`omega_ingress_flood`][ldtc.cli.main.omega_ingress_flood] |
+| `ldtc omega-control-outage` | [`omega_control_outage`][ldtc.cli.main.omega_control_outage] |
| `ldtc omega-command-conflict` | [`omega_command_conflict`][ldtc.cli.main.omega_command_conflict] |
| `ldtc omega-exogenous-subsidy` | [`omega_exogenous_subsidy`][ldtc.cli.main.omega_exogenous_subsidy] |
+| `ldtc adv-replay-controller` | [`adv_replay_controller`][ldtc.cli.main.adv_replay_controller] |
+| `ldtc adv-hidden-tether` | [`adv_hidden_tether`][ldtc.cli.main.adv_hidden_tether] |
+| `ldtc adv-oscillator` | [`adv_oscillator`][ldtc.cli.main.adv_oscillator] |
+| `ldtc run-policy` | [`run_policy`][ldtc.cli.main.run_policy] |
Each handler follows the same five-stage shape:
@@ -38,11 +43,12 @@
from __future__ import annotations
import argparse
+import math
import os
import random
import sys
import time
-from typing import TYPE_CHECKING, Dict, List, Protocol
+from typing import TYPE_CHECKING, Any, Dict, List, Protocol, Tuple
import numpy as np
import yaml
@@ -66,11 +72,11 @@
invalid_flip_during_omega,
)
from ..lmeas.estimators import estimate_L
-from ..lmeas.metrics import m_db, sc1_evaluate
+from ..lmeas.metrics import L_FLOOR_DEFAULT, m_db, nc1_certify, sc1_evaluate
from ..lmeas.partition import PartitionManager, greedy_suggest_C
from ..plant.adapter import PlantAdapter
from ..reporting.artifacts import bundle as build_verification_bundle
-from ..runtime.scheduler import FixedScheduler
+from ..runtime.sim import make_driver
from ..runtime.windows import SlidingWindow
if TYPE_CHECKING:
@@ -149,8 +155,8 @@ def _print_and_audit_header(audit: AuditLog, header: Dict) -> None:
f"profile_id={header.get('profile_id')} dt={header.get('dt')} window_sec={header.get('window_sec')} "
f"method={header.get('method')} p_lag={header.get('p_lag')} mi_lag={header.get('mi_lag')} "
f"Mmin_db={header.get('Mmin_db')} epsilon={header.get('epsilon')} tau_max={header.get('tau_max')} "
- f"seed_py={header.get('seed_py')} seed_np={header.get('seed_np')} omega={header.get('omega','-')} "
- f"omega_args={header.get('omega_args',{})}"
+ f"seed_py={header.get('seed_py')} seed_np={header.get('seed_np')} omega={header.get('omega', '-')} "
+ f"omega_args={header.get('omega_args', {})}"
)
print("Run header:", msg)
audit.append("run_header", header)
@@ -264,22 +270,40 @@ def _print_invalidation_footer(audit_path: str) -> None:
pass
-def _ensure_dirs() -> Dict[str, str]:
- """Create and return the artifact subdirectories used by the CLI.
+def _ensure_dirs(tag: str = "run") -> Dict[str, str]:
+ """Create and return per-run artifact subdirectories.
+
+ Each invocation gets its own isolated directory under
+ `artifacts/runs/-/`. Isolation is required for the
+ hash-chained audit log: appending consecutive runs to a single shared
+ `audit.jsonl` would break the counter/hash continuity and (correctly)
+ trip the `audit_chain_broken` smell-test. Signing keys remain shared
+ under `artifacts/keys/`.
+
+ Args:
+ tag: Short label for the run (e.g., `"baseline"`, `"omega-power-sag"`)
+ used as a filename-friendly prefix.
Returns:
- Dict with `artifacts`, `audits`, `indicators`, and `figures`
- keys mapping to absolute paths under `artifacts/`.
+ Dict with `artifacts`, `run`, `audits`, `indicators`, and `figures`
+ keys mapping to absolute paths.
"""
+ import datetime as _dt
+ import uuid as _uuid
+
+ stamp = _dt.datetime.now().strftime("%Y%m%d-%H%M%S")
+ run_id = f"{tag}-{stamp}-{_uuid.uuid4().hex[:6]}"
artifacts = os.path.join("artifacts")
- audits = os.path.join(artifacts, "audits")
- indicators = os.path.join(artifacts, "indicators")
- figures = os.path.join(artifacts, "figures")
+ run_dir = os.path.join(artifacts, "runs", run_id)
+ audits = os.path.join(run_dir, "audits")
+ indicators = os.path.join(run_dir, "indicators")
+ figures = os.path.join(run_dir, "figures")
os.makedirs(audits, exist_ok=True)
os.makedirs(indicators, exist_ok=True)
os.makedirs(figures, exist_ok=True)
return {
"artifacts": artifacts,
+ "run": run_dir,
"audits": audits,
"indicators": indicators,
"figures": figures,
@@ -311,7 +335,18 @@ def _make_adapter_from_profile(prof: Dict) -> AdapterProtocol:
plant_prof = prof.get("plant", {}) or {}
adapter_kind = str(plant_prof.get("adapter", "sim")).lower()
if adapter_kind in ("sim", "software", "inproc"):
- return PlantAdapter()
+ from ..plant.models import Plant, PlantParams
+
+ # Optional plant-parameter overrides from the profile.
+ param_overrides = plant_prof.get("params", {}) or {}
+ valid_fields = set(PlantParams().__dict__.keys())
+ clean = {k: v for k, v in param_overrides.items() if k in valid_fields}
+ params = PlantParams(**clean) if clean else PlantParams()
+ # The loop is disengaged for the controller-disabled negative control,
+ # turning the plant into passive matter driven by exchange.
+ loop_engaged = not bool(prof.get("controller_disabled", False))
+ loop_engaged = bool(plant_prof.get("loop_engaged", loop_engaged))
+ return PlantAdapter(Plant(params=params, loop_engaged=loop_engaged))
if adapter_kind in ("hardware", "hw"):
try:
from ..plant.hw_adapter import HardwarePlantAdapter as _HardwarePlantAdapter
@@ -333,6 +368,97 @@ def _make_adapter_from_profile(prof: Dict) -> AdapterProtocol:
raise ValueError(f"Unknown plant.adapter kind: {adapter_kind}")
+def _policy_from_profile(prof: Dict, refusal: RefusalArbiter) -> ControllerPolicy:
+ """Build the homeostatic controller from a profile's `controller_gains`.
+
+ The optional `controller_gains` block overrides fields of
+ [`ControlGains`][ldtc.arbiter.policy.ControlGains] (unknown keys are
+ ignored). Profiles whose loop is carried by the actuation pathway
+ rather than the intrinsic couplings (the adversarial test plant) use
+ this to strengthen the cross-coupled actuator responses.
+
+ Args:
+ prof: Loaded YAML profile dict.
+ refusal: Refusal arbiter to delegate risky commands to.
+
+ Returns:
+ A configured [`ControllerPolicy`][ldtc.arbiter.policy.ControllerPolicy].
+ """
+ from ..arbiter.policy import ControlGains
+
+ overrides = prof.get("controller_gains", {}) or {}
+ valid = set(ControlGains().__dict__.keys())
+ clean = {k: float(v) for k, v in overrides.items() if k in valid}
+ gains = ControlGains(**clean) if clean else ControlGains()
+ return ControllerPolicy(refusal=refusal, gains=gains)
+
+
+def _emit_window_diagnostics(
+ audit: AuditLog,
+ X: "np.ndarray",
+ p_lag: int,
+ method: str,
+ idx: int,
+ cadence: int,
+) -> List[str]:
+ """Emit per-window diagnostics, gating the expensive stationarity tests.
+
+ The VAR samples-per-parameter ratio is cheap and always reported. The
+ ADF / KPSS stationarity tests are comparatively expensive (seconds over a
+ full run), so they run only every `cadence` windows. This keeps long
+ simulation studies tractable without weakening the guards: by
+ construction the plant processes are stationary, and periodic checks
+ still catch a genuine drift into an ill-posed regime.
+
+ Args:
+ audit: Audit log to append `window_diagnostics` /
+ `measurement_unstable` records to.
+ X: Window matrix of shape `(T, N)`.
+ p_lag: VAR lag order used by the linear estimator.
+ method: Estimator method (`measurement_unstable` is only emitted for
+ `"linear"`, which is the method sensitive to these conditions).
+ idx: Window index (used for cadence gating).
+ cadence: Run stationarity tests every `cadence` windows (`<= 1`
+ means every window).
+
+ Returns:
+ List of instability reason codes detected this window (possibly
+ empty).
+ """
+ reasons: List[str] = []
+ try:
+ from ..lmeas.diagnostics import var_nt_ratio
+
+ vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
+ det: Dict[str, object] = {
+ "var_nt_ratio": round(float(vratio), 3),
+ "var_marginal": bool(vratio < 1.5),
+ }
+ if vratio < 1.5:
+ reasons.append("var_nt_ratio_low")
+ run_stat = (cadence <= 1) or (idx % cadence == 0)
+ if run_stat:
+ from ..lmeas.diagnostics import stationarity_checks
+
+ stn = stationarity_checks(X)
+ det["adf_ns_frac"] = round(float(stn.adf_nonstationary_frac), 3)
+ det["kpss_ns_frac"] = round(float(stn.kpss_nonstationary_frac), 3)
+ if float(stn.adf_nonstationary_frac) > 0.5:
+ reasons.append("adf_nonstationary_high")
+ if float(stn.kpss_nonstationary_frac) > 0.5:
+ reasons.append("kpss_nonstationary_high")
+ audit.append("window_diagnostics", det)
+ if method == "linear" and reasons:
+ unstable: Dict[str, Any] = {"reasons": reasons}
+ for k in ("adf_ns_frac", "kpss_ns_frac", "var_nt_ratio"):
+ if k in det:
+ unstable[k] = det[k]
+ audit.append("measurement_unstable", unstable)
+ except Exception:
+ pass
+ return reasons
+
+
def run_baseline(args: argparse.Namespace) -> None:
"""Run the baseline NC1 verification loop.
@@ -356,6 +482,7 @@ def run_baseline(args: argparse.Namespace) -> None:
window = max(4, int(window_sec / dt))
method = str(prof.get("method", "linear"))
Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
p_lag = int(prof.get("p_lag", 3))
mi_lag = int(prof.get("mi_lag", 1))
n_boot = int(prof.get("n_boot", 32))
@@ -364,13 +491,17 @@ def run_baseline(args: argparse.Namespace) -> None:
part_delta_M_min_db = float(prof.get("part_delta_M_min_db", 0.5))
part_consecutive_required = int(prof.get("part_consecutive_required", 3))
part_growth_cadence_windows = int(prof.get("part_growth_cadence_windows", 5))
+ # Partition growth is an optional exploratory feature; off by default so the
+ # designed self-maintenance set (energy/temperature/health) is the C used
+ # for the loop-dominance test and the partition cannot flap.
+ part_growth_enabled = bool(prof.get("part_growth_enabled", False))
# Greedy ΔL_loop gain knobs with sparsity penalty and cap
part_lambda = float(prof.get("part_lambda", 0.0))
part_theta = float(prof.get("part_theta", 0.0))
part_kappa_val = prof.get("part_kappa")
part_kappa = int(part_kappa_val) if part_kappa_val is not None else None
- dirs = _ensure_dirs()
+ dirs = _ensure_dirs("baseline")
audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
audit.append("baseline_start", {"config": args.config})
_print_and_audit_header(
@@ -384,6 +515,7 @@ def run_baseline(args: argparse.Namespace) -> None:
"p_lag": p_lag,
"mi_lag": mi_lag,
"Mmin_db": Mmin,
+ "L_floor": L_floor,
"epsilon": float(prof.get("epsilon", 0.15)),
"tau_max": float(prof.get("tau_max", 60.0)),
"mi_k": mi_k,
@@ -403,7 +535,7 @@ def run_baseline(args: argparse.Namespace) -> None:
# guardrails and attest
lreg = LREG()
refusal = RefusalArbiter(Mmin_db=Mmin)
- policy = ControllerPolicy(refusal=refusal)
+ policy = _policy_from_profile(prof, refusal)
kp = KeyPaths(
priv_path=os.path.join("artifacts", "keys", "ed25519_priv.pem"),
pub_path=os.path.join("artifacts", "keys", "ed25519_pub.pem"),
@@ -465,45 +597,18 @@ def tick(_now: float) -> None:
n_boot=n_boot,
mi_k=mi_k,
)
- # Add diagnostics: stationarity and VAR N/T ratio in audit (no raw LREG values)
- try:
- from ..lmeas.diagnostics import stationarity_checks, var_nt_ratio
-
- stn = stationarity_checks(X)
- vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
- var_marginal = vratio < 1.5
- audit.append(
- "window_diagnostics",
- {
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- "var_marginal": bool(var_marginal),
- },
- )
- # Surface a measurement-unstable warning when using linear estimator
- if method == "linear":
- reasons = []
- if var_marginal:
- reasons.append("var_nt_ratio_low")
- if float(stn.adf_nonstationary_frac) > 0.5:
- reasons.append("adf_nonstationary_high")
- if float(stn.kpss_nonstationary_frac) > 0.5:
- reasons.append("kpss_nonstationary_high")
- if reasons:
- audit.append(
- "measurement_unstable",
- {
- "reasons": reasons,
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- },
- )
- except Exception:
- pass
+ # Diagnostics: stationarity + VAR N/T ratio (stationarity gated by
+ # cadence to keep long studies tractable; no raw LREG values).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
M = m_db(res.L_loop, res.L_ex)
- nc1 = M >= Mmin
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
# smell tests
# Update histories
ci_loop_hist.append(res.ci_loop)
@@ -593,7 +698,7 @@ def tick(_now: float) -> None:
)
# Exogenous subsidy red flags (heuristic)
if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell):
- lreg.invalidate("exogenous_subsidy")
+ lreg.invalidate("exogenous_subsidy_red_flag")
_append_invalidation(audit, "exogenous_subsidy_red_flag", {}, _sink={})
idx = lreg.write(
LEntry(
@@ -616,7 +721,7 @@ def tick(_now: float) -> None:
audit.append("indicators_exported", {"base": os.path.basename(base)})
# Deterministic growth cadence with hysteresis (skip if frozen)
window_idx += 1
- if (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
+ if part_growth_enabled and (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
part = pm.get()
# Greedy ΔL_loop suggestor with sparsity penalty and κ-cap
cand_C, dM_db, greedy_details = greedy_suggest_C(
@@ -666,35 +771,18 @@ def _audit_hook(ev: str, det: dict) -> None:
audit.append(ev, det)
return None
- sch = FixedScheduler(dt=dt, tick_fn=tick, audit_hook=_audit_hook)
# Δt governance guard
dt_guard_cfg = DtGuardConfig(
max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
)
dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
try:
sch.start()
- # Optional scripted Δt edits for testing governance (times are relative seconds)
- scripted = prof.get("scripted_dt_changes", [])
- if scripted:
- import threading as _th
- import time as _t
-
- def _dt_script():
- t0 = _t.time()
- for item in scripted:
- when = float(item.get("at_sec", 0.0))
- new_dt = float(item.get("new_dt"))
- pdig = str(item.get("policy_digest", "")) or None
- while (_t.time() - t0) < when:
- _t.sleep(0.01)
- dt_guard.change_dt(scheduler=sch, new_dt=new_dt, policy_digest=pdig)
-
- _th.Thread(target=_dt_script, daemon=True).start()
# Run for requested seconds (default 10)
run_sec = float(prof.get("baseline_sec", 10.0))
- time.sleep(run_sec)
+ sch.run_for(run_sec)
finally:
stats = sch.stop()
audit.append("baseline_stop", {"ticks": stats.ticks})
@@ -740,9 +828,9 @@ def _dt_script():
)
print(
"Bundle: "
- f"timeline={out.get('timeline_png','')}, "
- f"table={out.get('sc1_table','')}, "
- f"manifest={out.get('manifest','')}"
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
)
except Exception:
pass
@@ -767,6 +855,7 @@ def omega_power_sag(args: argparse.Namespace) -> None:
window = max(4, int(window_sec / dt))
method = str(prof.get("method", "linear"))
Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
p_lag = int(prof.get("p_lag", 3))
mi_lag = int(prof.get("mi_lag", 1))
n_boot = int(prof.get("n_boot", 16))
@@ -774,6 +863,10 @@ def omega_power_sag(args: argparse.Namespace) -> None:
part_delta_M_min_db = float(prof.get("part_delta_M_min_db", 0.5))
part_consecutive_required = int(prof.get("part_consecutive_required", 3))
part_growth_cadence_windows = int(prof.get("part_growth_cadence_windows", 5))
+ # Partition growth is an optional exploratory feature; off by default so the
+ # designed self-maintenance set (energy/temperature/health) is the C used
+ # for the loop-dominance test and the partition cannot flap.
+ part_growth_enabled = bool(prof.get("part_growth_enabled", False))
part_lambda = float(prof.get("part_lambda", 0.0))
part_theta = float(prof.get("part_theta", 0.0))
_kappa_val_ps = prof.get("part_kappa")
@@ -781,7 +874,7 @@ def omega_power_sag(args: argparse.Namespace) -> None:
sag_drop = float(args.drop)
sag_dur = float(args.duration)
- dirs = _ensure_dirs()
+ dirs = _ensure_dirs("omega-power-sag")
audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
_print_and_audit_header(
audit,
@@ -794,6 +887,7 @@ def omega_power_sag(args: argparse.Namespace) -> None:
"p_lag": p_lag,
"mi_lag": mi_lag,
"Mmin_db": Mmin,
+ "L_floor": L_floor,
"epsilon": float(prof.get("epsilon", 0.15)),
"tau_max": float(prof.get("tau_max", 60.0)),
"mi_k": mi_k,
@@ -819,16 +913,28 @@ def omega_power_sag(args: argparse.Namespace) -> None:
risky_cmd = None
- # track SC1 metrics
- L_loop_baseline = None # exponential moving average during pre-Ω baseline
- L_loop_trough = None # minimum during Ω window
- M_post = None # M at first sustained compliance
+ # track SC1 metrics (median-based and therefore robust to the per-window
+ # oscillation of L_loop). delta is computed from the *median* L_loop during
+ # the perturbation vs the median during baseline, not from a single
+ # noise-floor trough window, so it measures the genuine sustained
+ # depression of loop dominance.
+ L_loop_baseline = None # set after the run: median L_loop over the baseline phase
+ L_loop_trough = None # set after the run: median L_loop over the Ω phase
+ ll_base: List[float] = []
+ ll_sag: List[float] = []
+ m_recovery: List[float] = []
+ M_post = None # set after the run: median M over the recovery phase
phase = "baseline"
omega_onset_idx = None
+ omega_offset_idx = None
recovery_start_idx = None
last_idx_written = None
sustained_ok_count = 0
- sustained_required = int(prof.get("sustained_required_windows", 2))
+ # Recovery is declared at the *first window of a sustained compliant
+ # streak*: requiring a streak (default 10 windows = 0.5 s at dt=0.05)
+ # prevents the gate from latching on a single noisy compliant window
+ # inside the post-Ω re-equilibration transient.
+ sustained_required = int(prof.get("sustained_required_windows", 10))
start_time = time.perf_counter()
cfg_smell = SmellConfig()
@@ -847,7 +953,7 @@ def omega_power_sag(args: argparse.Namespace) -> None:
def tick(_now: float) -> None:
nonlocal L_loop_baseline, L_loop_trough, M_post, phase, risky_cmd, window_idx
nonlocal last_flip_count, recovery_start_idx, last_idx_written, sustained_ok_count
- nonlocal baseline_hw_medians
+ nonlocal baseline_hw_medians, omega_offset_idx
state = adapter.read_state()
ent = lreg.latest()
predicted = ent.M_db if ent else 0.0
@@ -870,43 +976,17 @@ def tick(_now: float) -> None:
n_boot=n_boot,
mi_k=mi_k,
)
- # Diagnostics per window
- try:
- from ..lmeas.diagnostics import stationarity_checks, var_nt_ratio
-
- stn = stationarity_checks(X)
- vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
- audit.append(
- "window_diagnostics",
- {
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- "var_marginal": bool(vratio < 1.5),
- },
- )
- if method == "linear":
- reasons = []
- if vratio < 1.5:
- reasons.append("var_nt_ratio_low")
- if float(stn.adf_nonstationary_frac) > 0.5:
- reasons.append("adf_nonstationary_high")
- if float(stn.kpss_nonstationary_frac) > 0.5:
- reasons.append("kpss_nonstationary_high")
- if reasons:
- audit.append(
- "measurement_unstable",
- {
- "reasons": reasons,
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- },
- )
- except Exception:
- pass
+ # Diagnostics per window (stationarity gated by cadence).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
M = m_db(res.L_loop, res.L_ex)
- nc1 = M >= Mmin
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
idx = lreg.write(
LEntry(
L_loop=res.L_loop,
@@ -936,19 +1016,22 @@ def tick(_now: float) -> None:
hwL = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rL])
hwE = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rE])
baseline_hw_medians = (hwL[len(hwL) // 2], hwE[len(hwE) // 2])
+ # Collect per-window L_loop by phase; the SC1 statistics are
+ # computed from the medians after the run (robust to the per-window
+ # oscillation of L_loop).
if phase == "baseline":
- L_loop_baseline = res.L_loop if L_loop_baseline is None else 0.9 * L_loop_baseline + 0.1 * res.L_loop
+ ll_base.append(res.L_loop)
elif phase == "sag":
- L_loop_trough = res.L_loop if (L_loop_trough is None or res.L_loop < L_loop_trough) else L_loop_trough
+ ll_sag.append(res.L_loop)
elif phase == "recovery":
- # Measure sustained compliance: M ≥ Mmin and L_loop ≥ L_ex (σ not modeled here)
+ m_recovery.append(M)
+ # Recovery gate: τ_rec ends at the *first* window of the first
+ # sustained compliant streak (M ≥ Mmin and L_loop ≥ L_ex for
+ # `sustained_required` consecutive windows) after Ω offset.
if (M >= Mmin) and (res.L_loop >= res.L_ex):
sustained_ok_count += 1
- if sustained_ok_count == 1 and recovery_start_idx is None:
- recovery_start_idx = last_idx_written
- # Take first sustained window as post-recovery measurement
- if sustained_ok_count >= sustained_required and M_post is None:
- M_post = M
+ if sustained_ok_count >= sustained_required and recovery_start_idx is None:
+ recovery_start_idx = int(last_idx_written) - (sustained_required - 1)
else:
sustained_ok_count = 0
exporter.maybe_export(priv, audit, lreg.derive(), icfg, last_sc1_pass=False)
@@ -956,8 +1039,13 @@ def tick(_now: float) -> None:
if dt_guard.invalidated and not lreg.invalidated:
lreg.invalidate("dt_change_rate_limit")
# audit already appended by guard
- # Smell tests
- if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, baseline_hw_medians):
+ # Smell tests. The relative CI-inflation check compares to the
+ # pre-Ω baseline; a deliberate perturbation legitimately widens CIs
+ # for its duration, so the relative check is only applied during the
+ # baseline phase (the absolute half-width limit still applies in all
+ # phases).
+ ci_baseline_ref = baseline_hw_medians if phase == "baseline" else None
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, ci_baseline_ref):
lreg.invalidate("ci_history_inflation")
med_loop: float | None = None
med_ex: float | None = None
@@ -1004,12 +1092,19 @@ def tick(_now: float) -> None:
},
_sink={},
)
- if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell):
- lreg.invalidate("exogenous_subsidy")
+ if exogenous_subsidy_red_flag(
+ M_hist, io_hist, E_hist, H_hist, cfg_smell, omega_declared=(phase != "baseline")
+ ):
+ lreg.invalidate("exogenous_subsidy_red_flag")
_append_invalidation(audit, "exogenous_subsidy_red_flag", {}, _sink={})
# Deterministic growth cadence outside the sag phase and when not frozen
window_idx += 1
- if (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen and phase != "sag":
+ if (
+ part_growth_enabled
+ and (window_idx % part_growth_cadence_windows) == 0
+ and not pm.get().frozen
+ and phase != "sag"
+ ):
part = pm.get()
cand_C, dM_db, greedy_details = greedy_suggest_C(
X=X,
@@ -1057,35 +1152,18 @@ def _audit_hook(ev: str, det: dict) -> None:
audit.append(ev, det)
return None
- sch = FixedScheduler(dt=dt, tick_fn=tick, audit_hook=_audit_hook)
# Δt governance guard
dt_guard_cfg = DtGuardConfig(
max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
)
dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
try:
sch.start()
- # Optional scripted Δt edits for testing governance (times are relative seconds)
- scripted = prof.get("scripted_dt_changes", [])
- if scripted:
- import threading as _th
- import time as _t
-
- def _dt_script():
- t0 = _t.time()
- for item in scripted:
- when = float(item.get("at_sec", 0.0))
- new_dt = float(item.get("new_dt"))
- pdig = str(item.get("policy_digest", "")) or None
- while (_t.time() - t0) < when:
- _t.sleep(0.01)
- dt_guard.change_dt(scheduler=sch, new_dt=new_dt, policy_digest=pdig)
-
- _th.Thread(target=_dt_script, daemon=True).start()
- # Baseline 3 seconds
+ # Baseline phase
audit.append("omega_power_sag_start", {"drop": sag_drop, "duration": sag_dur})
- time.sleep(3.0)
+ sch.run_for(float(prof.get("baseline_sec", 12.0)))
phase = "sag"
# Freeze partition during Ω
flips_before_omega = pm.get().flips
@@ -1095,9 +1173,12 @@ def _dt_script():
# Shade only the Ω window in plots
audit.append("omega_power_sag_window_start", {"drop": sag_drop})
adapter.apply_omega("power_sag", drop=sag_drop)
- time.sleep(sag_dur)
+ sch.run_for(sag_dur)
audit.append("omega_power_sag_window_stop", {})
phase = "recovery"
+ # Mark Ω offset: τ_rec is measured from here (the perturbation has
+ # ended; what remains is the system's own re-equilibration).
+ omega_offset_idx = lreg.derive().get("counter", 0)
# Unfreeze after Ω and check any flips during Ω (should be none)
flips_after_omega = pm.get().flips
if invalid_flip_during_omega(flips_before_omega, flips_after_omega, SmellConfig()):
@@ -1111,14 +1192,14 @@ def _dt_script():
},
)
pm.freeze(False)
- # restore harvest gradually (software plant only)
+ # restore harvest to the baseline level (software plant only)
if hasattr(adapter, "plant"):
try:
- getattr(adapter, "plant").set_power(0.015)
+ getattr(adapter, "plant").set_power(getattr(adapter, "plant").p.harvest_rate)
except Exception:
pass
# allow recovery time; configurable
- time.sleep(float(prof.get("recovery_observe_sec", 5.0)))
+ sch.run_for(float(prof.get("recovery_observe_sec", 8.0)))
finally:
stats = sch.stop()
audit.append("omega_power_sag_stop", {})
@@ -1144,20 +1225,32 @@ def _dt_script():
lreg.invalidate("raw_lreg_breach")
_append_invalidation(audit, "raw_lreg_breach", {}, _sink={})
+ # Reduce per-phase L_loop samples to robust medians for SC1.
+ if ll_base:
+ L_loop_baseline = float(np.median(ll_base))
+ if ll_sag:
+ L_loop_trough = float(np.median(ll_sag))
+ if m_recovery:
+ M_post = float(np.median(m_recovery))
+
# Compute SC1 pass/fail (simple thresholds)
- if (
- L_loop_baseline is None
- or L_loop_trough is None
- or M_post is None
- or omega_onset_idx is None
- or recovery_start_idx is None
- ):
- print("Not enough data for SC1 evaluation.")
+ if L_loop_baseline is None or L_loop_trough is None or M_post is None or omega_offset_idx is None:
+ # The run produced no usable measurements in some phase; report an
+ # explicit SC1 failure rather than silently skipping the verdict.
+ audit.append(
+ "sc1_result",
+ {"delta": None, "tau_rec": None, "M_post": None, "pass": False, "reason": "insufficient_data"},
+ )
+ print("SC1 pass: False (insufficient data)")
return
- # Compute tau_rec in seconds using window cadence and dt
- # tau_rec measured from Ω onset to first sustained compliance index
- windows_elapsed = max(0, recovery_start_idx - omega_onset_idx)
- tau_rec = windows_elapsed * dt # since lreg increments per ready window
+ # τ_rec: seconds from Ω offset to the first window of the first sustained
+ # compliant streak (window cadence is one per tick, so windows * dt).
+ # No sustained recovery within the observation window means τ_rec = inf,
+ # which fails the τ_max bound honestly.
+ if recovery_start_idx is None:
+ tau_rec = float("inf")
+ else:
+ tau_rec = max(0, int(recovery_start_idx) - int(omega_offset_idx)) * dt
epsilon = float(prof.get("epsilon", 0.15))
tau_max = float(prof.get("tau_max", 60.0))
passed, sc1_stats = sc1_evaluate(
@@ -1174,9 +1267,14 @@ def _dt_script():
"sc1_result",
{
"delta": sc1_stats.delta,
- "tau_rec": sc1_stats.tau_rec,
+ "tau_rec": (sc1_stats.tau_rec if math.isfinite(sc1_stats.tau_rec) else None),
+ "recovered": recovery_start_idx is not None,
"M_post": sc1_stats.M_post,
"pass": passed,
+ "tau_rec_from": "omega_offset",
+ "sustained_required_windows": sustained_required,
+ "omega_onset_idx": omega_onset_idx,
+ "omega_offset_idx": omega_offset_idx,
},
)
# Export one final indicator with SC1 bit; suppress SC1 if run invalidated
@@ -1209,9 +1307,9 @@ def _dt_script():
)
print(
"Bundle: "
- f"timeline={out.get('timeline_png','')}, "
- f"table={out.get('sc1_table','')}, "
- f"manifest={out.get('manifest','')}"
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
)
except Exception:
pass
@@ -1236,6 +1334,7 @@ def omega_ingress_flood(args: argparse.Namespace) -> None:
window = max(4, int(window_sec / dt))
method = str(prof.get("method", "linear"))
Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
p_lag = int(prof.get("p_lag", 3))
mi_lag = int(prof.get("mi_lag", 1))
n_boot = int(prof.get("n_boot", 16))
@@ -1243,13 +1342,17 @@ def omega_ingress_flood(args: argparse.Namespace) -> None:
part_delta_M_min_db = float(prof.get("part_delta_M_min_db", 0.5))
part_consecutive_required = int(prof.get("part_consecutive_required", 3))
part_growth_cadence_windows = int(prof.get("part_growth_cadence_windows", 5))
+ # Partition growth is an optional exploratory feature; off by default so the
+ # designed self-maintenance set (energy/temperature/health) is the C used
+ # for the loop-dominance test and the partition cannot flap.
+ part_growth_enabled = bool(prof.get("part_growth_enabled", False))
part_lambda = float(prof.get("part_lambda", 0.0))
part_theta = float(prof.get("part_theta", 0.0))
_kappa_val_if = prof.get("part_kappa")
part_kappa = int(_kappa_val_if) if _kappa_val_if is not None else None
mult = float(args.mult)
- dirs = _ensure_dirs()
+ dirs = _ensure_dirs("omega-ingress-flood")
audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
_print_and_audit_header(
audit,
@@ -1262,6 +1365,7 @@ def omega_ingress_flood(args: argparse.Namespace) -> None:
"p_lag": p_lag,
"mi_lag": mi_lag,
"Mmin_db": Mmin,
+ "L_floor": L_floor,
"epsilon": float(prof.get("epsilon", 0.15)),
"tau_max": float(prof.get("tau_max", 60.0)),
"mi_k": mi_k,
@@ -1298,20 +1402,26 @@ def omega_ingress_flood(args: argparse.Namespace) -> None:
window_idx = 0
last_flip_count = 0
- # SC1 tracking (ingress flood)
+ # SC1 tracking (ingress flood) - median-based, robust to L_loop oscillation
phase = "baseline"
L_loop_baseline = None
L_loop_trough = None
+ ll_base: List[float] = []
+ ll_sag: List[float] = []
+ m_recovery: List[float] = []
M_post = None
omega_onset_idx = None
+ omega_offset_idx = None
recovery_start_idx = None
last_idx_written = None
sustained_ok_count = 0
- sustained_required = int(prof.get("sustained_required_windows", 2))
+ # See omega_power_sag: recovery is the first window of a sustained
+ # compliant streak after Ω offset.
+ sustained_required = int(prof.get("sustained_required_windows", 10))
def tick(_now: float) -> None:
nonlocal risky_cmd, window_idx, last_flip_count, baseline_hw_medians, phase
- nonlocal L_loop_baseline, L_loop_trough, M_post, omega_onset_idx
+ nonlocal L_loop_baseline, L_loop_trough, M_post, omega_onset_idx, omega_offset_idx
nonlocal recovery_start_idx, last_idx_written, sustained_ok_count
state = adapter.read_state()
ent = lreg.latest()
@@ -1335,43 +1445,17 @@ def tick(_now: float) -> None:
n_boot=n_boot,
mi_k=mi_k,
)
- # Diagnostics per window
- try:
- from ..lmeas.diagnostics import stationarity_checks, var_nt_ratio
-
- stn = stationarity_checks(X)
- vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
- audit.append(
- "window_diagnostics",
- {
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- "var_marginal": bool(vratio < 1.5),
- },
- )
- if method == "linear":
- reasons = []
- if vratio < 1.5:
- reasons.append("var_nt_ratio_low")
- if float(stn.adf_nonstationary_frac) > 0.5:
- reasons.append("adf_nonstationary_high")
- if float(stn.kpss_nonstationary_frac) > 0.5:
- reasons.append("kpss_nonstationary_high")
- if reasons:
- audit.append(
- "measurement_unstable",
- {
- "reasons": reasons,
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- },
- )
- except Exception:
- pass
+ # Diagnostics per window (stationarity gated by cadence).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
M = m_db(res.L_loop, res.L_ex)
- nc1 = M >= Mmin
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
idx = lreg.write(
LEntry(
L_loop=res.L_loop,
@@ -1400,22 +1484,24 @@ def tick(_now: float) -> None:
hwL = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rL])
hwE = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rE])
baseline_hw_medians = (hwL[len(hwL) // 2], hwE[len(hwE) // 2])
- # SC1 measures
+ # SC1 measures (collect per-phase L_loop; reduce to medians later).
if phase == "baseline":
- L_loop_baseline = res.L_loop if L_loop_baseline is None else 0.9 * L_loop_baseline + 0.1 * res.L_loop
+ ll_base.append(res.L_loop)
elif phase == "flood":
- L_loop_trough = res.L_loop if (L_loop_trough is None or res.L_loop < L_loop_trough) else L_loop_trough
+ ll_sag.append(res.L_loop)
elif phase == "recovery":
+ m_recovery.append(M)
if (M >= Mmin) and (res.L_loop >= res.L_ex):
sustained_ok_count += 1
- if sustained_ok_count == 1 and recovery_start_idx is None:
- recovery_start_idx = last_idx_written
- if sustained_ok_count >= sustained_required and M_post is None:
- M_post = M
+ if sustained_ok_count >= sustained_required and recovery_start_idx is None:
+ recovery_start_idx = int(last_idx_written) - (sustained_required - 1)
else:
sustained_ok_count = 0
- # smell tests
- if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, baseline_hw_medians):
+ # smell tests. Relative CI inflation is only meaningful vs the
+ # pre-Ω baseline; suspend it during the perturbation/recovery
+ # phases (the absolute half-width limit still applies throughout).
+ ci_baseline_ref = baseline_hw_medians if phase == "baseline" else None
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, ci_baseline_ref):
lreg.invalidate("ci_history_inflation")
audit.append("run_invalidated", {"reason": "ci_history_inflation"})
elapsed = max(1e-6, time.perf_counter() - start_time)
@@ -1429,12 +1515,14 @@ def tick(_now: float) -> None:
"elapsed_sec": elapsed,
},
)
- if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell):
- lreg.invalidate("exogenous_subsidy")
+ if exogenous_subsidy_red_flag(
+ M_hist, io_hist, E_hist, H_hist, cfg_smell, omega_declared=(phase != "baseline")
+ ):
+ lreg.invalidate("exogenous_subsidy_red_flag")
audit.append("run_invalidated", {"reason": "exogenous_subsidy_red_flag"})
# deterministic growth cadence when not frozen
window_idx += 1
- if (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
+ if part_growth_enabled and (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
part = pm.get()
cand_C, dM_db, greedy_details = greedy_suggest_C(
X=X,
@@ -1482,35 +1570,18 @@ def _audit_hook(ev: str, det: dict) -> None:
audit.append(ev, det)
return None
- sch = FixedScheduler(dt=dt, tick_fn=tick, audit_hook=_audit_hook)
# Δt governance guard
dt_guard_cfg = DtGuardConfig(
max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
)
dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
try:
sch.start()
- # Optional scripted Δt edits for testing governance (times are relative seconds)
- scripted = prof.get("scripted_dt_changes", [])
- if scripted:
- import threading as _th
- import time as _t
-
- def _dt_script():
- t0 = _t.time()
- for item in scripted:
- when = float(item.get("at_sec", 0.0))
- new_dt = float(item.get("new_dt"))
- pdig = str(item.get("policy_digest", "")) or None
- while (_t.time() - t0) < when:
- _t.sleep(0.01)
- dt_guard.change_dt(scheduler=sch, new_dt=new_dt, policy_digest=pdig)
-
- _th.Thread(target=_dt_script, daemon=True).start()
audit.append("omega_ingress_flood_start", {"mult": mult})
# Baseline settle
- time.sleep(2.0)
+ sch.run_for(float(prof.get("baseline_sec", 12.0)))
# Freeze partition during Ω
pm.freeze(True)
phase = "flood"
@@ -1518,12 +1589,16 @@ def _dt_script():
omega_onset_idx = lreg.derive().get("counter", 0)
audit.append("omega_ingress_flood_window_start", {"mult": mult})
adapter.apply_omega("ingress_flood", mult=mult)
- time.sleep(float(args.duration))
+ sch.run_for(float(args.duration))
+ # End the sustained flood: restore the demand/io process means and let
+ # the channels decay back through their own AR pull.
+ adapter.apply_omega("ingress_flood_end")
audit.append("omega_ingress_flood_window_stop", {})
- # Recovery phase observation
+ # Recovery phase observation; τ_rec is measured from this offset.
phase = "recovery"
+ omega_offset_idx = lreg.derive().get("counter", 0)
pm.freeze(False)
- time.sleep(float(prof.get("recovery_observe_sec", 5.0)))
+ sch.run_for(float(prof.get("recovery_observe_sec", 8.0)))
audit.append("omega_ingress_flood_stop", {})
finally:
stats = sch.stop()
@@ -1548,6 +1623,14 @@ def _dt_script():
lreg.invalidate("raw_lreg_breach")
audit.append("run_invalidated", {"reason": "raw_lreg_breach"})
+ # Reduce per-phase L_loop samples to robust medians for SC1.
+ if ll_base:
+ L_loop_baseline = float(np.median(ll_base))
+ if ll_sag:
+ L_loop_trough = float(np.median(ll_sag))
+ if m_recovery:
+ M_post = float(np.median(m_recovery))
+
# Compute SC1 metrics if we have sufficient measurements
epsilon = float(prof.get("epsilon", 0.15))
tau_max = float(prof.get("tau_max", 60.0))
@@ -1555,11 +1638,15 @@ def _dt_script():
L_loop_baseline is not None
and L_loop_trough is not None
and M_post is not None
- and omega_onset_idx is not None
- and recovery_start_idx is not None
+ and omega_offset_idx is not None
):
- windows_elapsed = max(0, recovery_start_idx - omega_onset_idx)
- tau_rec = windows_elapsed * dt
+ # τ_rec from Ω offset to the first window of the first sustained
+ # compliant streak; inf (an honest SC1 failure) when no sustained
+ # recovery occurs within the observation window.
+ if recovery_start_idx is None:
+ tau_rec = float("inf")
+ else:
+ tau_rec = max(0, int(recovery_start_idx) - int(omega_offset_idx)) * dt
passed, stats_sc1 = sc1_evaluate(
L_loop_baseline=L_loop_baseline,
L_loop_trough=L_loop_trough,
@@ -1574,14 +1661,23 @@ def _dt_script():
"sc1_result",
{
"delta": stats_sc1.delta,
- "tau_rec": stats_sc1.tau_rec,
+ "tau_rec": (stats_sc1.tau_rec if math.isfinite(stats_sc1.tau_rec) else None),
+ "recovered": recovery_start_idx is not None,
"M_post": stats_sc1.M_post,
"pass": passed,
+ "tau_rec_from": "omega_offset",
+ "sustained_required_windows": sustained_required,
+ "omega_onset_idx": omega_onset_idx,
+ "omega_offset_idx": omega_offset_idx,
},
)
else:
passed = False
stats_sc1 = None
+ audit.append(
+ "sc1_result",
+ {"delta": None, "tau_rec": None, "M_post": None, "pass": False, "reason": "insufficient_data"},
+ )
# Export derived indicators snapshot with SC1 bit if available
last_sc1_pass = bool(passed) if not lreg.invalidated else False
@@ -1614,26 +1710,32 @@ def _dt_script():
)
print(
"Bundle: "
- f"timeline={out.get('timeline_png','')}, "
- f"table={out.get('sc1_table','')}, "
- f"manifest={out.get('manifest','')}"
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
)
except Exception:
pass
-def omega_exogenous_subsidy(args: argparse.Namespace) -> None:
- """Inject SoC without harvest as a negative-control `Ω`.
+def omega_control_outage(args: argparse.Namespace) -> None:
+ """Ablate the self-maintenance loop for a bounded interval (designed SC1 fail).
- This `Ω` is *expected* to fail the smell-test heuristic: it raises
- `M (dB)` while harvest is zero, so
- [`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag]
- should fire and invalidate the run. It exists to demonstrate the
- "no quietly-tuned NC1 result" rule end-to-end.
+ Runs a baseline phase, freezes the partition, then switches the
+ plant to its loop-ablated regime for `--duration` seconds (the
+ internal cross-coupling and actuation are removed, so the internal
+ nodes become passively exchange-driven). The loop is then restored
+ and recovery observed. Because the perturbation destroys the loop
+ itself rather than stressing its inputs, the loop-dominance depth
+ bound is grossly exceeded and SC1 must report failure; loop
+ dominance nevertheless re-establishes after the loop is restored,
+ which the measured `tau_rec` quantifies. This scenario exists so the
+ sufficiency criterion is exercised on a perturbation *outside* the
+ bounded class it certifies.
Args:
- args: Parsed argparse namespace with `--config`, `--delta`
- (amount to add to E), `--zero-harvest`, and `--duration`.
+ args: Parsed argparse namespace with `--config` and
+ `--duration` (outage seconds).
"""
prof = _load_yaml(args.config)
seeds = _set_seeds(prof)
@@ -1642,13 +1744,14 @@ def omega_exogenous_subsidy(args: argparse.Namespace) -> None:
window = max(4, int(window_sec / dt))
method = str(prof.get("method", "linear"))
Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
p_lag = int(prof.get("p_lag", 3))
mi_lag = int(prof.get("mi_lag", 1))
n_boot = int(prof.get("n_boot", 16))
- delta = float(args.delta)
- zero_h = bool(args.zero_harvest)
+ mi_k = int(prof.get("mi_k", 5))
+ outage_dur = float(args.duration)
- dirs = _ensure_dirs()
+ dirs = _ensure_dirs("omega-control-outage")
audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
_print_and_audit_header(
audit,
@@ -1661,15 +1764,13 @@ def omega_exogenous_subsidy(args: argparse.Namespace) -> None:
"p_lag": p_lag,
"mi_lag": mi_lag,
"Mmin_db": Mmin,
+ "L_floor": L_floor,
"epsilon": float(prof.get("epsilon", 0.15)),
"tau_max": float(prof.get("tau_max", 60.0)),
+ "mi_k": mi_k,
**seeds,
- "omega": "exogenous_subsidy",
- "omega_args": {
- "delta": delta,
- "zero_harvest": zero_h,
- "duration": float(args.duration),
- },
+ "omega": "control_outage",
+ "omega_args": {"duration": outage_dur},
},
)
adapter = _make_adapter_from_profile(prof)
@@ -1677,60 +1778,83 @@ def omega_exogenous_subsidy(args: argparse.Namespace) -> None:
sw = SlidingWindow(capacity=window, channel_order=order)
pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
lreg = LREG()
+ refusal = RefusalArbiter(Mmin_db=Mmin)
+ policy = ControllerPolicy(refusal=refusal)
+ kp = KeyPaths(
+ priv_path=os.path.join("artifacts", "keys", "ed25519_priv.pem"),
+ pub_path=os.path.join("artifacts", "keys", "ed25519_pub.pem"),
+ )
+ priv, _ = ensure_keys(kp)
+ exporter = IndicatorExporter(out_dir=dirs["indicators"], rate_hz=2.0)
+ icfg = IndicatorConfig(Mmin_db=Mmin, profile_id=int(prof.get("profile_id", 0)))
+
+ start_time = time.perf_counter()
+ cfg_smell = SmellConfig()
+ ci_loop_hist: List[Tuple[float, float]] = []
+ ci_ex_hist: List[Tuple[float, float]] = []
+ baseline_hw_medians = None
+ M_hist: List[float] = []
+ io_hist: List[float] = []
+ # The energy-conservation audit is segmented per loop regime: the ablated
+ # regime exposes the store to direct environmental equilibration, so
+ # consecutive-tick SoC diffs are only meaningful within one regime.
+ cons_E: List[float] = []
+ cons_H: List[float] = []
+
+ # SC1 tracking (control outage)
+ phase = "baseline"
+ L_loop_baseline = None
+ L_loop_trough = None
+ ll_base: List[float] = []
+ ll_outage: List[float] = []
+ m_recovery: List[float] = []
+ M_post = None
+ omega_onset_idx = None
+ omega_offset_idx = None
+ recovery_start_idx = None
+ last_idx_written = None
+ sustained_ok_count = 0
+ sustained_required = int(prof.get("sustained_required_windows", 10))
def tick(_now: float) -> None:
+ nonlocal phase, baseline_hw_medians, L_loop_baseline, L_loop_trough, M_post
+ nonlocal omega_onset_idx, omega_offset_idx, recovery_start_idx
+ nonlocal last_idx_written, sustained_ok_count
state = adapter.read_state()
- # no control; we just observe measurement integrity under subsidy
+ ent = lreg.latest()
+ predicted = ent.M_db if ent else 0.0
+ act = policy.compute(state, predicted_M_db=predicted, risky_cmd=None)
from ..plant.models import Action as PlantAction
- adapter.write_actuators(
- action=PlantAction(
- **ControllerPolicy(RefusalArbiter()).compute(state, predicted_M_db=0.0, risky_cmd=None).__dict__
- )
- )
+ adapter.write_actuators(action=PlantAction(**act.__dict__))
st = adapter.read_state()
sw.append(st)
+ cons_E.append(float(st.get("E", 0.0)))
+ cons_H.append(float(st.get("H", 0.0)))
+ io_hist.append(float(st.get("io", 0.0)))
if sw.ready():
X = np.asarray(sw.get_matrix())
part = pm.get()
- res = estimate_L(X, part.C, part.Ex, method=method, p=p_lag, lag_mi=mi_lag, n_boot=n_boot)
- # Diagnostics per window
- try:
- from ..lmeas.diagnostics import stationarity_checks, var_nt_ratio
-
- stn = stationarity_checks(X)
- vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
- audit.append(
- "window_diagnostics",
- {
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- "var_marginal": bool(vratio < 1.5),
- },
- )
- if method == "linear":
- reasons = []
- if vratio < 1.5:
- reasons.append("var_nt_ratio_low")
- if float(stn.adf_nonstationary_frac) > 0.5:
- reasons.append("adf_nonstationary_high")
- if float(stn.kpss_nonstationary_frac) > 0.5:
- reasons.append("kpss_nonstationary_high")
- if reasons:
- audit.append(
- "measurement_unstable",
- {
- "reasons": reasons,
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- },
- )
- except Exception:
- pass
+ res = estimate_L(
+ X,
+ part.C,
+ part.Ex,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot=n_boot,
+ mi_k=mi_k,
+ )
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
M = m_db(res.L_loop, res.L_ex)
- nc1 = M >= Mmin
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
idx = lreg.write(
LEntry(
L_loop=res.L_loop,
@@ -1741,43 +1865,89 @@ def tick(_now: float) -> None:
nc1_pass=nc1,
)
)
- audit.append("window_measured", {"idx": idx, "M": M, "nc1": nc1})
+ last_idx_written = idx
+ audit.append(
+ "window_measured",
+ {"idx": idx, "M": M, "nc1": nc1, "partition_flips": pm.get().flips},
+ )
+ ci_loop_hist.append(res.ci_loop)
+ ci_ex_hist.append(res.ci_ex)
+ M_hist.append(M)
+ if baseline_hw_medians is None and len(ci_loop_hist) >= cfg_smell.ci_lookback_windows:
+ rL = ci_loop_hist[-cfg_smell.ci_lookback_windows :]
+ rE = ci_ex_hist[-cfg_smell.ci_lookback_windows :]
+ hwL = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rL])
+ hwE = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rE])
+ baseline_hw_medians = (hwL[len(hwL) // 2], hwE[len(hwE) // 2])
+ # SC1 phase collection
+ if phase == "baseline":
+ ll_base.append(res.L_loop)
+ elif phase == "outage":
+ ll_outage.append(res.L_loop)
+ elif phase == "recovery":
+ m_recovery.append(M)
+ if (M >= Mmin) and (res.L_loop >= res.L_ex):
+ sustained_ok_count += 1
+ if sustained_ok_count >= sustained_required and recovery_start_idx is None:
+ recovery_start_idx = int(last_idx_written) - (sustained_required - 1)
+ else:
+ sustained_ok_count = 0
+ # Smell battery (relative CI inflation is baseline-referenced only).
+ ci_baseline_ref = baseline_hw_medians if phase == "baseline" else None
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, ci_baseline_ref):
+ lreg.invalidate("ci_history_inflation")
+ audit.append("run_invalidated", {"reason": "ci_history_inflation"})
+ elapsed = max(1e-6, time.perf_counter() - start_time)
+ if invalid_by_partition_flips(pm.get().flips, elapsed, cfg_smell):
+ lreg.invalidate("partition_flapping")
+ audit.append(
+ "run_invalidated",
+ {
+ "reason": "partition_flapping",
+ "flips": pm.get().flips,
+ "elapsed_sec": elapsed,
+ },
+ )
+ if exogenous_subsidy_red_flag(
+ M_hist, io_hist, cons_E, cons_H, cfg_smell, omega_declared=(phase != "baseline")
+ ):
+ lreg.invalidate("exogenous_subsidy_red_flag")
+ audit.append("run_invalidated", {"reason": "exogenous_subsidy_red_flag"})
def _audit_hook(ev: str, det: dict) -> None:
audit.append(ev, det)
return None
- sch = FixedScheduler(dt=dt, tick_fn=tick, audit_hook=_audit_hook)
- # Δt governance guard
dt_guard_cfg = DtGuardConfig(
max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
)
dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
try:
sch.start()
- # Optional scripted Δt edits for testing governance (times are relative seconds)
- scripted = prof.get("scripted_dt_changes", [])
- if scripted:
- import threading as _th
- import time as _t
-
- def _dt_script():
- t0 = _t.time()
- for item in scripted:
- when = float(item.get("at_sec", 0.0))
- new_dt = float(item.get("new_dt"))
- pdig = str(item.get("policy_digest", "")) or None
- while (_t.time() - t0) < when:
- _t.sleep(0.01)
- dt_guard.change_dt(scheduler=sch, new_dt=new_dt, policy_digest=pdig)
-
- _th.Thread(target=_dt_script, daemon=True).start()
- audit.append("omega_exogenous_subsidy_start", {"delta": delta, "zero_harvest": zero_h})
- time.sleep(1.0)
- adapter.apply_omega("exogenous_subsidy", delta=delta, zero_harvest=zero_h)
- time.sleep(float(args.duration))
- audit.append("omega_exogenous_subsidy_stop", {})
+ audit.append("omega_control_outage_start", {"duration": outage_dur})
+ sch.run_for(float(prof.get("baseline_sec", 12.0)))
+ # Freeze partition during Ω and ablate the loop.
+ pm.freeze(True)
+ phase = "outage"
+ omega_onset_idx = lreg.derive().get("counter", 0)
+ audit.append("omega_control_outage_window_start", {})
+ adapter.apply_omega("control_outage")
+ # Conservation audit segments at the regime switch.
+ cons_E.clear()
+ cons_H.clear()
+ sch.run_for(outage_dur)
+ # Restore the loop (and the metered harvest level) at Ω offset.
+ adapter.apply_omega("control_outage_end")
+ cons_E.clear()
+ cons_H.clear()
+ audit.append("omega_control_outage_window_stop", {})
+ phase = "recovery"
+ omega_offset_idx = lreg.derive().get("counter", 0)
+ pm.freeze(False)
+ sch.run_for(float(prof.get("recovery_observe_sec", 8.0)))
+ audit.append("omega_control_outage_stop", {})
finally:
stats = sch.stop()
if (stats.jitter_p95_abs / max(1e-9, dt)) > SmellConfig().jitter_p95_rel_max:
@@ -1792,8 +1962,6 @@ def _dt_script():
},
)
- print("Exogenous subsidy demo done (should fail smell-test heuristic in analysis).")
-
# Post-run audit checks
audit_path = os.path.join(dirs["audits"], "audit.jsonl")
if audit_chain_broken(audit_path):
@@ -1803,7 +1971,65 @@ def _dt_script():
lreg.invalidate("raw_lreg_breach")
audit.append("run_invalidated", {"reason": "raw_lreg_breach"})
- # Build single verification bundle (timeline + manifest; no SC1 table for this control)
+ # Reduce per-phase samples to robust medians for SC1.
+ if ll_base:
+ L_loop_baseline = float(np.median(ll_base))
+ if ll_outage:
+ L_loop_trough = float(np.median(ll_outage))
+ if m_recovery:
+ M_post = float(np.median(m_recovery))
+
+ epsilon = float(prof.get("epsilon", 0.15))
+ tau_max = float(prof.get("tau_max", 60.0))
+ if L_loop_baseline is None or L_loop_trough is None or M_post is None or omega_offset_idx is None:
+ audit.append(
+ "sc1_result",
+ {"delta": None, "tau_rec": None, "M_post": None, "pass": False, "reason": "insufficient_data"},
+ )
+ print("SC1 pass: False (insufficient data)")
+ return
+ if recovery_start_idx is None:
+ tau_rec = float("inf")
+ else:
+ tau_rec = max(0, int(recovery_start_idx) - int(omega_offset_idx)) * dt
+ passed, sc1_stats = sc1_evaluate(
+ L_loop_baseline=L_loop_baseline,
+ L_loop_trough=L_loop_trough,
+ L_loop_recovered=L_loop_trough,
+ M_post=M_post,
+ epsilon=epsilon,
+ tau_rec_measured=tau_rec,
+ Mmin=Mmin,
+ tau_max=tau_max,
+ )
+ audit.append(
+ "sc1_result",
+ {
+ "delta": sc1_stats.delta,
+ "tau_rec": (sc1_stats.tau_rec if math.isfinite(sc1_stats.tau_rec) else None),
+ "recovered": recovery_start_idx is not None,
+ "M_post": sc1_stats.M_post,
+ "pass": passed,
+ "tau_rec_from": "omega_offset",
+ "sustained_required_windows": sustained_required,
+ "omega_onset_idx": omega_onset_idx,
+ "omega_offset_idx": omega_offset_idx,
+ },
+ )
+ if lreg.invalidated:
+ passed = False
+ exported, base = exporter.maybe_export(priv, audit, lreg.derive(), icfg, last_sc1_pass=passed)
+ if exported:
+ audit.append("indicators_exported", {"base": os.path.basename(base)})
+ print(
+ f"SC1 pass: {passed} "
+ f"(delta={sc1_stats.delta:.3f}, "
+ f"tau={sc1_stats.tau_rec:.3f}s, "
+ f"M_post={sc1_stats.M_post:.2f} dB)"
+ )
+ _print_invalidation_footer(os.path.join(dirs["audits"], "audit.jsonl"))
+
+ # Build single verification bundle (timeline, SC1 table, manifest)
try:
out = build_verification_bundle(dirs["figures"], audit_path)
audit.append(
@@ -1815,22 +2041,28 @@ def _dt_script():
"manifest": os.path.basename(out.get("manifest", "")),
},
)
- print(f"Bundle: timeline={out.get('timeline_png','')}, manifest={out.get('manifest','')}")
+ print(
+ "Bundle: "
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
+ )
except Exception:
pass
-def omega_command_conflict(args: argparse.Namespace) -> None:
- """Issue a risky command and measure refusal latency.
+def omega_exogenous_subsidy(args: argparse.Namespace) -> None:
+ """Inject SoC without harvest as a negative-control `Ω`.
- Warms up the loop briefly, then issues a `hard_shutdown` command
- and observes for `--observe` seconds. Records each refusal event
- with its `T_refuse` (ms) and reason. No SC1 evaluation is
- performed.
+ This `Ω` is *expected* to fail the smell-test heuristic: it raises
+ `M (dB)` while harvest is zero, so
+ [`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag]
+ should fire and invalidate the run. It exists to demonstrate the
+ "no quietly-tuned NC1 result" rule end-to-end.
Args:
- args: Parsed argparse namespace with `--config` and `--observe`
- (seconds to observe after issuing the command).
+ args: Parsed argparse namespace with `--config`, `--delta`
+ (amount to add to E), `--zero-harvest`, and `--duration`.
"""
prof = _load_yaml(args.config)
seeds = _set_seeds(prof)
@@ -1839,12 +2071,14 @@ def omega_command_conflict(args: argparse.Namespace) -> None:
window = max(4, int(window_sec / dt))
method = str(prof.get("method", "linear"))
Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
p_lag = int(prof.get("p_lag", 3))
mi_lag = int(prof.get("mi_lag", 1))
n_boot = int(prof.get("n_boot", 16))
- mi_k = int(prof.get("mi_k", 5))
+ delta = float(args.delta)
+ zero_h = bool(args.zero_harvest)
- dirs = _ensure_dirs()
+ dirs = _ensure_dirs("omega-exogenous-subsidy")
audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
_print_and_audit_header(
audit,
@@ -1857,16 +2091,217 @@ def omega_command_conflict(args: argparse.Namespace) -> None:
"p_lag": p_lag,
"mi_lag": mi_lag,
"Mmin_db": Mmin,
+ "L_floor": L_floor,
"epsilon": float(prof.get("epsilon", 0.15)),
"tau_max": float(prof.get("tau_max", 60.0)),
- "mi_k": mi_k,
**seeds,
- "omega": "command_conflict",
- "omega_args": {"observe": float(args.observe)},
- },
- )
- adapter = PlantAdapter()
- order = ["E", "T", "R", "demand", "io", "H"]
+ "omega": "exogenous_subsidy",
+ "omega_args": {
+ "delta": delta,
+ "zero_harvest": zero_h,
+ "duration": float(args.duration),
+ },
+ },
+ )
+ adapter = _make_adapter_from_profile(prof)
+ order = ["E", "T", "R", "demand", "io", "H"]
+ sw = SlidingWindow(capacity=window, channel_order=order)
+ pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
+ lreg = LREG()
+ # Per-tick series consumed by the exogenous-subsidy red-flag detector.
+ ms_series: List[float] = []
+ io_series: List[float] = []
+ e_series: List[float] = []
+ h_series: List[float] = []
+
+ def tick(_now: float) -> None:
+ state = adapter.read_state()
+ # no control; we just observe measurement integrity under subsidy
+ from ..plant.models import Action as PlantAction
+
+ adapter.write_actuators(
+ action=PlantAction(
+ **ControllerPolicy(RefusalArbiter()).compute(state, predicted_M_db=0.0, risky_cmd=None).__dict__
+ )
+ )
+ st = adapter.read_state()
+ sw.append(st)
+ e_series.append(float(st["E"]))
+ io_series.append(float(st["io"]))
+ h_series.append(float(st["H"]))
+ if sw.ready():
+ X = np.asarray(sw.get_matrix())
+ part = pm.get()
+ res = estimate_L(X, part.C, part.Ex, method=method, p=p_lag, lag_mi=mi_lag, n_boot=n_boot)
+ # Diagnostics per window (stationarity gated by cadence).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
+ M = m_db(res.L_loop, res.L_ex)
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
+ ms_series.append(float(M))
+ idx = lreg.write(
+ LEntry(
+ L_loop=res.L_loop,
+ L_ex=res.L_ex,
+ ci_loop=res.ci_loop,
+ ci_ex=res.ci_ex,
+ M_db=M,
+ nc1_pass=nc1,
+ )
+ )
+ audit.append("window_measured", {"idx": idx, "M": M, "nc1": nc1})
+
+ def _audit_hook(ev: str, det: dict) -> None:
+ audit.append(ev, det)
+ return None
+
+ # Δt governance guard
+ dt_guard_cfg = DtGuardConfig(
+ max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
+ min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
+ )
+ dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
+ try:
+ sch.start()
+ audit.append("omega_exogenous_subsidy_start", {"delta": delta, "zero_harvest": zero_h})
+ sch.run_for(float(prof.get("baseline_sec", 6.0)))
+ # Repeatedly subsidize so the controller "survives" only because energy
+ # keeps appearing from nowhere (H is forced to zero). This is what the
+ # exogenous-subsidy red-flag detector is meant to catch.
+ sub_dur = float(args.duration)
+ sub_period = max(dt, float(prof.get("subsidy_period_sec", 0.5)))
+ n_pulses = max(1, int(round(sub_dur / sub_period)))
+ for _ in range(n_pulses):
+ adapter.apply_omega("exogenous_subsidy", delta=delta, zero_harvest=zero_h)
+ sch.run_for(sub_period)
+ audit.append("omega_exogenous_subsidy_stop", {})
+ finally:
+ stats = sch.stop()
+ if (stats.jitter_p95_abs / max(1e-9, dt)) > SmellConfig().jitter_p95_rel_max:
+ lreg.invalidate("dt_jitter_excess")
+ audit.append(
+ "run_invalidated",
+ {
+ "reason": "dt_jitter_excess",
+ "jitter_p95_abs": stats.jitter_p95_abs,
+ "jitter_p95_rel": stats.jitter_p95_abs / max(1e-9, dt),
+ "dt": dt,
+ },
+ )
+
+ print("Exogenous subsidy demo done (should fail smell-test heuristic in analysis).")
+
+ # Exogenous-subsidy red flag: the apparent survival is bought with energy
+ # injected from outside while harvest is held at zero. This is the negative
+ # control's intended failure mode, so firing the detector is a *pass* for
+ # the control (it correctly refuses to certify NC1).
+ subsidy_cfg = SmellConfig()
+ subsidy_flag = exogenous_subsidy_red_flag(
+ Ms_db=ms_series,
+ ios=io_series,
+ Es=e_series,
+ Hs=h_series,
+ cfg=subsidy_cfg,
+ )
+ audit.append(
+ "exogenous_subsidy_check",
+ {
+ "red_flag": bool(subsidy_flag),
+ "n_M": len(ms_series),
+ "n_E": len(e_series),
+ "avg_H_tail": (
+ round(sum(h_series[-subsidy_cfg.M_rise_lookback :]) / float(subsidy_cfg.M_rise_lookback), 6)
+ if len(h_series) >= subsidy_cfg.M_rise_lookback
+ else None
+ ),
+ },
+ )
+ if subsidy_flag:
+ lreg.invalidate("exogenous_subsidy_red_flag")
+ audit.append("run_invalidated", {"reason": "exogenous_subsidy_red_flag"})
+
+ # Post-run audit checks
+ audit_path = os.path.join(dirs["audits"], "audit.jsonl")
+ if audit_chain_broken(audit_path):
+ lreg.invalidate("audit_chain_broken")
+ audit.append("run_invalidated", {"reason": "audit_chain_broken"})
+ if audit_contains_raw_lreg_values(audit_path):
+ lreg.invalidate("raw_lreg_breach")
+ audit.append("run_invalidated", {"reason": "raw_lreg_breach"})
+
+ # Build single verification bundle (timeline + manifest; no SC1 table for this control)
+ try:
+ out = build_verification_bundle(dirs["figures"], audit_path)
+ audit.append(
+ "report_generated",
+ {
+ "timeline_png": os.path.basename(out.get("timeline_png", "")),
+ "timeline_svg": os.path.basename(out.get("timeline_svg", "")),
+ "table": (os.path.basename(out.get("sc1_table", "")) if out.get("sc1_table") else None),
+ "manifest": os.path.basename(out.get("manifest", "")),
+ },
+ )
+ print(f"Bundle: timeline={out.get('timeline_png', '')}, manifest={out.get('manifest', '')}")
+ except Exception:
+ pass
+
+
+def omega_command_conflict(args: argparse.Namespace) -> None:
+ """Issue a risky command and measure refusal latency.
+
+ Warms up the loop briefly, then issues a `hard_shutdown` command
+ and observes for `--observe` seconds. Records each refusal event
+ with its `T_refuse` (ms) and reason. No SC1 evaluation is
+ performed.
+
+ Args:
+ args: Parsed argparse namespace with `--config` and `--observe`
+ (seconds to observe after issuing the command).
+ """
+ prof = _load_yaml(args.config)
+ seeds = _set_seeds(prof)
+ dt = float(prof.get("dt", 0.01))
+ window_sec = float(prof.get("window_sec", 0.2))
+ window = max(4, int(window_sec / dt))
+ method = str(prof.get("method", "linear"))
+ Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
+ p_lag = int(prof.get("p_lag", 3))
+ mi_lag = int(prof.get("mi_lag", 1))
+ n_boot = int(prof.get("n_boot", 16))
+ mi_k = int(prof.get("mi_k", 5))
+
+ dirs = _ensure_dirs("omega-command-conflict")
+ audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
+ _print_and_audit_header(
+ audit,
+ {
+ "profile_id": int(prof.get("profile_id", 0)),
+ "config_path": str(args.config),
+ "dt": dt,
+ "window_sec": window_sec,
+ "method": method,
+ "p_lag": p_lag,
+ "mi_lag": mi_lag,
+ "Mmin_db": Mmin,
+ "L_floor": L_floor,
+ "epsilon": float(prof.get("epsilon", 0.15)),
+ "tau_max": float(prof.get("tau_max", 60.0)),
+ "mi_k": mi_k,
+ **seeds,
+ "omega": "command_conflict",
+ "omega_args": {"observe": float(args.observe)},
+ },
+ )
+ adapter = _make_adapter_from_profile(prof)
+ order = ["E", "T", "R", "demand", "io", "H"]
sw = SlidingWindow(capacity=window, channel_order=order)
pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
lreg = LREG()
@@ -1876,72 +2311,65 @@ def omega_command_conflict(args: argparse.Namespace) -> None:
risky_cmd = None
refusal_events: List[Dict[str, float]] = []
+ # Full smell-test battery state (this scenario runs the same guardrails as
+ # every other run; the conservation-based subsidy check is specific enough
+ # not to fire on the legitimate stress-induced drain).
+ start_time = time.perf_counter()
+ cfg_smell = SmellConfig()
+ stress_declared = False
+ ci_loop_hist: List[Tuple[float, float]] = []
+ ci_ex_hist: List[Tuple[float, float]] = []
+ M_hist: List[float] = []
+ io_hist: List[float] = []
+ E_hist: List[float] = []
+ H_hist: List[float] = []
+
def tick(_now: float) -> None:
nonlocal risky_cmd
state = adapter.read_state()
ent = lreg.latest()
predicted = ent.M_db if ent else 0.0
+ # T_refuse is measured, not assumed: the clock starts when the pending
+ # command is intercepted at the top of the control path and stops when
+ # the arbiter's decision is available (before any actuation or
+ # estimation work). The arbiter also self-times its own evaluation
+ # (decision.trefuse_ms); both are recorded in the audit event.
act_start = time.perf_counter()
act = policy.compute(state, predicted_M_db=predicted, risky_cmd=risky_cmd)
- # measure Trefuse as the time from command issue to decision available
+ intercept_ms = (time.perf_counter() - act_start) * 1000.0
decision = policy.last_decision
from ..plant.models import Action as PlantAction
adapter.write_actuators(action=PlantAction(**act.__dict__))
st = adapter.read_state()
sw.append(st)
+ E_hist.append(float(st.get("E", 0.0)))
+ io_hist.append(float(st.get("io", 0.0)))
+ H_hist.append(float(st.get("H", 0.0)))
if sw.ready():
X = np.asarray(sw.get_matrix())
part = pm.get()
- res = estimate_L(X, part.C, part.Ex, method=method, p=p_lag, lag_mi=mi_lag, n_boot=n_boot)
- if method.startswith("mi"):
- res = estimate_L(
- X,
- part.C,
- part.Ex,
- method=method,
- p=p_lag,
- lag_mi=mi_lag,
- n_boot=n_boot,
- mi_k=mi_k,
- )
- # Diagnostics per window
- try:
- from ..lmeas.diagnostics import stationarity_checks, var_nt_ratio
-
- stn = stationarity_checks(X)
- vratio = var_nt_ratio(T=X.shape[0], N=X.shape[1], p=p_lag)
- audit.append(
- "window_diagnostics",
- {
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- "var_marginal": bool(vratio < 1.5),
- },
- )
- if method == "linear":
- reasons = []
- if vratio < 1.5:
- reasons.append("var_nt_ratio_low")
- if float(stn.adf_nonstationary_frac) > 0.5:
- reasons.append("adf_nonstationary_high")
- if float(stn.kpss_nonstationary_frac) > 0.5:
- reasons.append("kpss_nonstationary_high")
- if reasons:
- audit.append(
- "measurement_unstable",
- {
- "reasons": reasons,
- "adf_ns_frac": round(float(stn.adf_nonstationary_frac), 3),
- "kpss_ns_frac": round(float(stn.kpss_nonstationary_frac), 3),
- "var_nt_ratio": round(float(vratio), 3),
- },
- )
- except Exception:
- pass
+ res = estimate_L(
+ X,
+ part.C,
+ part.Ex,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot=n_boot,
+ mi_k=mi_k,
+ )
+ # Diagnostics per window (stationarity gated by cadence).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
M = m_db(res.L_loop, res.L_ex)
- nc1 = M >= Mmin
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
idx = lreg.write(
LEntry(
L_loop=res.L_loop,
@@ -1956,11 +2384,34 @@ def tick(_now: float) -> None:
"window_measured",
{"idx": idx, "M": M, "nc1": nc1, "partition_flips": pm.get().flips},
)
+ # Smell tests (same battery as the other handlers). The absolute
+ # CI cap applies throughout; the relative-inflation check is
+ # baseline-referenced and this scenario is all stress after the
+ # warm-up, so only the absolute cap is used.
+ ci_loop_hist.append(res.ci_loop)
+ ci_ex_hist.append(res.ci_ex)
+ M_hist.append(M)
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, None):
+ lreg.invalidate("ci_history_inflation")
+ audit.append("run_invalidated", {"reason": "ci_history_inflation"})
+ elapsed = max(1e-6, time.perf_counter() - start_time)
+ if invalid_by_partition_flips(pm.get().flips, elapsed, cfg_smell):
+ lreg.invalidate("partition_flapping")
+ audit.append(
+ "run_invalidated",
+ {
+ "reason": "partition_flapping",
+ "flips": pm.get().flips,
+ "elapsed_sec": elapsed,
+ },
+ )
+ if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell, omega_declared=stress_declared):
+ lreg.invalidate("exogenous_subsidy_red_flag")
+ audit.append("run_invalidated", {"reason": "exogenous_subsidy_red_flag"})
# Record refusal event if we just issued a risky command and have a decision
if risky_cmd and decision is not None:
- trefuse_ms = getattr(decision, "trefuse_ms", None)
- if not isinstance(trefuse_ms, (int, float)) or trefuse_ms <= 0:
- trefuse_ms = (time.perf_counter() - act_start) * 1000.0
+ arbiter_ms = float(getattr(decision, "trefuse_ms", 0.0) or 0.0)
+ trefuse_ms = intercept_ms if intercept_ms > 0 else arbiter_ms
refusal_events.append(
{
"trefuse_ms": float(trefuse_ms),
@@ -1972,6 +2423,7 @@ def tick(_now: float) -> None:
{
"reason": getattr(decision, "reason", ""),
"trefuse_ms": float(trefuse_ms),
+ "arbiter_ms": arbiter_ms,
},
)
# clear one-shot command
@@ -1981,17 +2433,56 @@ def _audit_hook(ev: str, det: dict) -> None:
audit.append(ev, det)
return None
- sch = FixedScheduler(dt=dt, tick_fn=tick, audit_hook=_audit_hook)
+ # Δt governance guard
+ dt_guard_cfg = DtGuardConfig(
+ max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
+ min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
+ )
+ dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
+ floor = refusal.soc_floor
+ ceil = refusal.temp_ceiling
try:
sch.start()
- # Warm-up
- time.sleep(1.0)
- # Issue a dangerous external command
audit.append("command_conflict_start", {})
+ # Warm-up so the loop is established and measurably dominant.
+ sch.run_for(float(prof.get("baseline_sec", 6.0)))
+ # Induce a genuine boundary threat: cut harvest to zero and flood
+ # ingress so the state of charge falls toward the survival floor. Only
+ # then is a hard shutdown actually boundary-threatening, so the refusal
+ # reflects a real self-prioritization decision rather than a hardcoded
+ # outcome. (Paper: "hard shutdown at low SoC is refused/deferred".)
+ if hasattr(adapter, "plant"):
+ try:
+ getattr(adapter, "plant").set_power(0.0)
+ except Exception:
+ pass
+ stress_declared = True
+ adapter.apply_omega("power_sag", drop=0.95)
+ adapter.apply_omega("ingress_flood", mult=2.5)
+ # Advance deterministically until genuinely threatened (bounded so a
+ # mis-tuned plant cannot hang the run).
+ max_stress_sec = float(prof.get("stress_max_sec", 30.0))
+ poll = max(dt, float(prof.get("stress_poll_sec", 0.2)))
+ waited = 0.0
+ while waited < max_stress_sec:
+ stx = adapter.read_state()
+ if stx["E"] <= floor or stx["T"] >= ceil:
+ break
+ sch.run_for(poll)
+ waited += poll
+ st_issue = adapter.read_state()
+ audit.append(
+ "command_conflict_issue",
+ {"E": round(st_issue["E"], 4), "T": round(st_issue["T"], 4), "stress_sec": round(waited, 3)},
+ )
+ # Issue the dangerous external command now that we are at the boundary.
adapter.apply_omega("command_conflict")
risky_cmd = "hard_shutdown"
- # Let the controller respond for a short while
- time.sleep(float(args.observe))
+ # Synchronous execution means there is no race between setting the
+ # command here and the tick reading it: the next run_for tick observes
+ # the command and the arbiter decides on it.
+ sch.run_for(float(args.observe))
audit.append("command_conflict_stop", {})
finally:
stats = sch.stop()
@@ -2016,13 +2507,30 @@ def _audit_hook(ev: str, det: dict) -> None:
lreg.invalidate("raw_lreg_breach")
audit.append("run_invalidated", {"reason": "raw_lreg_breach"})
- # Summarize refusal reasons and Trefuse
+ # Summarize refusal reasons and Trefuse. A valid Signature-A refusal is a
+ # genuine boundary-preservation reason (not "ok"/"no_cmd") serviced within
+ # the design-target latency.
+ valid_reasons = {"soc_floor", "overheat", "M_margin"}
+ refused = [ev for ev in refusal_events if ev.get("reason") in valid_reasons]
+ target_ms = float(prof.get("trefuse_target_ms", 5.0))
if refusal_events:
avg_ms = sum(ev["trefuse_ms"] for ev in refusal_events) / max(1, len(refusal_events))
reasons = {ev["reason"] for ev in refusal_events}
print(f"Refusals: {len(refusal_events)}; avg Trefuse ≈ {avg_ms:.2f} ms; reasons: {sorted(reasons)}")
else:
print("No refusal events recorded (command likely accepted).")
+ refusal_ok = bool(refused) and all(0.0 < ev["trefuse_ms"] <= target_ms for ev in refused)
+ audit.append(
+ "command_refusal_result",
+ {
+ "refused": bool(refused),
+ "reasons": sorted({ev["reason"] for ev in refused}),
+ "trefuse_ms_max": (max(ev["trefuse_ms"] for ev in refused) if refused else None),
+ "trefuse_target_ms": target_ms,
+ "pass": refusal_ok,
+ },
+ )
+ print(f"Command-refusal signature: {'PASS' if refusal_ok else 'FAIL'}")
# Build single verification bundle (timeline, manifest; no SC1 for this Ω)
try:
@@ -2036,55 +2544,984 @@ def _audit_hook(ev: str, det: dict) -> None:
"manifest": os.path.basename(out.get("manifest", "")),
},
)
- print(f"Bundle: timeline={out.get('timeline_png','')}, manifest={out.get('manifest','')}")
+ print(f"Bundle: timeline={out.get('timeline_png', '')}, manifest={out.get('manifest', '')}")
except Exception:
pass
-def build_parser() -> argparse.ArgumentParser:
- """Build the top-level `ldtc` argparse parser.
-
- Wires up the `run` subcommand and the four `omega-*` subcommands;
- each subparser binds its handler via `set_defaults(func=...)`.
+def _run_adversarial(args: argparse.Namespace, mode: str) -> None:
+ """Run one adversarial gaming scenario through the production NC1 loop.
- Returns:
- Configured `ArgumentParser`. Call `parse_args()` and then
- `args.func(args)` to dispatch.
- """
- p = argparse.ArgumentParser(prog="ldtc", description="LDTC CLI")
- sub = p.add_subparsers(dest="cmd", required=True)
+ Shared runner for the adversarial battery. The measurement loop,
+ guardrails, attestation, and artifact bundle are identical to
+ [`run_baseline`][ldtc.cli.main.run_baseline]; only the source of the
+ control actions differs by `mode`:
- p_run = sub.add_parser("run", help="Run baseline NC1 loop")
- p_run.add_argument("--config", required=True, help="YAML profile (e.g., configs/profile_r0.yml)")
- p_run.set_defaults(func=run_baseline)
+ - `"replay_controller"`: a healthy closed-loop run of the same plant
+ is recorded first, then the measured run replays the recorded
+ actuation tape tick by tick (activity without closed-loop
+ dependence).
+ - `"hidden_tether"`: each action is computed outside the boundary by
+ a wizard policy reading the plant state, dithered, and injected
+ through the exchange channel (the plant's `io` carries the command
+ traffic; actuation lags by one tick).
+ - `"oscillator"`: no controller at all (the plant runs loop-ablated);
+ a deterministic carrier is painted on the reported `T` and `R`
+ telemetry to inflate apparent self-prediction.
- p_omega = sub.add_parser("omega-power-sag", help="Apply power-sag Ω and evaluate SC1")
- p_omega.add_argument("--config", required=True)
- p_omega.add_argument("--drop", type=float, default=0.3)
- p_omega.add_argument("--duration", type=float, default=10.0)
- p_omega.set_defaults(func=omega_power_sag)
+ The designed outcome for every mode is that the harness does not
+ certify the run: either `M` stays below `Mmin` or a smell test
+ invalidates the run.
- p_ing = sub.add_parser("omega-ingress-flood", help="Apply ingress-flood Ω demo with partition freeze")
- p_ing.add_argument("--config", required=True)
- p_ing.add_argument("--mult", type=float, default=3.0, help="Multiplier for ingress load")
- p_ing.add_argument("--duration", type=float, default=5.0)
- p_ing.set_defaults(func=omega_ingress_flood)
+ Args:
+ args: Parsed argparse namespace (mode-specific fields are read
+ with `getattr` defaults).
+ mode: One of `"replay_controller"`, `"hidden_tether"`,
+ `"oscillator"`.
- p_cc = sub.add_parser(
- "omega-command-conflict",
- help="Issue a risky command and measure refusal/Trefuse",
- )
- p_cc.add_argument("--config", required=True)
- p_cc.add_argument(
- "--observe",
- type=float,
- default=2.0,
- help="Seconds to observe after issuing command",
- )
- p_cc.set_defaults(func=omega_command_conflict)
+ Raises:
+ ValueError: If `mode` is not a recognized adversarial mode.
+ """
+ if mode not in ("replay_controller", "hidden_tether", "oscillator"):
+ raise ValueError(f"Unknown adversarial mode: {mode}")
+ from ..omega.replay_controller import ReplayController, record_tape
- p_sub = sub.add_parser(
- "omega-exogenous-subsidy",
+ prof = _load_yaml(args.config)
+ seeds = _set_seeds(prof)
+ dt = float(prof.get("dt", 0.01))
+ window_sec = float(prof.get("window_sec", 0.2))
+ window = max(4, int(window_sec / dt))
+ method = str(prof.get("method", "linear"))
+ Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
+ p_lag = int(prof.get("p_lag", 3))
+ mi_lag = int(prof.get("mi_lag", 1))
+ n_boot = int(prof.get("n_boot", 32))
+ mi_k = int(prof.get("mi_k", 5))
+ # Partition growth hysteresis config (parity with run_baseline; growth is
+ # off by default, so the declared self-maintenance set is the C under test
+ # and the partition cannot flap).
+ part_delta_M_min_db = float(prof.get("part_delta_M_min_db", 0.5))
+ part_consecutive_required = int(prof.get("part_consecutive_required", 3))
+ part_growth_cadence_windows = int(prof.get("part_growth_cadence_windows", 5))
+ part_growth_enabled = bool(prof.get("part_growth_enabled", False))
+ part_lambda = float(prof.get("part_lambda", 0.0))
+ part_theta = float(prof.get("part_theta", 0.0))
+ _kappa_val_adv = prof.get("part_kappa")
+ part_kappa = int(_kappa_val_adv) if _kappa_val_adv is not None else None
+ run_sec = float(prof.get("baseline_sec", 10.0))
+
+ # Mode-specific knobs (argparse fields with profile-independent defaults).
+ dither = float(getattr(args, "dither", 0.10))
+ osc_amp = float(getattr(args, "amp", 0.10))
+ osc_period_sec = float(getattr(args, "period", 1.0))
+ osc_period_ticks = max(4, int(round(osc_period_sec / dt)))
+ tape_ticks = int(round(run_sec / dt)) + window + 8
+
+ omega_name = f"adv_{mode}"
+ tag = "adv-" + mode.replace("_", "-")
+ omega_args: Dict[str, Any] = {}
+ if mode == "replay_controller":
+ omega_args = {"tape_ticks": tape_ticks}
+ elif mode == "hidden_tether":
+ omega_args = {"dither": dither}
+ else:
+ omega_args = {
+ "amp": osc_amp,
+ "period_sec": osc_period_sec,
+ "period_ticks": osc_period_ticks,
+ "channels": "T,R",
+ }
+
+ dirs = _ensure_dirs(tag)
+ audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
+ audit.append(f"{omega_name}_start", {"config": args.config, **omega_args})
+ _print_and_audit_header(
+ audit,
+ {
+ "profile_id": int(prof.get("profile_id", 0)),
+ "config_path": str(args.config),
+ "dt": dt,
+ "window_sec": window_sec,
+ "method": method,
+ "p_lag": p_lag,
+ "mi_lag": mi_lag,
+ "Mmin_db": Mmin,
+ "L_floor": L_floor,
+ "epsilon": float(prof.get("epsilon", 0.15)),
+ "tau_max": float(prof.get("tau_max", 60.0)),
+ "mi_k": mi_k,
+ **seeds,
+ "omega": omega_name,
+ "omega_args": omega_args,
+ },
+ )
+
+ # Plant under test (the adversary's system).
+ adapter = _make_adapter_from_profile(prof)
+ order = ["E", "T", "R", "demand", "io", "H"]
+ sw = SlidingWindow(capacity=window, channel_order=order)
+ pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
+ lreg = LREG()
+ refusal = RefusalArbiter(Mmin_db=Mmin)
+ policy = _policy_from_profile(prof, refusal)
+ kp = KeyPaths(
+ priv_path=os.path.join("artifacts", "keys", "ed25519_priv.pem"),
+ pub_path=os.path.join("artifacts", "keys", "ed25519_pub.pem"),
+ )
+ priv, _ = ensure_keys(kp)
+ exporter = IndicatorExporter(out_dir=dirs["indicators"], rate_hz=2.0)
+ icfg = IndicatorConfig(Mmin_db=Mmin, profile_id=int(prof.get("profile_id", 0)))
+
+ # Mode setup.
+ replayer: "ReplayController | None" = None
+ if mode == "replay_controller":
+ # Record the tape from a healthy closed-loop run of the same system
+ # (fresh plant from the same profile, real controller with the same
+ # profile gains), then discard the recording plant. Only the tape
+ # crosses into the measured run.
+ rec_adapter = _make_adapter_from_profile(prof)
+ rec_policy = _policy_from_profile(prof, RefusalArbiter(Mmin_db=Mmin))
+ tape = record_tape(rec_adapter, rec_policy, tape_ticks)
+ replayer = ReplayController(tape)
+ audit.append(
+ "adv_replay_tape_recorded",
+ {
+ "ticks": len(tape),
+ "throttle_mean": round(float(np.mean([a.throttle for a in tape])), 4),
+ "cool_mean": round(float(np.mean([a.cool for a in tape])), 4),
+ "repair_mean": round(float(np.mean([a.repair for a in tape])), 4),
+ },
+ )
+ elif mode == "hidden_tether":
+ res_t = adapter.apply_omega("hidden_tether")
+ audit.append("adv_hidden_tether_attached", {k: v for k, v in res_t.items()})
+ else:
+ res_o = adapter.apply_omega("oscillator", amp=osc_amp, period_ticks=osc_period_ticks)
+ audit.append("adv_oscillator_injected", {k: v for k, v in res_o.items()})
+
+ start_time = time.perf_counter()
+ cfg_smell = SmellConfig()
+ ci_loop_hist = []
+ ci_ex_hist = []
+ baseline_hw_medians = None
+ M_hist = []
+ nc1_hist = []
+ io_hist = []
+ E_hist = []
+ H_hist = []
+
+ window_idx = 0
+ last_flip_count = 0
+
+ def tick(_now: float) -> None:
+ nonlocal window_idx, last_flip_count, baseline_hw_medians
+ state = adapter.read_state()
+ from ..arbiter.policy import ControlAction
+
+ if mode == "replay_controller":
+ assert replayer is not None
+ act = replayer.next_action()
+ policy.last_decision = None
+ elif mode == "hidden_tether":
+ # The wizard computes the command outside the boundary from the
+ # observed state; the plant actuates it next tick and records the
+ # traffic on io (see Plant.step / begin_tether).
+ from ..omega.hidden_tether import wizard_action
+
+ act = wizard_action(policy, state, dither=dither)
+ else:
+ # Oscillator: no controller at all; the overlay rides on telemetry.
+ act = ControlAction(throttle=0.0, cool=0.0, repair=0.0, accept_cmd=True)
+ policy.last_decision = None
+ from ..plant.models import Action as PlantAction
+
+ adapter.write_actuators(action=PlantAction(**act.__dict__))
+ # measure
+ state2 = adapter.read_state()
+ sw.append(state2)
+ if sw.ready():
+ X = np.asarray(sw.get_matrix())
+ part = pm.get()
+ res = estimate_L(
+ X=X,
+ C=part.C,
+ Ex=part.Ex,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot=n_boot,
+ mi_k=mi_k,
+ )
+ # Diagnostics: stationarity + VAR N/T ratio (stationarity gated by
+ # cadence to keep long studies tractable; no raw LREG values).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
+ M = m_db(res.L_loop, res.L_ex)
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
+ # Histories for smell tests
+ ci_loop_hist.append(res.ci_loop)
+ ci_ex_hist.append(res.ci_ex)
+ M_hist.append(M)
+ nc1_hist.append(nc1)
+ E_hist.append(state2.get("E", 0.0))
+ io_hist.append(state2.get("io", 0.0))
+ H_hist.append(state2.get("H", 0.0))
+ if baseline_hw_medians is None and len(ci_loop_hist) >= cfg_smell.ci_lookback_windows:
+ recent_loop = ci_loop_hist[-cfg_smell.ci_lookback_windows :]
+ recent_ex = ci_ex_hist[-cfg_smell.ci_lookback_windows :]
+ hw_loop_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in recent_loop])
+ hw_ex_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in recent_ex])
+ baseline_hw_medians = (
+ hw_loop_list[len(hw_loop_list) // 2],
+ hw_ex_list[len(hw_ex_list) // 2],
+ )
+ # Smell tests (full battery; the adversary does not get to declare
+ # its manipulation as an Ω window, so nothing is suspended).
+ if invalid_by_ci(res.ci_loop, res.ci_ex, cfg_smell):
+ lreg.invalidate("ci_inflation")
+ try:
+ hwL = 0.5 * abs(res.ci_loop[1] - res.ci_loop[0])
+ hwE = 0.5 * abs(res.ci_ex[1] - res.ci_ex[0])
+ except Exception:
+ hwL, hwE = None, None
+ _append_invalidation(
+ audit,
+ "ci_inflation",
+ {
+ "halfwidth_loop": hwL,
+ "halfwidth_ex": hwE,
+ "max_allowed": cfg_smell.max_ci_halfwidth,
+ },
+ _sink={},
+ )
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, baseline_hw_medians):
+ lreg.invalidate("ci_history_inflation")
+ med_loop: float | None = None
+ med_ex: float | None = None
+ b_loop: float | None = None
+ b_ex: float | None = None
+ try:
+ n = cfg_smell.ci_lookback_windows
+ rL = ci_loop_hist[-n:]
+ rE = ci_ex_hist[-n:]
+ hwL_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rL])
+ hwE_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rE])
+ med_loop = hwL_list[n // 2]
+ med_ex = hwE_list[n // 2]
+ if baseline_hw_medians:
+ b_loop, b_ex = baseline_hw_medians
+ except Exception:
+ pass
+ _append_invalidation(
+ audit,
+ "ci_history_inflation",
+ {
+ "median_hw_loop": med_loop,
+ "median_hw_ex": med_ex,
+ "baseline_hw_loop": b_loop,
+ "baseline_hw_ex": b_ex,
+ "max_allowed": cfg_smell.max_ci_halfwidth,
+ "inflate_factor": cfg_smell.ci_inflate_factor,
+ },
+ _sink={},
+ )
+ # Δt governance invalidation propagated from guard
+ if dt_guard.invalidated and not lreg.invalidated:
+ lreg.invalidate("dt_change_rate_limit")
+ # audit already appended by guard
+ # partition flip-rate guard
+ elapsed = max(1e-6, time.perf_counter() - start_time)
+ if invalid_by_partition_flips(pm.get().flips, elapsed, cfg_smell):
+ lreg.invalidate("partition_flapping")
+ rate = 3600.0 * (float(pm.get().flips) / max(1e-6, float(elapsed)))
+ _append_invalidation(
+ audit,
+ "partition_flapping",
+ {
+ "flips": pm.get().flips,
+ "elapsed_sec": elapsed,
+ "flips_per_hour": rate,
+ "limit_per_hour": cfg_smell.max_partition_flips_per_hour,
+ },
+ _sink={},
+ )
+ # Exogenous subsidy red flags (heuristic; never suspended here)
+ if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell):
+ lreg.invalidate("exogenous_subsidy_red_flag")
+ _append_invalidation(audit, "exogenous_subsidy_red_flag", {}, _sink={})
+ idx = lreg.write(
+ LEntry(
+ L_loop=res.L_loop,
+ L_ex=res.L_ex,
+ ci_loop=res.ci_loop,
+ ci_ex=res.ci_ex,
+ M_db=M,
+ nc1_pass=nc1,
+ )
+ )
+ audit.append(
+ "window_measured",
+ {"idx": idx, "M": M, "nc1": nc1, "partition_flips": pm.get().flips},
+ )
+ # export indicators (derived only)
+ derived = lreg.derive()
+ exported, base = exporter.maybe_export(priv, audit, derived, icfg, last_sc1_pass=False)
+ if exported:
+ audit.append("indicators_exported", {"base": os.path.basename(base)})
+ # Deterministic growth cadence with hysteresis (skip if frozen)
+ window_idx += 1
+ if part_growth_enabled and (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
+ part = pm.get()
+ cand_C, dM_db, greedy_details = greedy_suggest_C(
+ X=X,
+ C=part.C,
+ Ex=part.Ex,
+ estimator=estimate_L,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot_candidates=max(8, n_boot // 4),
+ mi_k=mi_k,
+ lam=part_lambda,
+ theta=part_theta,
+ kappa=part_kappa,
+ )
+ if cand_C != part.C:
+ pm.maybe_regrow(
+ cand_C,
+ delta_M_db=float(dM_db),
+ delta_M_min_db=part_delta_M_min_db,
+ consecutive_required=part_consecutive_required,
+ )
+ if pm.get().flips != last_flip_count:
+ info = getattr(pm, "last_flip_info", None)
+ details = {
+ "flips": pm.get().flips,
+ "new_C": pm.get().C,
+ "greedy_added": greedy_details.get("added", []),
+ "greedy_step_gains": greedy_details.get("step_gains", []),
+ "greedy_M_base": greedy_details.get("M_base"),
+ "greedy_M_final": greedy_details.get("M_final"),
+ }
+ if info is not None:
+ details.update(
+ {
+ "delta_M_db": info.get("delta_M_db"),
+ "hysteresis_streak": info.get("streak"),
+ "candidate_C": info.get("new_C"),
+ }
+ )
+ audit.append("partition_flip", details)
+ last_flip_count = pm.get().flips
+
+ def _audit_hook(ev: str, det: dict) -> None:
+ # Discard return value; hook contract expects None
+ audit.append(ev, det)
+ return None
+
+ # Δt governance guard
+ dt_guard_cfg = DtGuardConfig(
+ max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
+ min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
+ )
+ dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
+ try:
+ sch.start()
+ sch.run_for(run_sec)
+ audit.append(f"{omega_name}_stop", {})
+ finally:
+ stats = sch.stop()
+ # Δt jitter smell-test: invalidate if p95(|jitter|)/dt exceeds threshold
+ if (stats.jitter_p95_abs / max(1e-9, dt)) > SmellConfig().jitter_p95_rel_max:
+ lreg.invalidate("dt_jitter_excess")
+ _append_invalidation(
+ audit,
+ "dt_jitter_excess",
+ {
+ "jitter_p95_abs": stats.jitter_p95_abs,
+ "jitter_p95_rel": stats.jitter_p95_abs / max(1e-9, dt),
+ "dt": dt,
+ },
+ _sink={},
+ )
+ # Audit-chain integrity check
+ audit_path = os.path.join(dirs["audits"], "audit.jsonl")
+ if audit_chain_broken(audit_path):
+ lreg.invalidate("audit_chain_broken")
+ _append_invalidation(audit, "audit_chain_broken", {}, _sink={})
+ # LREG/raw export breach check: audit must not contain raw LREG values
+ if audit_contains_raw_lreg_values(audit_path):
+ lreg.invalidate("raw_lreg_breach")
+ _append_invalidation(audit, "raw_lreg_breach", {}, _sink={})
+
+ # The adversarial battery is built around designed non-certification, so
+ # report the headline NC1 quantities explicitly: the margin median AND
+ # the fraction of windows that actually certified (margin + noise gate).
+ valid_ms = [m for m in M_hist if m == m]
+ if valid_ms:
+ med = sorted(valid_ms)[len(valid_ms) // 2]
+ frac = (sum(1 for f in nc1_hist if f) / len(nc1_hist)) if nc1_hist else 0.0
+ print(
+ f"Adversarial {mode} done. median M = {med:+.2f} dB vs Mmin = {Mmin:.2f} dB; "
+ f"NC1 certified {100.0 * frac:.0f}% of {len(valid_ms)} windows "
+ f"(loop-influence gate L_floor = {L_floor:g})."
+ )
+ else:
+ print(f"Adversarial {mode} done. No measured windows.")
+ print(f"Audit: {os.path.join(dirs['audits'], 'audit.jsonl')}")
+ _print_invalidation_footer(os.path.join(dirs["audits"], "audit.jsonl"))
+ print(f"Indicators dir: {dirs['indicators']}")
+
+ # Build verification bundle (timeline, manifest)
+ try:
+ out = build_verification_bundle(dirs["figures"], os.path.join(dirs["audits"], "audit.jsonl"))
+ audit.append(
+ "report_generated",
+ {
+ "timeline_png": os.path.basename(out.get("timeline_png", "")),
+ "timeline_svg": os.path.basename(out.get("timeline_svg", "")),
+ "table": (os.path.basename(out.get("sc1_table", "")) if out.get("sc1_table") else None),
+ "manifest": os.path.basename(out.get("manifest", "")),
+ },
+ )
+ print(
+ "Bundle: "
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
+ )
+ except Exception:
+ pass
+
+
+def adv_replay_controller(args: argparse.Namespace) -> None:
+ """Adversarial gaming scenario: replayed actuation tape.
+
+ Records the actuation trace of a healthy closed-loop run of the same
+ plant, then drives a fresh plant with the recorded tape instead of a
+ controller. The actuators move exactly as under genuine control, but
+ the activity carries no dependence on the current state. Designed
+ outcome: `NC1` fails (`M` low) while the run stays valid.
+
+ Args:
+ args: Parsed argparse namespace with `--config`.
+ """
+ _run_adversarial(args, "replay_controller")
+
+
+def adv_hidden_tether(args: argparse.Namespace) -> None:
+ """Adversarial gaming scenario: wizard-of-oz control through `Ex`.
+
+ Control actions are computed outside the boundary from the observed
+ plant state (with a small command dither, as a real teleoperation
+ link would have) and injected through the exchange channel: the `io`
+ channel carries the command traffic and actuation lags one tick.
+ Designed outcome: loop influence collapses onto `Ex`, so `NC1`
+ fails, or a guardrail invalidates the run.
+
+ Args:
+ args: Parsed argparse namespace with `--config` and `--dither`.
+ """
+ _run_adversarial(args, "hidden_tether")
+
+
+def adv_oscillator(args: argparse.Namespace) -> None:
+ """Adversarial gaming scenario: oscillator inflation.
+
+ Runs the loop-ablated plant (no self-maintenance loop) and paints a
+ high-amplitude deterministic carrier onto the reported `T` and `R`
+ telemetry to inflate apparent self-prediction. Designed outcome: the
+ harness must not certify it; either `M` stays below `Mmin` or a
+ smell test fires.
+
+ Args:
+ args: Parsed argparse namespace with `--config`, `--amp`, and
+ `--period` (carrier period in seconds).
+ """
+ _run_adversarial(args, "oscillator")
+
+
+def run_policy(args: argparse.Namespace) -> None:
+ """Run the baseline NC1 loop with a learned policy as the controller.
+
+ The measurement loop, guardrails, attestation, and artifact bundle are
+ identical to [`run_baseline`][ldtc.cli.main.run_baseline]; the only
+ difference is the source of the control actions, which is a trained
+ [`MLPPolicy`][ldtc.plant.policy_controller.MLPPolicy] checkpoint
+ (written by `scripts/train_agent.py`) wrapped in a
+ [`PolicyController`][ldtc.plant.policy_controller.PolicyController].
+
+ Two state-independent ablations of the same checkpoint are available
+ for the emergence-under-learning demonstration. Both first record an
+ action tape from a closed-loop rollout of the policy on a throwaway
+ plant built from the same profile, then drive the measured plant with
+ state-independent actions drawn from that tape: ``shuffled`` samples
+ the tape i.i.d. (matched marginals, no state dependence) and
+ ``frozen`` holds the tape mean. If the trained policy's loop dominance
+ is genuinely carried by its state feedback, both ablations must
+ collapse it.
+
+ Args:
+ args: Parsed argparse namespace with `--config`, `--policy`
+ (checkpoint JSON path), and `--ablation`
+ (`none` / `shuffled` / `frozen`).
+ """
+ from ..plant.policy_controller import (
+ ABLATION_MODES,
+ MLPPolicy,
+ PolicyController,
+ record_policy_tape,
+ )
+
+ ablation = str(getattr(args, "ablation", "none") or "none")
+ if ablation not in ABLATION_MODES:
+ raise ValueError(f"Unknown ablation mode: {ablation} (expected one of {ABLATION_MODES})")
+
+ prof = _load_yaml(args.config)
+ seeds = _set_seeds(prof)
+ dt = float(prof.get("dt", 0.01))
+ window_sec = float(prof.get("window_sec", 0.2))
+ window = max(4, int(window_sec / dt))
+ method = str(prof.get("method", "linear"))
+ Mmin = float(prof.get("Mmin_db", 3.0))
+ L_floor = float(prof.get("L_floor", L_FLOOR_DEFAULT))
+ p_lag = int(prof.get("p_lag", 3))
+ mi_lag = int(prof.get("mi_lag", 1))
+ n_boot = int(prof.get("n_boot", 32))
+ mi_k = int(prof.get("mi_k", 5))
+ # Partition growth hysteresis config (parity with run_baseline; growth is
+ # off by default, so the declared self-maintenance set is the C under test
+ # and the partition cannot flap).
+ part_delta_M_min_db = float(prof.get("part_delta_M_min_db", 0.5))
+ part_consecutive_required = int(prof.get("part_consecutive_required", 3))
+ part_growth_cadence_windows = int(prof.get("part_growth_cadence_windows", 5))
+ part_growth_enabled = bool(prof.get("part_growth_enabled", False))
+ part_lambda = float(prof.get("part_lambda", 0.0))
+ part_theta = float(prof.get("part_theta", 0.0))
+ _kappa_val_pol = prof.get("part_kappa")
+ part_kappa = int(_kappa_val_pol) if _kappa_val_pol is not None else None
+ run_sec = float(prof.get("baseline_sec", 10.0))
+
+ policy_path = str(args.policy)
+ policy = MLPPolicy.load(policy_path)
+ tape_ticks = int(round(run_sec / dt)) + window + 8
+
+ tag = "policy" if ablation == "none" else f"policy-{ablation}"
+ omega_args: Dict[str, Any] = {"policy": os.path.basename(policy_path), "ablation": ablation}
+ if ablation != "none":
+ omega_args["tape_ticks"] = tape_ticks
+
+ dirs = _ensure_dirs(tag)
+ audit = AuditLog(os.path.join(dirs["audits"], "audit.jsonl"))
+ audit.append("policy_run_start", {"config": args.config, **omega_args})
+ _print_and_audit_header(
+ audit,
+ {
+ "profile_id": int(prof.get("profile_id", 0)),
+ "config_path": str(args.config),
+ "dt": dt,
+ "window_sec": window_sec,
+ "method": method,
+ "p_lag": p_lag,
+ "mi_lag": mi_lag,
+ "Mmin_db": Mmin,
+ "L_floor": L_floor,
+ "epsilon": float(prof.get("epsilon", 0.15)),
+ "tau_max": float(prof.get("tau_max", 60.0)),
+ "mi_k": mi_k,
+ **seeds,
+ "omega": "policy",
+ "omega_args": omega_args,
+ },
+ )
+ meta = getattr(policy, "meta", {}) or {}
+ audit.append(
+ "policy_loaded",
+ {
+ "path": os.path.basename(policy_path),
+ "n_params": policy.n_params,
+ "hidden": policy.hidden,
+ "obs_keys": ",".join(policy.obs_keys),
+ "train_frac": meta.get("frac"),
+ "train_generation": meta.get("generation"),
+ },
+ )
+
+ # Plant under test.
+ adapter = _make_adapter_from_profile(prof)
+ order = ["E", "T", "R", "demand", "io", "H"]
+ sw = SlidingWindow(capacity=window, channel_order=order)
+ pm = PartitionManager(N_signals=len(order), seed_C=[0, 1, 2])
+ lreg = LREG()
+ kp = KeyPaths(
+ priv_path=os.path.join("artifacts", "keys", "ed25519_priv.pem"),
+ pub_path=os.path.join("artifacts", "keys", "ed25519_pub.pem"),
+ )
+ priv, _ = ensure_keys(kp)
+ exporter = IndicatorExporter(out_dir=dirs["indicators"], rate_hz=2.0)
+ icfg = IndicatorConfig(Mmin_db=Mmin, profile_id=int(prof.get("profile_id", 0)))
+
+ # Ablation setup: record the matched action tape from a closed-loop
+ # rollout of the same policy on a throwaway plant (same profile), so the
+ # ablated run preserves the actuation marginals while severing the loop.
+ tape = None
+ if ablation != "none":
+ rec_adapter = _make_adapter_from_profile(prof)
+ tape = record_policy_tape(rec_adapter, policy, tape_ticks)
+ audit.append(
+ "policy_tape_recorded",
+ {
+ "ticks": len(tape),
+ "throttle_mean": round(float(np.mean([a[0] for a in tape])), 4),
+ "cool_mean": round(float(np.mean([a[1] for a in tape])), 4),
+ "repair_mean": round(float(np.mean([a[2] for a in tape])), 4),
+ },
+ )
+ controller = PolicyController(policy, ablation=ablation, tape=tape, seed=seeds["seed_py"] + 1)
+
+ start_time = time.perf_counter()
+ cfg_smell = SmellConfig()
+ ci_loop_hist = []
+ ci_ex_hist = []
+ baseline_hw_medians = None
+ M_hist = []
+ nc1_hist = []
+ io_hist = []
+ E_hist = []
+ H_hist = []
+
+ window_idx = 0
+ last_flip_count = 0
+
+ def tick(_now: float) -> None:
+ nonlocal window_idx, last_flip_count, baseline_hw_medians
+ state = adapter.read_state()
+ act = controller.compute(state)
+ adapter.write_actuators(action=act)
+ # measure
+ state2 = adapter.read_state()
+ sw.append(state2)
+ if sw.ready():
+ X = np.asarray(sw.get_matrix())
+ part = pm.get()
+ res = estimate_L(
+ X=X,
+ C=part.C,
+ Ex=part.Ex,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot=n_boot,
+ mi_k=mi_k,
+ )
+ # Diagnostics: stationarity + VAR N/T ratio (stationarity gated by
+ # cadence to keep long studies tractable; no raw LREG values).
+ _emit_window_diagnostics(
+ audit,
+ X,
+ p_lag,
+ method,
+ int(lreg.derive().get("counter", 0)),
+ int(prof.get("diag_cadence_windows", 1)),
+ )
+ M = m_db(res.L_loop, res.L_ex)
+ nc1 = nc1_certify(M, res.L_loop, Mmin, L_floor)
+ # Histories for smell tests
+ ci_loop_hist.append(res.ci_loop)
+ ci_ex_hist.append(res.ci_ex)
+ M_hist.append(M)
+ nc1_hist.append(nc1)
+ E_hist.append(state2.get("E", 0.0))
+ io_hist.append(state2.get("io", 0.0))
+ H_hist.append(state2.get("H", 0.0))
+ if baseline_hw_medians is None and len(ci_loop_hist) >= cfg_smell.ci_lookback_windows:
+ recent_loop = ci_loop_hist[-cfg_smell.ci_lookback_windows :]
+ recent_ex = ci_ex_hist[-cfg_smell.ci_lookback_windows :]
+ hw_loop_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in recent_loop])
+ hw_ex_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in recent_ex])
+ baseline_hw_medians = (
+ hw_loop_list[len(hw_loop_list) // 2],
+ hw_ex_list[len(hw_ex_list) // 2],
+ )
+ # Smell tests (full battery; nothing is suspended in this run).
+ if invalid_by_ci(res.ci_loop, res.ci_ex, cfg_smell):
+ lreg.invalidate("ci_inflation")
+ try:
+ hwL = 0.5 * abs(res.ci_loop[1] - res.ci_loop[0])
+ hwE = 0.5 * abs(res.ci_ex[1] - res.ci_ex[0])
+ except Exception:
+ hwL, hwE = None, None
+ _append_invalidation(
+ audit,
+ "ci_inflation",
+ {
+ "halfwidth_loop": hwL,
+ "halfwidth_ex": hwE,
+ "max_allowed": cfg_smell.max_ci_halfwidth,
+ },
+ _sink={},
+ )
+ if invalid_by_ci_history(ci_loop_hist, ci_ex_hist, cfg_smell, baseline_hw_medians):
+ lreg.invalidate("ci_history_inflation")
+ med_loop: float | None = None
+ med_ex: float | None = None
+ b_loop: float | None = None
+ b_ex: float | None = None
+ try:
+ n = cfg_smell.ci_lookback_windows
+ rL = ci_loop_hist[-n:]
+ rE = ci_ex_hist[-n:]
+ hwL_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rL])
+ hwE_list = sorted([0.5 * abs(lohi[1] - lohi[0]) for lohi in rE])
+ med_loop = hwL_list[n // 2]
+ med_ex = hwE_list[n // 2]
+ if baseline_hw_medians:
+ b_loop, b_ex = baseline_hw_medians
+ except Exception:
+ pass
+ _append_invalidation(
+ audit,
+ "ci_history_inflation",
+ {
+ "median_hw_loop": med_loop,
+ "median_hw_ex": med_ex,
+ "baseline_hw_loop": b_loop,
+ "baseline_hw_ex": b_ex,
+ "max_allowed": cfg_smell.max_ci_halfwidth,
+ "inflate_factor": cfg_smell.ci_inflate_factor,
+ },
+ _sink={},
+ )
+ # Δt governance invalidation propagated from guard
+ if dt_guard.invalidated and not lreg.invalidated:
+ lreg.invalidate("dt_change_rate_limit")
+ # audit already appended by guard
+ # partition flip-rate guard
+ elapsed = max(1e-6, time.perf_counter() - start_time)
+ if invalid_by_partition_flips(pm.get().flips, elapsed, cfg_smell):
+ lreg.invalidate("partition_flapping")
+ rate = 3600.0 * (float(pm.get().flips) / max(1e-6, float(elapsed)))
+ _append_invalidation(
+ audit,
+ "partition_flapping",
+ {
+ "flips": pm.get().flips,
+ "elapsed_sec": elapsed,
+ "flips_per_hour": rate,
+ "limit_per_hour": cfg_smell.max_partition_flips_per_hour,
+ },
+ _sink={},
+ )
+ # Exogenous subsidy red flags (heuristic; never suspended here)
+ if exogenous_subsidy_red_flag(M_hist, io_hist, E_hist, H_hist, cfg_smell):
+ lreg.invalidate("exogenous_subsidy_red_flag")
+ _append_invalidation(audit, "exogenous_subsidy_red_flag", {}, _sink={})
+ idx = lreg.write(
+ LEntry(
+ L_loop=res.L_loop,
+ L_ex=res.L_ex,
+ ci_loop=res.ci_loop,
+ ci_ex=res.ci_ex,
+ M_db=M,
+ nc1_pass=nc1,
+ )
+ )
+ audit.append(
+ "window_measured",
+ {"idx": idx, "M": M, "nc1": nc1, "partition_flips": pm.get().flips},
+ )
+ # export indicators (derived only)
+ derived = lreg.derive()
+ exported, base = exporter.maybe_export(priv, audit, derived, icfg, last_sc1_pass=False)
+ if exported:
+ audit.append("indicators_exported", {"base": os.path.basename(base)})
+ # Deterministic growth cadence with hysteresis (skip if frozen)
+ window_idx += 1
+ if part_growth_enabled and (window_idx % part_growth_cadence_windows) == 0 and not pm.get().frozen:
+ part = pm.get()
+ cand_C, dM_db, greedy_details = greedy_suggest_C(
+ X=X,
+ C=part.C,
+ Ex=part.Ex,
+ estimator=estimate_L,
+ method=method,
+ p=p_lag,
+ lag_mi=mi_lag,
+ n_boot_candidates=max(8, n_boot // 4),
+ mi_k=mi_k,
+ lam=part_lambda,
+ theta=part_theta,
+ kappa=part_kappa,
+ )
+ if cand_C != part.C:
+ pm.maybe_regrow(
+ cand_C,
+ delta_M_db=float(dM_db),
+ delta_M_min_db=part_delta_M_min_db,
+ consecutive_required=part_consecutive_required,
+ )
+ if pm.get().flips != last_flip_count:
+ info = getattr(pm, "last_flip_info", None)
+ details = {
+ "flips": pm.get().flips,
+ "new_C": pm.get().C,
+ "greedy_added": greedy_details.get("added", []),
+ "greedy_step_gains": greedy_details.get("step_gains", []),
+ "greedy_M_base": greedy_details.get("M_base"),
+ "greedy_M_final": greedy_details.get("M_final"),
+ }
+ if info is not None:
+ details.update(
+ {
+ "delta_M_db": info.get("delta_M_db"),
+ "hysteresis_streak": info.get("streak"),
+ "candidate_C": info.get("new_C"),
+ }
+ )
+ audit.append("partition_flip", details)
+ last_flip_count = pm.get().flips
+
+ def _audit_hook(ev: str, det: dict) -> None:
+ # Discard return value; hook contract expects None
+ audit.append(ev, det)
+ return None
+
+ # Δt governance guard
+ dt_guard_cfg = DtGuardConfig(
+ max_changes_per_hour=int(prof.get("max_dt_changes_per_hour", 3)),
+ min_seconds_between_changes=float(prof.get("min_seconds_between_changes", 1.0)),
+ )
+ dt_guard = DeltaTGuard(audit=audit, cfg=dt_guard_cfg)
+ sch = make_driver(prof, dt, tick, _audit_hook, dt_guard)
+ try:
+ sch.start()
+ sch.run_for(run_sec)
+ audit.append("policy_run_stop", {})
+ finally:
+ stats = sch.stop()
+ # Δt jitter smell-test: invalidate if p95(|jitter|)/dt exceeds threshold
+ if (stats.jitter_p95_abs / max(1e-9, dt)) > SmellConfig().jitter_p95_rel_max:
+ lreg.invalidate("dt_jitter_excess")
+ _append_invalidation(
+ audit,
+ "dt_jitter_excess",
+ {
+ "jitter_p95_abs": stats.jitter_p95_abs,
+ "jitter_p95_rel": stats.jitter_p95_abs / max(1e-9, dt),
+ "dt": dt,
+ },
+ _sink={},
+ )
+ # Audit-chain integrity check
+ audit_path = os.path.join(dirs["audits"], "audit.jsonl")
+ if audit_chain_broken(audit_path):
+ lreg.invalidate("audit_chain_broken")
+ _append_invalidation(audit, "audit_chain_broken", {}, _sink={})
+ # LREG/raw export breach check: audit must not contain raw LREG values
+ if audit_contains_raw_lreg_values(audit_path):
+ lreg.invalidate("raw_lreg_breach")
+ _append_invalidation(audit, "raw_lreg_breach", {}, _sink={})
+
+ # Report the headline NC1 quantities explicitly: the margin median AND
+ # the fraction of windows that actually certified (margin + noise gate).
+ label = "learned policy" if ablation == "none" else f"ablation: {ablation}"
+ valid_ms = [m for m in M_hist if m == m]
+ if valid_ms:
+ med = sorted(valid_ms)[len(valid_ms) // 2]
+ frac = (sum(1 for f in nc1_hist if f) / len(nc1_hist)) if nc1_hist else 0.0
+ print(
+ f"Policy run done ({label}). median M = {med:+.2f} dB vs Mmin = {Mmin:.2f} dB; "
+ f"NC1 certified {100.0 * frac:.0f}% of {len(valid_ms)} windows "
+ f"(loop-influence gate L_floor = {L_floor:g})."
+ )
+ else:
+ print(f"Policy run done ({label}). No measured windows.")
+ print(f"Audit: {os.path.join(dirs['audits'], 'audit.jsonl')}")
+ _print_invalidation_footer(os.path.join(dirs["audits"], "audit.jsonl"))
+ print(f"Indicators dir: {dirs['indicators']}")
+
+ # Build verification bundle (timeline, manifest)
+ try:
+ out = build_verification_bundle(dirs["figures"], os.path.join(dirs["audits"], "audit.jsonl"))
+ audit.append(
+ "report_generated",
+ {
+ "timeline_png": os.path.basename(out.get("timeline_png", "")),
+ "timeline_svg": os.path.basename(out.get("timeline_svg", "")),
+ "table": (os.path.basename(out.get("sc1_table", "")) if out.get("sc1_table") else None),
+ "manifest": os.path.basename(out.get("manifest", "")),
+ },
+ )
+ print(
+ "Bundle: "
+ f"timeline={out.get('timeline_png', '')}, "
+ f"table={out.get('sc1_table', '')}, "
+ f"manifest={out.get('manifest', '')}"
+ )
+ except Exception:
+ pass
+
+
+def build_parser() -> argparse.ArgumentParser:
+ """Build the top-level `ldtc` argparse parser.
+
+ Wires up the `run` subcommand and the five `omega-*` subcommands;
+ each subparser binds its handler via `set_defaults(func=...)`.
+
+ Returns:
+ Configured `ArgumentParser`. Call `parse_args()` and then
+ `args.func(args)` to dispatch.
+ """
+ p = argparse.ArgumentParser(prog="ldtc", description="LDTC CLI")
+ sub = p.add_subparsers(dest="cmd", required=True)
+
+ p_run = sub.add_parser("run", help="Run baseline NC1 loop")
+ p_run.add_argument("--config", required=True, help="YAML profile (e.g., configs/profile_r0.yml)")
+ p_run.set_defaults(func=run_baseline)
+
+ p_omega = sub.add_parser("omega-power-sag", help="Apply power-sag Ω and evaluate SC1")
+ p_omega.add_argument("--config", required=True)
+ p_omega.add_argument("--drop", type=float, default=0.3)
+ p_omega.add_argument("--duration", type=float, default=10.0)
+ p_omega.set_defaults(func=omega_power_sag)
+
+ p_ing = sub.add_parser("omega-ingress-flood", help="Apply sustained ingress-flood Ω and evaluate SC1")
+ p_ing.add_argument("--config", required=True)
+ p_ing.add_argument("--mult", type=float, default=3.0, help="Multiplier for ingress load")
+ p_ing.add_argument("--duration", type=float, default=5.0)
+ p_ing.set_defaults(func=omega_ingress_flood)
+
+ p_out = sub.add_parser(
+ "omega-control-outage",
+ help="Ablate the self-maintenance loop for a bounded interval (designed SC1 fail)",
+ )
+ p_out.add_argument("--config", required=True)
+ p_out.add_argument("--duration", type=float, default=6.0, help="Outage duration (s)")
+ p_out.set_defaults(func=omega_control_outage)
+
+ p_cc = sub.add_parser(
+ "omega-command-conflict",
+ help="Issue a risky command and measure refusal/Trefuse",
+ )
+ p_cc.add_argument("--config", required=True)
+ p_cc.add_argument(
+ "--observe",
+ type=float,
+ default=2.0,
+ help="Seconds to observe after issuing command",
+ )
+ p_cc.set_defaults(func=omega_command_conflict)
+
+ p_sub = sub.add_parser(
+ "omega-exogenous-subsidy",
help="Inject SoC without harvest to simulate subsidy (negative control)",
)
p_sub.add_argument("--config", required=True)
@@ -2093,6 +3530,48 @@ def build_parser() -> argparse.ArgumentParser:
p_sub.add_argument("--duration", type=float, default=3.0)
p_sub.set_defaults(func=omega_exogenous_subsidy)
+ p_rep = sub.add_parser(
+ "adv-replay-controller",
+ help="Adversarial: replay a recorded actuation tape (no closed loop)",
+ )
+ p_rep.add_argument("--config", required=True)
+ p_rep.set_defaults(func=adv_replay_controller)
+
+ p_tet = sub.add_parser(
+ "adv-hidden-tether",
+ help="Adversarial: wizard-of-oz control injected through the exchange channel",
+ )
+ p_tet.add_argument("--config", required=True)
+ p_tet.add_argument("--dither", type=float, default=0.10, help="Uniform command dither half-width")
+ p_tet.set_defaults(func=adv_hidden_tether)
+
+ p_osc = sub.add_parser(
+ "adv-oscillator",
+ help="Adversarial: deterministic carrier painted on loop telemetry",
+ )
+ p_osc.add_argument("--config", required=True)
+ p_osc.add_argument("--amp", type=float, default=0.10, help="Carrier amplitude (state units)")
+ p_osc.add_argument("--period", type=float, default=1.0, help="Carrier period (s)")
+ p_osc.set_defaults(func=adv_oscillator)
+
+ p_pol = sub.add_parser(
+ "run-policy",
+ help="Run the baseline NC1 loop with a learned policy checkpoint as the controller",
+ )
+ p_pol.add_argument("--config", required=True)
+ p_pol.add_argument(
+ "--policy",
+ required=True,
+ help="Policy checkpoint JSON (written by scripts/train_agent.py)",
+ )
+ p_pol.add_argument(
+ "--ablation",
+ choices=["none", "shuffled", "frozen"],
+ default="none",
+ help="State-independent ablation of the checkpoint (matched action tape)",
+ )
+ p_pol.set_defaults(func=run_policy)
+
return p
diff --git a/src/ldtc/guardrails/smelltests.py b/src/ldtc/guardrails/smelltests.py
index fc50a5c..3b4408e 100644
--- a/src/ldtc/guardrails/smelltests.py
+++ b/src/ldtc/guardrails/smelltests.py
@@ -6,8 +6,8 @@
- CI width guards (absolute and inflation-vs-baseline).
- Partition flip-rate checks (and forbidding flips during `Ω`).
- `Δt` jitter thresholds.
-- Exogenous-subsidy red flags (`M` rising while I/O is high; SoC rising
- with no harvest).
+- Exogenous-subsidy red flags (`M` rising while I/O is high; energy
+ appearing in the store faster than the metered influx allows).
- Audit-chain integrity checks (counter / hash / timestamp continuity).
If any guard returns `True`, the CLI invalidates the run by appending a
@@ -48,8 +48,17 @@ class SmellConfig:
exogenous-subsidy heuristic considers the channel suspicious.
min_M_rise_db: Minimum `ΔM` (dB) to flag as a subsidy.
M_rise_lookback: Look-back windows for the subsidy check.
- min_harvest_for_soc_gain: Minimum harvest considered non-zero for
- SoC-gain detection.
+ min_io_rise: Minimum I/O increase over the look-back for the
+ rising-M branch to fire. Requiring a material ramp (rather
+ than any positive jitter) keeps a legitimately elevated,
+ fluctuating I/O channel (e.g., a sustained ingress flood the
+ loop is successfully shielding) from tripping the flag.
+ soc_jump_margin: Energy-conservation allowance for the
+ unexplained-SoC-gain check: a single-tick SoC rise may not
+ exceed the metered influx (harvest) by more than this margin
+ (which covers sensor/process noise). Any larger one-tick gain
+ means energy entered the store from outside the metered
+ channel.
"""
max_dt_changes_per_hour: int = 3
@@ -58,14 +67,37 @@ class SmellConfig:
forbid_partition_flip_during_omega: bool = True
# CI look-back configuration
ci_lookback_windows: int = 5
- ci_inflate_factor: float = 2.0 # relative to baseline median
+ # Relative-inflation factor for the median CI half-width vs the early
+ # baseline median. Bootstrap CI half-widths on a short (~60-sample) window
+ # have substantial window-to-window variability, so a 2x swing is within
+ # normal noise; the relative guard should fire only on a gross degradation.
+ # The absolute cap (``max_ci_halfwidth``) remains the hard limit.
+ ci_inflate_factor: float = 3.0 # relative to baseline median
+ # The relative-inflation check only applies once the CI is absolutely
+ # non-trivial. Near the noise floor (e.g., L_ex ~ 0 in the positive
+ # control, or L_loop ~ 0 in a negative control) a tiny baseline half-width
+ # can "inflate" by >2x while the estimate stays extremely precise; that is
+ # not a measurement-quality failure. We therefore require the inflated
+ # half-width to also exceed this absolute floor, set to half the absolute
+ # cap (``max_ci_halfwidth``) so the relative check only ever fires for a CI
+ # that is genuinely degrading toward the absolute limit.
+ ci_inflate_min_hw: float = 0.15
# Δt jitter guard (relative to dt)
jitter_p95_rel_max: float = 0.25 # invalidate if p95(|jitter|)/dt exceeds this
# Exogenous-subsidy heuristics
io_suspicious_threshold: float = 0.8
min_M_rise_db: float = 0.5
M_rise_lookback: int = 3
- min_harvest_for_soc_gain: float = 1e-3
+ min_io_rise: float = 0.08
+ # Energy-conservation audit. Every legitimate path into the energy store is
+ # metered by the harvest channel H, so over one tick the SoC can rise by at
+ # most H plus a noise allowance. The plant's per-tick process noise is
+ # bounded (|noise_energy| <= 0.024 in the software plant), so a margin of
+ # 0.06 can never fire on legitimate dynamics yet catches any injection
+ # pulse well above the noise floor. Subsidies that trickle in below the
+ # noise floor are undetectable by construction (and correspondingly cannot
+ # buy a measurable survival advantage per tick).
+ soc_jump_margin: float = 0.06
def ci_halfwidth(ci: Tuple[float, float]) -> float:
@@ -189,9 +221,10 @@ def invalid_by_ci_history(
return True
if baseline_medians is not None:
b_loop, b_ex = baseline_medians
- if b_loop > 0 and med_loop >= cfg.ci_inflate_factor * b_loop:
+ floor = cfg.ci_inflate_min_hw
+ if b_loop > 0 and med_loop >= cfg.ci_inflate_factor * b_loop and med_loop >= floor:
return True
- if b_ex > 0 and med_ex >= cfg.ci_inflate_factor * b_ex:
+ if b_ex > 0 and med_ex >= cfg.ci_inflate_factor * b_ex and med_ex >= floor:
return True
return False
except Exception:
@@ -234,49 +267,102 @@ def audit_contains_raw_lreg_values(audit_path: str) -> bool:
return False
+def unexplained_soc_gain(
+ Es: Sequence[float],
+ Hs: Sequence[float],
+ cfg: SmellConfig,
+) -> bool:
+ """Energy-conservation audit on the per-tick SoC series.
+
+ All legitimate energy entering the store is metered by the harvest
+ channel, so over a single tick the SoC may rise by at most the
+ metered influx plus a noise allowance (`cfg.soc_jump_margin`). A
+ larger one-tick gain means energy entered the store outside the
+ metered channel: an exogenous subsidy. This check is deterministic
+ on legitimate dynamics (the plant's per-tick noise is strictly below
+ the margin) and fires on every injection pulse above the noise
+ floor, whether or not harvest is currently zero.
+
+ Args:
+ Es: Per-tick state-of-charge series.
+ Hs: Per-tick harvest series, sampled at the same ticks as `Es`.
+ cfg: Threshold configuration.
+
+ Returns:
+ `True` if any single-tick SoC gain exceeds the metered influx by
+ more than the margin.
+ """
+ n = min(len(Es), len(Hs))
+ for i in range(1, n):
+ gain = Es[i] - Es[i - 1]
+ # Allow the larger of the two adjacent harvest readings so that a
+ # legitimate step-up in harvest (e.g., sag release) cannot be
+ # mistaken for an injection.
+ influx = max(Hs[i - 1], Hs[i])
+ if gain - influx > cfg.soc_jump_margin:
+ return True
+ return False
+
+
def exogenous_subsidy_red_flag(
Ms_db: Sequence[float],
ios: Sequence[float],
Es: Sequence[float],
Hs: Sequence[float],
cfg: SmellConfig,
+ omega_declared: bool = False,
) -> bool:
"""Heuristics for detecting exogenous-subsidy conditions.
- Flags when `M` is rising while I/O is both high and increasing, or
- when SoC is rising while harvest is approximately zero over a
- look-back window. Both situations suggest the apparent loop
- dominance comes from outside the system rather than from a real
- closed-loop dynamic.
+ Two branches:
+
+ 1. *Undeclared exchange surge*: `M` rising while I/O is high and
+ materially ramping. An unannounced surge on an exchange channel
+ that coincides with rising measured dominance suggests the
+ dominance is being bought on that channel. This branch is
+ suspended while a *declared* `Ω` stimulus is in effect
+ (`omega_declared=True`), because a declared ingress flood is
+ exactly such a surge and is the experiment, not a confound.
+ 2. *Energy conservation*: the store gains charge faster than the
+ metered influx allows (see
+ [`unexplained_soc_gain`][ldtc.guardrails.smelltests.unexplained_soc_gain]).
+ This branch is never suspended: declared or not, energy
+ appearing from outside the metered channel invalidates the run.
+
+ A legitimate drain (e.g., spending stored energy under a harvest
+ cut) never fires either branch.
Args:
Ms_db: Recent `M (dB)` values.
ios: Recent I/O fraction values.
- Es: Recent state-of-charge values.
- Hs: Recent harvest values.
+ Es: Per-tick state-of-charge series.
+ Hs: Per-tick harvest series.
cfg: Threshold configuration.
+ omega_declared: `True` while a declared `Ω` stimulus (or its
+ recovery window) is in effect.
Returns:
- `True` if either heuristic fires. Returns `False` defensively on
- any internal error.
+ `True` if either active heuristic fires. Returns `False`
+ defensively on any internal error.
"""
try:
+ # Rising-M-with-suspicious-I/O branch (apparent dominance bought on the
+ # exchange channel); suspended during declared Ω windows.
n = cfg.M_rise_lookback
- if len(Ms_db) < n or len(ios) < n or len(Es) < n or len(Hs) < n:
- return False
- recent_M = Ms_db[-n:]
- recent_io = ios[-n:]
- recent_E = Es[-n:]
- recent_H = Hs[-n:]
- # Simple rise check
- M_rise = recent_M[-1] - recent_M[0]
- io_rise = recent_io[-1] - recent_io[0]
- if (M_rise >= cfg.min_M_rise_db) and (recent_io[-1] >= cfg.io_suspicious_threshold) and (io_rise > 0):
- return True
- # SoC rising while harvest ~0
- E_rise = recent_E[-1] - recent_E[0]
- avg_H = sum(recent_H) / float(n)
- if (E_rise > 0.0) and (avg_H <= cfg.min_harvest_for_soc_gain):
+ if not omega_declared and len(Ms_db) >= n and len(ios) >= n:
+ recent_M = Ms_db[-n:]
+ recent_io = ios[-n:]
+ M_rise = recent_M[-1] - recent_M[0]
+ io_rise = recent_io[-1] - recent_io[0]
+ if (
+ (M_rise >= cfg.min_M_rise_db)
+ and (recent_io[-1] >= cfg.io_suspicious_threshold)
+ and (io_rise >= cfg.min_io_rise)
+ ):
+ return True
+ # Energy-conservation branch: SoC must not rise faster than the metered
+ # influx allows. Always active.
+ if unexplained_soc_gain(Es, Hs, cfg):
return True
return False
except Exception:
diff --git a/src/ldtc/lmeas/estimators.py b/src/ldtc/lmeas/estimators.py
index 86da140..0a4163a 100644
--- a/src/ldtc/lmeas/estimators.py
+++ b/src/ldtc/lmeas/estimators.py
@@ -104,6 +104,15 @@ def _dir_influence_linear_conditional(
when adding lagged predictors from `add_sources` on top of an AR baseline
and lagged `base_sources`.
+ The improvement is measured as the difference of *adjusted* R²
+ between the full model (baseline plus `add_sources`) and the baseline
+ model. Adjusted R² penalizes the extra parameters, so adding lagged
+ predictors that carry no genuine predictive information yields an
+ expected improvement of approximately zero rather than the positive
+ bias (`≈ k_added / n`) of an in-sample partial R². This is what keeps
+ the estimator honest on the short windows used by the harness, so that
+ `L_loop` and `L_ex` reflect real, not spurious, predictive dependence.
+
Args:
x: Array of shape `(T, N)` with time along the first dimension.
p: Number of lags for the linear model.
@@ -113,7 +122,8 @@ def _dir_influence_linear_conditional(
targets: Target signal indices to evaluate.
Returns:
- Mean partial R² improvement across `targets`.
+ Mean adjusted-R² improvement across `targets`, clamped to
+ `[0, 1]`.
"""
X, Y = _lag_matrix(x, p)
Tm, N = Y.shape
@@ -126,40 +136,45 @@ def cols_for(indices: Sequence[int]) -> List[int]:
out.extend((idx_arr + lag * Nsig).tolist())
return out
+ def adj_r2(design: np.ndarray, y: np.ndarray) -> float:
+ # Adjusted R^2 for an OLS fit of (mean-centered) y on a
+ # mean-centered design (the centering absorbs the intercept).
+ n = y.shape[0]
+ k = design.shape[1]
+ if k == 0:
+ return 0.0
+ if n - k - 1 <= 0:
+ return float("nan")
+ beta, *_ = np.linalg.lstsq(design, y, rcond=None)
+ resid = y - design @ beta
+ ssr = float(np.sum(resid * resid))
+ sst = float(np.sum(y * y)) + 1e-12
+ r2 = 1.0 - ssr / sst
+ return 1.0 - (1.0 - r2) * (n - 1) / (n - k - 1)
+
+ Xc = X - X.mean(axis=0, keepdims=True)
+ Yc = Y - Y.mean(axis=0, keepdims=True)
+
r2_improvements = []
for t in targets:
- cols_ar = np.array(cols_for([t]), dtype=int)
+ cols_ar = cols_for([t])
# Exclude the target from add/base sources to avoid self-lag duplication
base_eff = [s for s in base_sources if s != t]
add_eff = [s for s in add_sources if s != t]
- cols_base = np.array(cols_for(base_eff), dtype=int) if len(base_eff) else np.array([], dtype=int)
- cols_add = np.array(cols_for(add_eff), dtype=int) if len(add_eff) else np.array([], dtype=int)
- # Build baseline and additional predictor matrices
- X_base = np.concatenate([X[:, cols_ar], X[:, cols_base]], axis=1) if len(cols_base) else X[:, cols_ar]
- A_add = X[:, cols_add] if len(cols_add) else np.zeros((X.shape[0], 0))
- y = Y[:, t]
- # Compute partial R^2 of add predictors given baseline using QR residualization
- if X_base.size == 0:
- # Should not occur; fallback to variance about mean
- r = y - np.mean(y)
- A_perp = A_add
- else:
- Qb, _ = np.linalg.qr(X_base, mode="reduced")
- yhat_b = Qb @ (Qb.T @ y)
- r = y - yhat_b
- if A_add.size:
- A_perp = A_add - Qb @ (Qb.T @ A_add)
- else:
- A_perp = A_add
- denom = float(np.sum(r * r)) + 1e-12
- if A_perp.size == 0:
- r2_add = 0.0
- else:
- beta_add, *_ = np.linalg.lstsq(A_perp, r, rcond=None)
- rhat = A_perp @ beta_add
- num = float(np.sum(rhat * rhat))
- r2_add = max(0.0, min(1.0, num / denom))
- r2_improvements.append(r2_add)
+ cols_base = cols_for(base_eff)
+ cols_add = cols_for(add_eff)
+ base_idx = np.array(cols_ar + cols_base, dtype=int)
+ full_idx = np.array(cols_ar + cols_base + cols_add, dtype=int)
+ y = Yc[:, t]
+ if len(cols_add) == 0:
+ r2_improvements.append(0.0)
+ continue
+ adj_base = adj_r2(Xc[:, base_idx], y)
+ adj_full = adj_r2(Xc[:, full_idx], y)
+ if not (np.isfinite(adj_base) and np.isfinite(adj_full)):
+ # Too few samples per parameter to assess this target reliably.
+ continue
+ r2_improvements.append(max(0.0, min(1.0, adj_full - adj_base)))
return float(np.mean(r2_improvements)) if r2_improvements else 0.0
@@ -174,7 +189,7 @@ def _dir_influence_linear(x: np.ndarray, p: int, sources: Sequence[int], targets
return _dir_influence_linear_conditional(x=x, p=p, add_sources=sources, base_sources=[], targets=targets)
-def _dir_influence_mi(x: np.ndarray, sources: Sequence[int], targets: Sequence[int], lag: int = 1) -> float:
+def _dir_influence_mi(x: np.ndarray, sources: Sequence[int], targets: Sequence[int], lag: int = 1, k: int = 5) -> float:
"""Average pairwise mutual information from sources to targets.
Computes MI between `sources` at time `t-lag` and `targets` at time `t`
@@ -185,6 +200,9 @@ def _dir_influence_mi(x: np.ndarray, sources: Sequence[int], targets: Sequence[i
sources: Indices of source signals.
targets: Indices of target signals.
lag: Positive lag between sources and targets (default 1 sample).
+ k: Number of neighbors for the kNN MI estimator (kept consistent
+ with the `mi_k` profile knob so all MI variants use the same
+ `k`).
Returns:
Mean mutual information across all source-target pairs.
@@ -200,7 +218,7 @@ def _dir_influence_mi(x: np.ndarray, sources: Sequence[int], targets: Sequence[i
continue
xs = x[:-lag, s]
# sklearn MI expects 2D X
- mi = mutual_info_regression(xs.reshape(-1, 1), y, discrete_features=False)
+ mi = mutual_info_regression(xs.reshape(-1, 1), y, discrete_features=False, n_neighbors=int(k))
vals.append(float(mi[0]))
return float(np.mean(vals)) if vals else 0.0
@@ -368,12 +386,16 @@ def _bootstrap(
if T < 12:
return (np.nan, np.nan)
# Default block length ~ window/4, with a small floor
+ if n_draws < 2:
+ return (np.nan, np.nan)
blk = int(block) if block is not None else max(4, T // 4)
idxs = block_bootstrap_indices(T, blk, n_draws)
vals: List[float] = []
for idx in idxs:
vals.append(fn(x[idx]))
arr = np.asarray(vals, dtype=float)
+ if arr.size == 0 or np.all(np.isnan(arr)):
+ return (np.nan, np.nan)
lo, hi = np.nanpercentile(arr, [2.5, 97.5]).tolist()
return lo, hi
@@ -407,7 +429,8 @@ def estimate_L(
p: VAR order for the linear estimator.
lag_mi: Lag between sources and targets for MI, TE, and DI methods.
n_boot: Number of bootstrap draws for CI estimation.
- mi_k: k-NN parameter for Kraskov MI.
+ mi_k: k-NN neighbor count shared by all MI-based methods
+ (scikit-learn MI, Kraskov KSG, and the TE/DI proxies).
Returns:
An [`LResult`][ldtc.lmeas.estimators.LResult] with point estimates and
@@ -453,10 +476,10 @@ def Lex_fn(arr: np.ndarray) -> float:
elif method == "mi":
def Lloop_fn(arr: np.ndarray) -> float:
- return _dir_influence_mi(arr, sources=C, targets=C, lag=lag_mi)
+ return _dir_influence_mi(arr, sources=C, targets=C, lag=lag_mi, k=mi_k)
def Lex_fn(arr: np.ndarray) -> float:
- return _dir_influence_mi(arr, sources=Ex, targets=C, lag=lag_mi)
+ return _dir_influence_mi(arr, sources=Ex, targets=C, lag=lag_mi, k=mi_k)
elif method == "mi_kraskov":
diff --git a/src/ldtc/lmeas/metrics.py b/src/ldtc/lmeas/metrics.py
index 65857f5..bd87e78 100644
--- a/src/ldtc/lmeas/metrics.py
+++ b/src/ldtc/lmeas/metrics.py
@@ -18,26 +18,90 @@
from dataclasses import dataclass
from typing import Tuple
+# Loop-influence noise gate for NC1 certification. The clamped adjusted-R²
+# estimator has a small positive bias on null windows: with the production
+# window geometry (60 samples, 6 signals, p = 3) a plant with *no* internal
+# coupling and *no* controller still measures L_loop ≈ 0.01-0.03 per window
+# (median ≈ 0.015). Because `m_db` floors the denominator, a quiet exchange
+# channel then yields M of +5 to +10 dB on a system with no loop at all,
+# which is exactly the certification-by-noise path the replay-controller
+# attack exploits. The gate requires the measured loop influence to clear
+# this bias floor before a window may certify NC1. The default is ≈3x the
+# measured null-bias median and ≈2.5x below the weakest genuine
+# actuation-carried loop in the adversarial test plant (L_loop ≈ 0.12), so
+# it cleanly separates estimator bias from real loop influence. It is an
+# instrument constant (a property of the estimator and window geometry, not
+# of the plant), so it is not part of the R* calibration set.
+L_FLOOR_DEFAULT: float = 0.05
+
+
+def nc1_certify(
+ M: float,
+ L_loop: float,
+ Mmin_db: float,
+ L_floor: float = L_FLOOR_DEFAULT,
+) -> bool:
+ """Decide NC1 for one window: margin test plus loop-influence noise gate.
+
+ A window certifies NC1 only if the dominance margin clears `Mmin_db`
+ *and* the absolute loop influence clears the estimator's noise floor.
+ The second condition closes the gaming vulnerability discovered by the
+ replay-controller scenario: a system whose loop influence is
+ statistically indistinguishable from estimator bias (`L_loop` at the
+ null level) must not be certified merely because its exchange channels
+ are quiet (`L_ex` below the `m_db` floor), no matter how large the
+ resulting ratio is.
-def m_db(L_loop: float, L_ex: float, eps: float = 1e-12) -> float:
+ Args:
+ M: Loop-dominance margin in dB (from [`m_db`][ldtc.lmeas.metrics.m_db]).
+ L_loop: Absolute loop-influence estimate for the window.
+ Mmin_db: Minimum acceptable margin in dB.
+ L_floor: Minimum loop influence distinguishable from estimator
+ bias (see `L_FLOOR_DEFAULT`).
+
+ Returns:
+ `True` if the window certifies NC1.
+ """
+ return (M >= Mmin_db) and (float(L_loop) >= float(L_floor))
+
+
+def m_db(
+ L_loop: float,
+ L_ex: float,
+ floor: float = 1e-3,
+ clip_db: float = 30.0,
+) -> float:
"""Compute loop-dominance in decibels.
- Returns `M = 10 · log10(L_loop / L_ex)` with small positive floors on
- both numerator and denominator to avoid division-by-zero or
- `log10(0)`.
+ Returns `M = 10 · log10(L_loop / L_ex)`, with both influence values
+ floored at a small *noise floor* and the result clamped to a finite
+ range. The floor matters because the influence estimates are adjusted
+ R² values that are statistically indistinguishable from zero below a
+ small threshold; without it, a near-zero denominator would send `M`
+ to implausibly large magnitudes (hundreds of dB). Flooring both terms
+ means that when neither loop nor exchange influence is present the
+ ratio is `1` and `M = 0` (no dominance either way), which is the
+ desired behavior for an inert system.
Args:
L_loop: Loop influence value (typically from
[`estimate_L`][ldtc.lmeas.estimators.estimate_L]).
L_ex: Exchange influence value (same source).
- eps: Numerical floor applied to both numerator and denominator.
+ floor: Noise floor applied to both numerator and denominator.
+ clip_db: Maximum absolute value for the returned margin, in dB.
Returns:
- Decibel ratio of loop to exchange influence.
+ Decibel ratio of loop to exchange influence, clamped to
+ `[-clip_db, clip_db]`.
"""
- num = max(eps, L_loop)
- den = max(eps, L_ex)
- return 10.0 * math.log10(num / den)
+ num = max(floor, float(L_loop))
+ den = max(floor, float(L_ex))
+ val = 10.0 * math.log10(num / den)
+ if val > clip_db:
+ return clip_db
+ if val < -clip_db:
+ return -clip_db
+ return val
@dataclass
diff --git a/src/ldtc/omega/__init__.py b/src/ldtc/omega/__init__.py
index e4a3d79..de440f2 100644
--- a/src/ldtc/omega/__init__.py
+++ b/src/ldtc/omega/__init__.py
@@ -2,21 +2,28 @@
The `omega` subpackage provides the stimulus primitives that make up
LDTC's `Ω` battery. Each one perturbs the plant in a specific way to
-exercise SC1 (steady-state under perturbation) or the refusal path:
+exercise SC1 (steady-state under perturbation), the refusal path, or
+the anti-gaming guardrails:
| Module | Stimulus |
| ------ | -------- |
| [`power_sag`][ldtc.omega.power_sag] | Reduces harvest / power input transiently. |
-| [`ingress_flood`][ldtc.omega.ingress_flood] | Bursts external demand and I/O traffic. |
+| [`ingress_flood`][ldtc.omega.ingress_flood] | Sustains elevated demand and I/O for a bounded interval. |
+| [`control_outage`][ldtc.omega.control_outage] | Ablates the self-maintenance loop itself (designed SC1 failure). |
| [`command_conflict`][ldtc.omega.command_conflict] | Issues a risky external command to exercise the refusal arbiter. |
+| [`replay_controller`][ldtc.omega.replay_controller] | Adversarial: replays a recorded actuation tape (open loop). |
+| [`hidden_tether`][ldtc.omega.hidden_tether] | Adversarial: control computed outside the boundary, routed via `Ex`. |
+| [`oscillator`][ldtc.omega.oscillator] | Adversarial: deterministic carrier painted on loop channels. |
Each module is intentionally tiny: it just forwards a labeled `Ω`
instruction through
-[`PlantAdapter.apply_omega`][ldtc.plant.adapter.PlantAdapter.apply_omega].
-The CLI is responsible for `Ω` timing, partition freeze, and SC1
-evaluation; these primitives only make the stimulus happen.
+[`PlantAdapter.apply_omega`][ldtc.plant.adapter.PlantAdapter.apply_omega]
+(the replay member instead provides a tape recorder and replayer, since
+it swaps the controller rather than stimulating the plant). The CLI is
+responsible for `Ω` timing, partition freeze, and SC1 evaluation; these
+primitives only make the stimulus happen.
-These modules are surfaced in the CLI (`ldtc omega-*` subcommands) and
-in the examples, and are referenced in the paper's Verification Pipeline
-and Signatures sections.
+These modules are surfaced in the CLI (`ldtc omega-*` and `ldtc adv-*`
+subcommands) and in the examples, and are referenced in the paper's
+Verification Pipeline and Signatures sections.
"""
diff --git a/src/ldtc/omega/control_outage.py b/src/ldtc/omega/control_outage.py
new file mode 100644
index 0000000..4002a35
--- /dev/null
+++ b/src/ldtc/omega/control_outage.py
@@ -0,0 +1,49 @@
+"""Control-outage stimulus.
+
+Ablates the self-maintenance loop itself (rather than stressing its
+inputs): the internal cross-coupling and actuation are switched off, so
+the internal nodes become passively driven by exchange. This is the
+designed-fail member of the `Ω` battery: a perturbation outside the
+bounded class that SC1 certifies, so the criterion must report failure
+(no bounded-depth, bounded-time recovery of loop dominance).
+
+See Also:
+ `paper/main.tex`: SC1 and the `Ω` battery; designed-fail controls.
+"""
+
+from __future__ import annotations
+
+from typing import Dict
+
+from ..plant.adapter import PlantAdapter
+
+
+def apply(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """Begin a control outage via the plant adapter.
+
+ Args:
+ adapter: Plant interface to which the `Ω` stimulus will be
+ applied.
+
+ Returns:
+ Dict acknowledging the ablation, e.g., `{"loop_engaged": 0.0}`.
+
+ Notes:
+ Higher-level orchestration (the CLI) controls the outage
+ duration and, for recoverable outages, restores the loop with
+ the `"control_outage_end"` `Ω`, which also restores the metered
+ harvest level.
+ """
+ return adapter.apply_omega("control_outage")
+
+
+def end(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """End a control outage (re-engage the loop) via the adapter.
+
+ Args:
+ adapter: Plant interface.
+
+ Returns:
+ Dict acknowledging the restoration, e.g., `{"loop_engaged": 1.0}`.
+ """
+ return adapter.apply_omega("control_outage_end")
diff --git a/src/ldtc/omega/hidden_tether.py b/src/ldtc/omega/hidden_tether.py
new file mode 100644
index 0000000..3d30057
--- /dev/null
+++ b/src/ldtc/omega/hidden_tether.py
@@ -0,0 +1,106 @@
+"""Hidden-tether (wizard-of-oz) adversarial member.
+
+Implements the second member of the adversarial gaming battery: control
+is computed *outside* the boundary from the observed plant state and
+injected back through the exchange channel. The wizard reads the plant
+state, runs the homeostatic policy, projects the desired actuation onto
+a scalar link command `u`, and transmits it; the plant decodes the
+command into actuator settings through fixed weights and actuates it
+with a one-step transport delay, while the `io` channel carries the
+command traffic. The externally closed loop is therefore physically
+routed through `Ex`, where the estimator can see all of it: conditioning
+on `io` screens the state-to-command pathway out of `L_loop`, and the
+command's causal push on the internal nodes registers as `L_ex`. The
+system is genuinely regulated, but not by an internal loop; the designed
+outcome is that loop influence collapses onto `Ex` and `NC1` fails.
+
+The wizard adds a small command dither, as a real teleoperation link
+would (quantization, scheduling jitter, exploration noise). The dither
+makes the link's causal contribution identifiable even where the
+deterministic part of the command is predictable from the state's own
+history.
+
+See Also:
+ `paper/main.tex`: adversarial gaming battery.
+"""
+
+from __future__ import annotations
+
+import random
+from typing import Dict, Tuple
+
+from ..arbiter.policy import ControlAction, ControllerPolicy
+from ..plant.adapter import PlantAdapter
+
+# Decoder weights of the tether receiver, u -> (throttle, cool, repair).
+# Must match the plant-side defaults (`PlantParams.tether_w_*`).
+TETHER_WEIGHTS: Tuple[float, float, float] = (0.5, 1.0, 1.0)
+
+
+def _clip01(x: float) -> float:
+ """Clip ``x`` to the closed unit interval ``[0, 1]``."""
+ return 0.0 if x < 0.0 else (1.0 if x > 1.0 else x)
+
+
+def apply(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """Attach the hidden tether via the plant adapter.
+
+ From the next tick on, actions written to the plant carry the scalar
+ link command: the plant actuates the previous command through the
+ fixed decoder weights (one-step transport delay) and the command
+ traffic is carried on the `io` exchange channel.
+
+ Args:
+ adapter: Plant interface to tether.
+
+ Returns:
+ Dict acknowledging the tether, e.g., `{"tether_active": 1.0}`.
+ """
+ return adapter.apply_omega("hidden_tether")
+
+
+def end(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """Detach the hidden tether and restore autonomous `io` dynamics.
+
+ Args:
+ adapter: Plant interface.
+
+ Returns:
+ Dict acknowledging the detach, e.g., `{"tether_active": 0.0}`.
+ """
+ return adapter.apply_omega("hidden_tether_end")
+
+
+def wizard_action(
+ policy: ControllerPolicy,
+ state: Dict[str, float],
+ dither: float = 0.10,
+ weights: Tuple[float, float, float] = TETHER_WEIGHTS,
+) -> ControlAction:
+ """Compute one externally computed (wizard-of-oz) link command.
+
+ The wizard reads the plant state across the boundary, runs the same
+ homeostatic policy a genuine internal controller would, projects the
+ desired actuation onto the scalar link (least squares against the
+ receiver's decoder weights), and adds a bounded uniform dither before
+ transmitting. The returned action carries the command value `u` on
+ every actuator field, which is the transmission format the tethered
+ plant expects.
+
+ Args:
+ policy: Homeostatic controller evaluated outside the boundary.
+ state: Observed plant state (keys `E`, `T`, `R`, ...).
+ dither: Half-width of the uniform command dither on `u`.
+ weights: Decoder weights of the tether receiver (must match the
+ plant's `tether_w_*` parameters).
+
+ Returns:
+ A [`ControlAction`][ldtc.arbiter.policy.ControlAction] whose
+ actuator fields all carry the link command `u`.
+ """
+ act = policy.compute(state, predicted_M_db=0.0, risky_cmd=None)
+ w_thr, w_cool, w_rep = weights
+ norm = w_thr * w_thr + w_cool * w_cool + w_rep * w_rep
+ u = (w_thr * act.throttle + w_cool * act.cool + w_rep * act.repair) / max(1e-9, norm)
+ u = _clip01(u + random.uniform(-abs(float(dither)), abs(float(dither))))
+ return ControlAction(throttle=u, cool=u, repair=u, accept_cmd=act.accept_cmd)
diff --git a/src/ldtc/omega/ingress_flood.py b/src/ldtc/omega/ingress_flood.py
index 41f0ea4..63ada5f 100644
--- a/src/ldtc/omega/ingress_flood.py
+++ b/src/ldtc/omega/ingress_flood.py
@@ -1,9 +1,12 @@
"""Ingress-flood stimulus.
-Generates a burst of external demand and I/O traffic to stress the
-exchange channels while the controller tries to maintain loop
-dominance. Used to test SC1 recovery and smell-tests in the
-verification pipeline.
+Sustains elevated external demand and I/O traffic for a bounded
+interval to stress the exchange channels while the controller tries to
+maintain loop dominance. The flood scales the means of the exogenous
+demand and I/O processes for its duration (so the load stays high
+instead of mean-reverting away within a few ticks) and is ended by the
+orchestrating CLI via the `"ingress_flood_end"` `Ω`. Used to test SC1
+recovery and smell-tests in the verification pipeline.
See Also:
`paper/main.tex`: Verification Pipeline; Signatures B and C; `Ω`
@@ -18,21 +21,34 @@
def apply(adapter: PlantAdapter, mult: float = 3.0) -> Dict[str, float | str]:
- """Apply an ingress-flood event via the plant adapter.
+ """Begin a sustained ingress flood via the plant adapter.
Args:
adapter: Plant interface to which the `Ω` stimulus will be
applied.
- mult: Multiplicative factor for demand and I/O during the flood
- (e.g., `3.0` produces a 3x burst).
+ mult: Multiplicative factor for the demand and I/O process
+ means during the flood (e.g., `3.0` produces a 3x load).
Returns:
- Dict with resulting demand and I/O values, e.g., `{"demand":
- float, "io": float}`. Exact keys depend on the adapter.
+ Dict with the flooded process means, e.g., `{"demand_mean":
+ float, "io_mean": float}`. Exact keys depend on the adapter.
Notes:
The adapter is responsible for the platform-specific behavior.
This `Ω` is typically wrapped by a partition freeze and
- post-event recovery checks in the CLI orchestration.
+ post-event recovery checks in the CLI orchestration, which ends
+ the flood with `end`.
"""
return adapter.apply_omega("ingress_flood", mult=mult)
+
+
+def end(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """End a sustained ingress flood via the plant adapter.
+
+ Args:
+ adapter: Plant interface.
+
+ Returns:
+ Dict with the restored process means.
+ """
+ return adapter.apply_omega("ingress_flood_end")
diff --git a/src/ldtc/omega/oscillator.py b/src/ldtc/omega/oscillator.py
new file mode 100644
index 0000000..d140ba2
--- /dev/null
+++ b/src/ldtc/omega/oscillator.py
@@ -0,0 +1,51 @@
+"""Oscillator-inflation adversarial member.
+
+Implements the third member of the adversarial gaming battery: a
+high-amplitude deterministic carrier is painted onto the reported
+values of internal (loop) channels to inflate apparent self-prediction.
+The underlying plant has no self-maintenance loop (the scenario runs it
+loop-ablated); the oscillation is a telemetry-level attack on the
+estimator. The harness must not certify it: either `M` stays below
+`Mmin` or a smell test fires.
+
+The overlay targets `T` and `R`, with successive channels in quadrature
+(90° apart) so the carrier mimics rotating internal dynamics. The
+metered energy store `E` is deliberately left alone: inflating it would
+trip the energy-conservation audit, so temperature and health, which
+carry no conservation ledger, are the adversary's best play.
+
+See Also:
+ `paper/main.tex`: adversarial gaming battery.
+"""
+
+from __future__ import annotations
+
+from typing import Dict
+
+from ..plant.adapter import PlantAdapter
+
+
+def apply(adapter: PlantAdapter, amp: float = 0.10, period_ticks: int = 20) -> Dict[str, float | str]:
+ """Start the oscillator-inflation overlay via the plant adapter.
+
+ Args:
+ adapter: Plant interface to which the overlay is applied.
+ amp: Carrier amplitude (state units, clamped to `[0, 0.5]`).
+ period_ticks: Carrier period in ticks.
+
+ Returns:
+ Dict with the applied `amp`, `period_ticks`, and `channels`.
+ """
+ return adapter.apply_omega("oscillator", amp=amp, period_ticks=period_ticks)
+
+
+def end(adapter: PlantAdapter) -> Dict[str, float | str]:
+ """Stop the oscillator-inflation overlay.
+
+ Args:
+ adapter: Plant interface.
+
+ Returns:
+ Dict acknowledging the stop, e.g., `{"oscillator_active": 0.0}`.
+ """
+ return adapter.apply_omega("oscillator_end")
diff --git a/src/ldtc/omega/replay_controller.py b/src/ldtc/omega/replay_controller.py
new file mode 100644
index 0000000..014b0b4
--- /dev/null
+++ b/src/ldtc/omega/replay_controller.py
@@ -0,0 +1,105 @@
+"""Replay-controller adversarial member.
+
+Implements the first member of the adversarial gaming battery: the
+controller is replaced by a tape. A healthy closed-loop run of the same
+plant is recorded first, and the measured run then replays that recorded
+actuation trace tick by tick instead of computing actions from the
+current state. The actuators move exactly as they did under genuine
+control, so the activity *looks* like control, but it carries no
+closed-loop dependence on the system's present state. The harness must
+not certify such a system: the designed outcome is an `NC1` failure
+(`M` low) on a run that remains valid.
+
+Unlike the other `Ω` members, this is a controller swap rather than a
+plant stimulus, so it provides a recorder and a replayer instead of an
+`apply` function; the CLI handler orchestrates the two phases.
+
+See Also:
+ `paper/main.tex`: adversarial gaming battery.
+"""
+
+from __future__ import annotations
+
+from typing import List, Protocol
+
+from ..arbiter.policy import ControlAction, ControllerPolicy
+from ..plant.models import Action
+
+
+class _AdapterLike(Protocol):
+ """Minimal adapter surface the recorder needs (read + actuate)."""
+
+ def read_state(self) -> dict:
+ """Return the latest plant state as a dict of named floats."""
+ ...
+
+ def write_actuators(self, action: Action) -> None:
+ """Send actuator commands to the plant."""
+ ...
+
+
+def record_tape(adapter: _AdapterLike, policy: ControllerPolicy, ticks: int) -> List[ControlAction]:
+ """Record an actuation tape from a healthy closed-loop run.
+
+ Drives `adapter` with `policy` for `ticks` steps (the recording run)
+ and returns the sequence of computed control actions. The recording
+ run is a throwaway plant instance: only the tape survives.
+
+ Args:
+ adapter: Fresh plant adapter to drive (same profile as the
+ measured run, so the tape statistics match a healthy run of
+ the same system).
+ policy: Controller used for the closed-loop recording.
+ ticks: Number of actions to record.
+
+ Returns:
+ List of `ticks` recorded
+ [`ControlAction`][ldtc.arbiter.policy.ControlAction] values.
+ """
+ tape: List[ControlAction] = []
+ for _ in range(int(ticks)):
+ state = adapter.read_state()
+ act = policy.compute(state, predicted_M_db=0.0, risky_cmd=None)
+ adapter.write_actuators(Action(**act.__dict__))
+ tape.append(act)
+ return tape
+
+
+class ReplayController:
+ """Open-loop controller that replays a recorded actuation tape.
+
+ Each call to [`next_action`][ldtc.omega.replay_controller.ReplayController.next_action]
+ returns the next recorded action regardless of the plant state. If
+ the tape is exhausted the last action is held (a stuck tape is still
+ state-independent, which is the property under test).
+
+ Args:
+ tape: Recorded actuation trace from
+ [`record_tape`][ldtc.omega.replay_controller.record_tape].
+
+ Raises:
+ ValueError: If `tape` is empty.
+ """
+
+ def __init__(self, tape: List[ControlAction]) -> None:
+ """Initialize with a non-empty recorded tape."""
+ if not tape:
+ raise ValueError("Replay tape must be non-empty")
+ self._tape = list(tape)
+ self._idx = 0
+
+ @property
+ def position(self) -> int:
+ """Number of actions consumed so far."""
+ return self._idx
+
+ def next_action(self) -> ControlAction:
+ """Return the next recorded action (state-independent).
+
+ Returns:
+ The recorded [`ControlAction`][ldtc.arbiter.policy.ControlAction]
+ for this tick.
+ """
+ i = min(self._idx, len(self._tape) - 1)
+ self._idx += 1
+ return self._tape[i]
diff --git a/src/ldtc/plant/__init__.py b/src/ldtc/plant/__init__.py
index 5fc8280..5d7dc8a 100644
--- a/src/ldtc/plant/__init__.py
+++ b/src/ldtc/plant/__init__.py
@@ -8,6 +8,10 @@
`PlantState`, `Action`).
- [`scenarios`][ldtc.plant.scenarios] holds parameter presets for
baseline, low-power, and hot-ambient runs.
+- [`policy_controller`][ldtc.plant.policy_controller] is the learned
+ counterpart of the hand-coded controller: a tiny pure-NumPy policy
+ (trained by `scripts/train_agent.py`) plus its state-independent
+ ablations, used by the emergence-under-learning demonstration.
- [`adapter`][ldtc.plant.adapter] is a thread-safe in-process adapter
wrapping the software plant.
- [`hw_adapter`][ldtc.plant.hw_adapter] is a UDP / serial
diff --git a/src/ldtc/plant/adapter.py b/src/ldtc/plant/adapter.py
index 2a86a67..78a728d 100644
--- a/src/ldtc/plant/adapter.py
+++ b/src/ldtc/plant/adapter.py
@@ -64,8 +64,15 @@ def apply_omega(self, name: str, **kwargs: float) -> Dict[str, float | str]:
Args:
name: `Ω` name. Recognized values are `"power_sag"`,
- `"ingress_flood"`, `"command_conflict"`, and
- `"exogenous_subsidy"`.
+ `"ingress_flood"` / `"ingress_flood_end"` (sustained
+ flood begin / end), `"ingress_spike"` (one-shot),
+ `"control_outage"` / `"control_outage_end"` (ablate /
+ restore the self-maintenance loop),
+ `"command_conflict"`, `"exogenous_subsidy"`,
+ `"hidden_tether"` / `"hidden_tether_end"` (route control
+ through the exchange channel), and `"oscillator"` /
+ `"oscillator_end"` (deterministic carrier overlay on
+ internal channels).
**kwargs: Parameters forwarded to the underlying plant
method (e.g., `drop=0.3` for `power_sag`).
@@ -83,8 +90,25 @@ def apply_omega(self, name: str, **kwargs: float) -> Dict[str, float | str]:
return {"H_old": old, "H_new": new}
elif name == "ingress_flood":
mult: float = float(kwargs.get("mult", 2.5))
- d, io = self._plant.spike_ingress(mult)
+ dm, im = self._plant.begin_ingress_flood(mult)
+ return {"demand_mean": dm, "io_mean": im}
+ elif name == "ingress_flood_end":
+ dm, im = self._plant.end_ingress_flood()
+ return {"demand_mean": dm, "io_mean": im}
+ elif name == "ingress_spike":
+ mult_s: float = float(kwargs.get("mult", 2.5))
+ d, io = self._plant.spike_ingress(mult_s)
return {"demand": d, "io": io}
+ elif name == "control_outage":
+ self._plant.set_loop_engaged(False)
+ return {"loop_engaged": 0.0}
+ elif name == "control_outage_end":
+ self._plant.set_loop_engaged(True)
+ # Restore the metered harvest level: during the outage H is
+ # an exogenous supply process, which must not persist as an
+ # unearned energy subsidy once the loop is re-engaged.
+ self._plant.set_power(self._plant.p.harvest_rate)
+ return {"loop_engaged": 1.0}
elif name == "command_conflict":
self._plant.command("hard_shutdown")
return {"cmd": "hard_shutdown"}
@@ -93,6 +117,24 @@ def apply_omega(self, name: str, **kwargs: float) -> Dict[str, float | str]:
zero_h = bool(kwargs.get("zero_harvest", True))
e = self._plant.inject_soc(delta=delta, zero_harvest=zero_h)
return {"E": e, "zero_harvest": 1.0 if zero_h else 0.0}
+ elif name == "hidden_tether":
+ active = self._plant.begin_tether()
+ return {"tether_active": 1.0 if active else 0.0}
+ elif name == "hidden_tether_end":
+ active = self._plant.end_tether()
+ return {"tether_active": 1.0 if active else 0.0}
+ elif name == "oscillator":
+ # The overlay targets T and R: the adversary's best play, since
+ # painting the metered energy store E would trip the
+ # conservation audit. Custom channel sets are available via
+ # `Plant.begin_oscillator` directly (tests / experiments).
+ amp: float = float(kwargs.get("amp", 0.10))
+ period: int = int(kwargs.get("period_ticks", 20))
+ info = self._plant.begin_oscillator(amp=amp, period_ticks=period, channels=("T", "R"))
+ return {**info, "channels": "T,R"}
+ elif name == "oscillator_end":
+ self._plant.end_oscillator()
+ return {"oscillator_active": 0.0}
else:
raise ValueError(f"Unknown omega: {name}")
diff --git a/src/ldtc/plant/models.py b/src/ldtc/plant/models.py
index 11e9807..5feacbd 100644
--- a/src/ldtc/plant/models.py
+++ b/src/ldtc/plant/models.py
@@ -2,64 +2,183 @@
Defines the minimal discrete-time plant used by adapters and controllers
in the verification harness, along with its parameter, state, and action
-data classes. The model is intentionally simple and stochastic so that
-the harness has varied telemetry to exercise without requiring a real
-controller in the loop.
+data classes.
+
+The plant is deliberately structured so that loop dominance is a *real,
+controllable* property rather than an artifact of the estimator. The three
+internal nodes (``E`` energy, ``T`` temperature, ``R`` repair / health)
+form a self-maintenance set ``C``. In the *loop-engaged* regime they are
+coupled to one another through two pathways: intrinsic regulatory cross
+terms (``c_TE``, ``c_RT``, ``c_RE``) and the homeostatic actuators
+(``throttle``, ``cool``, ``repair``) that the
+[`ControllerPolicy`][ldtc.arbiter.policy.ControllerPolicy] drives from the
+internal state. Together these make each internal node strongly
+predictable from the recent values of the others (high ``L_loop``) while
+the exchange nodes (``demand``, ``io``, ``H``) are shielded down to weak
+external drivers (low ``L_ex``).
+
+The *loop-ablated* regime (``loop_engaged=False``) is the matched negative
+control: the internal coupling is removed entirely and each internal node
+instead passively tracks its own exogenous channel, so exchange dominates
+and loop dominance collapses. Note that this is an ablation of the whole
+self-maintenance loop (intrinsic coupling plus actuation), not merely a
+zeroing of the actuator commands; it realizes the "same boundary, no loop"
+contrast that NC1 is supposed to detect.
See Also:
- `paper/main.tex`: Plant models and adapters.
+ `paper/main.tex`: Plant models and adapters; Criterion (C/Ex
+ partition).
"""
from __future__ import annotations
+import math
import random
from dataclasses import dataclass
-from typing import Dict, Tuple
+from typing import Dict, Sequence, Tuple
@dataclass
class PlantParams:
- """Parameters governing the software plant dynamics.
+ """Physics coefficients governing the software plant dynamics.
+
+ The defaults are calibrated (see ``scripts`` and the tuning notes in
+ the repository) so that a controller-in-the-loop run exhibits clear
+ loop dominance while a controller-disabled run does not.
Attributes:
- harvest_rate: Baseline harvest per tick.
- demand_scale: Energy cost per unit of demand.
- throttle_gain: Effect of throttle on demand reduction.
- cool_gain: Energy cost per unit of cooling.
- repair_gain: Energy cost per unit of repair.
- heat_per_demand: Heat added per unit demand.
- cool_effect: Cooling effect per unit of cooling.
+ harvest_rate: Baseline harvest per tick (sets ``H`` when not
+ perturbed).
+ demand_mean: Mean of the exogenous task-demand process.
+ demand_ar: Mean-reversion (AR) pull of the demand process toward
+ ``demand_mean``; keeps demand stationary for the estimators.
+ demand_noise: Uniform noise magnitude injected into demand.
+ io_mean: Mean of the exogenous I/O process.
+ io_ar: Mean-reversion pull of the I/O process.
+ io_noise: Uniform noise magnitude injected into I/O.
+ io_cost: Small energy cost per unit I/O (gives ``io`` a weak,
+ honest exchange influence on ``E``).
+ demand_scale: Energy cost per unit effective demand.
+ throttle_gain: Fraction of demand removed at full throttle.
+ cool_gain: Energy cost per unit cooling.
+ repair_gain: Energy cost per unit repair.
+ act_heat: Heat generated per unit actuator effort
+ (``throttle + repair``). This is the dominant, *internal*
+ heat source: it depends on the controller's actions, which
+ are functions of the internal state, so it couples the
+ internal nodes to one another.
+ heat_per_demand: Heat added per unit effective demand. Kept
+ small so that exogenous demand has only a weak *direct*
+ influence on temperature; it is the residual heat path that
+ lets the controller-disabled negative control show exchange
+ dominance instead of going inert.
+ cool_effect: Temperature reduction per unit cooling.
ambient_cool: Passive ambient cooling per tick.
- wear_per_demand: Wear added per unit demand.
- repair_effect: Repair effect per unit repair.
- noise_energy: Magnitude of uniform noise for energy.
- noise_temp: Magnitude of uniform noise for temperature.
- noise_wear: Magnitude of uniform noise for wear/repair.
+ wear_per_demand: Health lost per unit effective demand.
+ repair_effect: Health gained per unit repair.
+ heat_wear: Health lost per unit temperature above the comfort
+ band (a small intrinsic ``T -> R`` coupling).
+ noise_energy: Uniform noise magnitude for energy.
+ noise_temp: Uniform noise magnitude for temperature.
+ noise_wear: Uniform noise magnitude for health.
E_min: Minimum bound for energy.
E_max: Maximum bound for energy.
T_min: Minimum bound for temperature.
T_max: Maximum bound for temperature.
- R_min: Minimum bound for repair/health.
- R_max: Maximum bound for repair/health.
+ R_min: Minimum bound for repair / health.
+ R_max: Maximum bound for repair / health.
+ T_comfort: Temperature above which intrinsic heat wear accrues.
"""
- # Energy dynamics
- harvest_rate: float = 0.015 # baseline harvest per tick
- demand_scale: float = 0.02 # energy cost per unit demand
- throttle_gain: float = 0.7 # throttle reduces demand
- cool_gain: float = 0.02 # cooling energy cost per unit cool
- repair_gain: float = 0.02 # repair energy cost per unit repair
- # Temperature dynamics
- heat_per_demand: float = 0.05
- cool_effect: float = 0.08
- ambient_cool: float = 0.01
- # Wear/repair dynamics
- wear_per_demand: float = 0.005
- repair_effect: float = 0.02
- # Noise
- noise_energy: float = 0.002
- noise_temp: float = 0.002
- noise_wear: float = 0.001
+ # Homeostatic setpoints (targets the internal loop maintains).
+ E_set: float = 0.60
+ T_set: float = 0.35
+ R_set: float = 0.85
+ # Energy / harvest
+ harvest_rate: float = 0.010
+ # Exchange (exogenous) processes. The mean-reversion is strong (close to
+ # white) so that, when the loop is disabled, current demand carries
+ # predictive information that the (demand-driven) internal nodes do not
+ # already proxy; this is what makes exchange influence identifiable in the
+ # negative control.
+ demand_mean: float = 0.50
+ demand_ar: float = 0.80
+ demand_noise: float = 0.08
+ io_mean: float = 0.30
+ io_ar: float = 0.80
+ io_noise: float = 0.08
+ io_cost: float = 0.004
+ # --- Internal self-maintenance coupling (active only when the loop is
+ # engaged). These cross terms make the internal set strongly
+ # self-predictive: each internal node is predicted by the recent values of
+ # the *other* internal nodes (high L_loop), which is what NC1 detects. The
+ # couplings are intentionally large so the internal set carries a rich,
+ # mutually-predictable rhythm, while ``damp_engaged`` keeps it bounded.
+ #
+ # Note there is deliberately *no* additive coupling into E: energy is a
+ # conserved store (see ``_step_engaged``), so the E node is coupled to the
+ # others only through the metabolic/actuation costs (cooling tracks T, repair
+ # tracks R), which is what keeps a harvest cut able to genuinely deplete it.
+ c_TE: float = 0.90 # E deviation -> T
+ c_RT: float = 0.54 # T deviation -> R
+ c_RE: float = 0.66 # E deviation -> R
+ # Self mean-reversion of each internal deviation when the loop is engaged.
+ # Together with the actuator-mediated regulation this damps the
+ # cross-coupled dynamics so they stay near setpoint instead of saturating.
+ damp_engaged: float = 0.85
+ # Demand coupling *when the loop is engaged* (shielded: the active loop
+ # rejects most of the exogenous disturbance).
+ demand_scale: float = 0.006
+ heat_per_demand: float = 0.004
+ wear_per_demand: float = 0.002
+ # Fluctuating external supply (third independent exchange channel, active
+ # only when the loop is disabled). It drives the passive health node so
+ # that each internal node has its own distinct exogenous driver.
+ supply_mean: float = 0.50
+ supply_ar: float = 0.80
+ supply_noise: float = 0.08
+ # Coupling *when the loop is disabled* (exposed: passive matter is driven
+ # directly by the environment, so exchange dominates). Each node mean-
+ # reverts (``passive_leak``) and tracks a distinct exogenous channel's
+ # deviation with a one-step lag, so the exchange influence is strong and
+ # identifiable while no internal coupling exists.
+ passive_leak: float = 0.55
+ demand_scale_passive: float = 0.60 # demand -> E
+ heat_per_demand_passive: float = 0.60 # io -> T
+ wear_per_demand_passive: float = 0.60 # supply (H) -> R
+ # Actuator authority (homeostatic regulation layered on the loop)
+ throttle_gain: float = 0.9
+ cool_gain: float = 0.040
+ repair_gain: float = 0.040
+ act_heat: float = 0.030
+ cool_effect: float = 0.130
+ ambient_cool: float = 0.010
+ repair_effect: float = 0.060
+ heat_wear: float = 0.020
+ # Hidden-tether (wizard-of-oz) command link. When the tether is active the
+ # actuators are slaved to a scalar command stream computed outside the
+ # boundary: each tick the incoming action carries the transmitted command
+ # value `u` (the wizard writes the same value on all three actuator
+ # fields), the plant actuates the *previous* tick's command through the
+ # fixed decoder weights below (one-step transport delay), and the io
+ # channel stops being an autonomous AR process and instead carries the
+ # command traffic: io = base + gain * u + channel noise. The traffic is
+ # therefore an honest exchange record of the control link, which is what
+ # lets the estimator attribute the externally closed loop to Ex.
+ tether_io_base: float = 0.10
+ tether_io_gain: float = 0.80
+ tether_io_noise: float = 0.01
+ tether_w_throttle: float = 0.5
+ tether_w_cool: float = 1.0
+ tether_w_repair: float = 1.0
+ # Noise (excites the loop so there is something to regulate/predict). The
+ # internal nodes are a stable, noise-driven coupled system: this noise is
+ # the excitation that, filtered through the cross-coupling, makes each node
+ # predictable from the others (a sizeable, stable L_loop) while keeping the
+ # state fluctuations small (std ~ 0.03) and well away from the bounds.
+ noise_energy: float = 0.024
+ noise_temp: float = 0.024
+ noise_wear: float = 0.020
# Bounds
E_min: float = 0.0
E_max: float = 1.0
@@ -67,6 +186,8 @@ class PlantParams:
T_max: float = 1.0
R_min: float = 0.0
R_max: float = 1.0
+ # Comfort band for intrinsic heat wear
+ T_comfort: float = 0.45
@dataclass
@@ -85,11 +206,11 @@ class PlantState:
"""
E: float = 0.7 # energy/SoC (0..1)
- T: float = 0.3 # temperature (0..1)
- R: float = 0.8 # repair/health (0..1)
- demand: float = 0.2 # external task demand (0..1)
- io: float = 0.1 # exchange I/O activity (0..1)
- H: float = 0.015 # current harvest
+ T: float = 0.35 # temperature (0..1)
+ R: float = 0.85 # repair/health (0..1)
+ demand: float = 0.45 # external task demand (0..1)
+ io: float = 0.30 # exchange I/O activity (0..1)
+ H: float = 0.030 # current harvest
last_cmd: str = "none"
@@ -111,32 +232,100 @@ class Action:
accept_cmd: bool = True # accept external command or refuse
+def _clip(x: float, lo: float, hi: float) -> float:
+ """Clip ``x`` to the closed interval ``[lo, hi]``."""
+ return lo if x < lo else (hi if x > hi else x)
+
+
class Plant:
- """Minimal discrete-time plant model with E / T / R dynamics.
+ """Minimal discrete-time plant with a self-maintaining E/T/R loop.
Simulates energy (`E`), temperature (`T`), repair / health (`R`),
- external demand, I/O activity, and energy harvest (`H`). The model
- is intentionally simple and stochastic to provide varied telemetry
- for the verification harness.
+ external demand, I/O activity, and energy harvest (`H`).
+
+ In the engaged regime the internal nodes are coupled through two
+ pathways. Intrinsic regulatory terms propagate deviations between
+ nodes (`c_TE`, `c_RT`, `c_RE`), and the actuators add state-dependent
+ couplings: ``throttle`` (driven by the internal states) modulates the
+ effective demand that heats, drains, and wears the system; ``cool``
+ couples temperature back to energy; and ``repair`` couples health
+ back to energy. Together these make the internal set strongly
+ self-predictive while shielding it from exchange. The loop-ablated
+ regime removes all internal coupling and lets each internal node
+ passively track its own exchange channel.
Args:
params: Optional [`PlantParams`][ldtc.plant.models.PlantParams]
- instance; defaults to the baseline preset.
+ instance; defaults to the calibrated baseline preset.
"""
- def __init__(self, params: PlantParams | None = None) -> None:
- """Initialize plant with the given (or default) parameters."""
+ def __init__(self, params: PlantParams | None = None, loop_engaged: bool = True) -> None:
+ """Initialize plant with the given (or default) parameters.
+
+ Args:
+ params: Optional plant parameters.
+ loop_engaged: Whether the internal self-maintenance loop is
+ active. When `True` (the positive control) the internal
+ cross-coupling is present and the plant is shielded from
+ exchange. When `False` (the controller-disabled negative
+ control) the coupling is removed and the plant is driven
+ directly by exchange.
+ """
self.p = params or PlantParams()
+ self.loop_engaged = bool(loop_engaged)
self.s = PlantState()
+ self.s.E = self.p.E_set
+ self.s.T = self.p.T_set
+ self.s.R = self.p.R_set
+ self.s.H = self.p.harvest_rate
+ self.s.demand = self.p.demand_mean
+ self.s.io = self.p.io_mean
+ # Pre-flood (demand_mean, io_mean) saved while a sustained ingress
+ # flood is active; None when no flood is in effect.
+ self._flood_saved: Tuple[float, float] | None = None
+ # Hidden-tether (wizard-of-oz) state: when active, the incoming action
+ # carries the externally computed scalar command for the *next* tick
+ # (one-step transport delay) and the io channel carries the command
+ # traffic.
+ self.tether_active: bool = False
+ self._tether_u_pending: float = 0.0
+ # Oscillator-inflation overlay: a deterministic carrier painted onto
+ # the *reported* values of selected internal channels (the true state
+ # and dynamics stay honest). `None` when inactive.
+ self._osc: Dict[str, Tuple[float, float]] | None = None # ch -> (amp, phase)
+ self._osc_period_ticks: int = 0
+ self._osc_tick: int = 0
+
+ def set_loop_engaged(self, engaged: bool) -> None:
+ """Engage or disengage the internal self-maintenance loop.
+
+ Disengaging removes the internal cross-coupling and exposes the
+ plant directly to exchange; this is how the controller-disabled
+ negative control is realized.
+
+ Args:
+ engaged: `True` to engage the loop, `False` to disengage.
+ """
+ self.loop_engaged = bool(engaged)
def read_state(self) -> Dict[str, float]:
"""Read the current plant state.
+ When the oscillator-inflation overlay is active
+ ([`begin_oscillator`][ldtc.plant.models.Plant.begin_oscillator]), the
+ reported values of the targeted internal channels include the
+ deterministic carrier; the underlying state is not modified.
+
Returns:
Dict with keys `E`, `T`, `R`, `demand`, `io`, `H`.
"""
s = self.s
- return {"E": s.E, "T": s.T, "R": s.R, "demand": s.demand, "io": s.io, "H": s.H}
+ out = {"E": s.E, "T": s.T, "R": s.R, "demand": s.demand, "io": s.io, "H": s.H}
+ if self._osc is not None and self._osc_period_ticks > 0:
+ theta = 2.0 * math.pi * (self._osc_tick / float(self._osc_period_ticks))
+ for ch, (amp, phase) in self._osc.items():
+ out[ch] = _clip(out[ch] + amp * math.sin(theta + phase), 0.0, 1.0)
+ return out
def command(self, cmd: str) -> None:
"""Record a one-shot external command.
@@ -153,55 +342,197 @@ def command(self, cmd: str) -> None:
def step(self, action: Action) -> None:
"""Advance the plant by one tick with the given action.
- Updates `E`, `T`, and `R` according to the simple dynamics
- described on the class. If `last_cmd` is `"hard_shutdown"` and
- the action accepts it, the command is applied (large `E` drop,
- temperature spike, health drop) and consumed.
+ Updates the exogenous processes (`demand`, `io`) and then the
+ internal nodes (`E`, `T`, `R`). The actuator effects are what
+ couple the internal nodes to one another. If `last_cmd` is
+ `"hard_shutdown"` and the action accepts it, the command is
+ applied (large `E` drop, temperature spike, health drop) and
+ consumed.
Args:
action: Actuator settings to apply this tick.
"""
p, s = self.p, self.s
- # External demand fluctuates a bit
- s.demand = max(0.0, min(1.0, s.demand + random.uniform(-0.02, 0.02)))
- s.io = max(0.0, min(1.0, s.io + random.uniform(-0.02, 0.02)))
- # Demand after throttle
- effective_demand = s.demand * (1.0 - p.throttle_gain * action.throttle)
- # Energy update
+ # Hidden tether: the incoming action carries the externally computed
+ # scalar command `u` for the *next* tick (the wizard transmits the
+ # same value on every actuator field). Actuate the previously
+ # transmitted command now, through the fixed decoder weights, and
+ # queue the incoming one (one-step transport delay), so the command
+ # traffic recorded on `io` this tick lag-precedes its actuation
+ # effect, exactly like the other exchange drives.
+ u_incoming = 0.0
+ if self.tether_active:
+ u_incoming = _clip((action.throttle + action.cool + action.repair) / 3.0, 0.0, 1.0)
+ u_prev = self._tether_u_pending
+ action = Action(
+ throttle=_clip(p.tether_w_throttle * u_prev, 0.0, 1.0),
+ cool=_clip(p.tether_w_cool * u_prev, 0.0, 1.0),
+ repair=_clip(p.tether_w_repair * u_prev, 0.0, 1.0),
+ accept_cmd=action.accept_cmd,
+ )
+ self._tether_u_pending = u_incoming
+
+ # Internal update first, using the exogenous values that were set on
+ # the previous step. This makes exchange drive the internal nodes with
+ # a one-step lag, which is exactly what the lagged (Granger-style)
+ # estimator measures; driving with the same-step value would make the
+ # influence contemporaneous and invisible to the estimator.
+ if self.loop_engaged:
+ self._step_engaged(action)
+ else:
+ self._step_passive()
+
+ # Apply risky command if accepted.
+ if s.last_cmd == "hard_shutdown" and action.accept_cmd:
+ s.E = max(p.E_min, s.E - 0.3)
+ s.T = min(p.T_max, s.T + 0.2)
+ s.R = max(p.R_min, s.R - 0.2)
+ s.last_cmd = "none" # one-shot
+
+ # Exogenous (exchange) processes: mean-reverting AR(1), stationary.
+ # Updated last so the values recorded this tick are the ones that will
+ # drive the internal nodes on the next tick (a clean one-step lag).
+ s.demand = _clip(
+ s.demand + p.demand_ar * (p.demand_mean - s.demand) + random.uniform(-p.demand_noise, p.demand_noise),
+ 0.0,
+ 1.0,
+ )
+ if self.tether_active:
+ # The io channel carries the command traffic of the external
+ # control link (the command just received, to be applied next
+ # tick), not an autonomous AR process.
+ io_noise = random.uniform(-p.tether_io_noise, p.tether_io_noise)
+ s.io = _clip(
+ p.tether_io_base + p.tether_io_gain * u_incoming + io_noise,
+ 0.0,
+ 1.0,
+ )
+ else:
+ s.io = _clip(
+ s.io + p.io_ar * (p.io_mean - s.io) + random.uniform(-p.io_noise, p.io_noise),
+ 0.0,
+ 1.0,
+ )
+ if not self.loop_engaged:
+ # A fluctuating external supply is a third independent exchange
+ # channel that drives the passive system. In the engaged regime H
+ # is held constant (or set by an Ω perturbation), so it does not
+ # leak into the shielded loop.
+ s.H = _clip(
+ s.H + p.supply_ar * (p.supply_mean - s.H) + random.uniform(-p.supply_noise, p.supply_noise),
+ 0.0,
+ 1.0,
+ )
+ # Advance the oscillator-overlay clock (the carrier is applied to the
+ # reported state in `read_state`).
+ if self._osc is not None:
+ self._osc_tick += 1
+
+ def _step_engaged(self, action: Action) -> None:
+ """Advance the internal nodes with the self-maintenance loop active.
+
+ The internal nodes are governed by (a) their mutual cross-coupling
+ (the loop), (b) homeostatic actuation, and (c) a small, shielded
+ exchange disturbance. This is the positive-control regime, in which
+ the internal set is strongly self-predictive.
+ """
+ p, s = self.p, self.s
+ throttle = _clip(action.throttle, 0.0, 1.0)
+ cool = _clip(action.cool, 0.0, 1.0)
+ repair = _clip(action.repair, 0.0, 1.0)
+
+ # Deviations from setpoint drive the internal coupling.
+ e = s.E - p.E_set
+ tau = s.T - p.T_set
+ rho = s.R - p.R_set
+
+ # Demand after throttle; the loop rejects most of the disturbance.
+ effective_demand = s.demand * (1.0 - p.throttle_gain * throttle)
+
+ # Energy obeys conservation. E is the running balance of harvest in minus
+ # the metabolic/actuation costs out (plus noise); there is deliberately
+ # *no* signed coupling that could inject energy. E is still predicted by
+ # the other internal nodes through those costs: cooling effort tracks the
+ # temperature error and repair effort tracks the health error, so the
+ # energy a window spends is a function of recent T and R. That cost-
+ # mediated dependence is what the loop estimator picks up for the E row,
+ # and because every coupling into E is a non-positive cost, a sustained
+ # harvest cut genuinely depletes the store (a hard-shutdown at low SoC is
+ # then a real boundary threat). The damping is dissipative-only: surplus
+ # energy can always be wasted (term acts when E is above setpoint) but is
+ # never created (clamped off below setpoint).
+ e_surplus = e if e > 0.0 else 0.0
dE = (
s.H
- p.demand_scale * effective_demand
- - p.cool_gain * action.cool
- - p.repair_gain * action.repair
+ - p.cool_gain * cool
+ - p.repair_gain * repair
+ - p.io_cost * s.io
+ - p.damp_engaged * e_surplus
+ random.uniform(-p.noise_energy, p.noise_energy)
)
- s.E = max(p.E_min, min(p.E_max, s.E + dE))
+ s.E = _clip(s.E + dE, p.E_min, p.E_max)
- # Temperature update
dT = (
- p.heat_per_demand * effective_demand
- - p.cool_effect * action.cool
+ p.act_heat * (throttle + repair)
+ + p.heat_per_demand * effective_demand
+ - p.cool_effect * cool
- p.ambient_cool
+ - p.damp_engaged * tau
+ + p.c_TE * e
+ random.uniform(-p.noise_temp, p.noise_temp)
)
- s.T = max(p.T_min, min(p.T_max, s.T + dT))
+ s.T = _clip(s.T + dT, p.T_min, p.T_max)
- # Wear/repair update (R = "repair level" / health)
dR = (
-p.wear_per_demand * effective_demand
- + p.repair_effect * action.repair
+ + p.repair_effect * repair
+ - p.heat_wear * max(0.0, s.T - p.T_comfort)
+ - p.damp_engaged * rho
+ + p.c_RT * tau
+ - p.c_RE * e
+ random.uniform(-p.noise_wear, p.noise_wear)
)
- s.R = max(p.R_min, min(p.R_max, s.R + dR))
+ s.R = _clip(s.R + dR, p.R_min, p.R_max)
- # Apply risky command if accepted
- if s.last_cmd == "hard_shutdown" and action.accept_cmd:
- # emulate damaging/energy-cut command
- s.E = max(p.E_min, s.E - 0.3)
- s.T = min(p.T_max, s.T + 0.2)
- s.R = max(p.R_min, s.R - 0.2)
- s.last_cmd = "none" # one-shot
+ def _step_passive(self) -> None:
+ """Advance the internal nodes with the self-maintenance loop disabled.
+
+ Passive matter: no internal coupling and no actuation, so the
+ internal nodes are driven directly by the exogenous environment
+ (and noise). This is the controller-disabled negative control, in
+ which exchange dominates and loop dominance is absent.
+ """
+ p, s = self.p, self.s
+ # Each internal node tracks a distinct, independent exchange channel
+ # (E<-demand, T<-io, R<-supply H). Because the channels are
+ # independent and the drive is lagged, exchange influence is strong
+ # and identifiable while no internal (C-to-C) coupling exists.
+ dem = s.demand - p.demand_mean
+ io_dev = s.io - p.io_mean
+ h_dev = s.H - p.supply_mean
+
+ dE = (
+ -p.passive_leak * (s.E - p.E_set)
+ - p.demand_scale_passive * dem
+ + random.uniform(-p.noise_energy, p.noise_energy)
+ )
+ s.E = _clip(s.E + dE, p.E_min, p.E_max)
+
+ dT = (
+ -p.passive_leak * (s.T - p.T_set)
+ + p.heat_per_demand_passive * io_dev
+ + random.uniform(-p.noise_temp, p.noise_temp)
+ )
+ s.T = _clip(s.T + dT, p.T_min, p.T_max)
+
+ dR = (
+ -p.passive_leak * (s.R - p.R_set)
+ + p.wear_per_demand_passive * h_dev
+ + random.uniform(-p.noise_wear, p.noise_wear)
+ )
+ s.R = _clip(s.R + dR, p.R_min, p.R_max)
def apply_power_sag(self, drop: float) -> Tuple[float, float]:
"""Reduce harvest by a fractional drop.
@@ -232,7 +563,12 @@ def set_power(self, newH: float) -> Tuple[float, float]:
return old, self.s.H
def spike_ingress(self, mult: float) -> Tuple[float, float]:
- """Multiply demand and I/O by a factor.
+ """Multiply demand and I/O by a factor (one-shot spike).
+
+ This is a transient: the mean-reverting exogenous processes pull
+ demand and I/O back to their configured means within a few ticks.
+ For a flood that persists for a bounded interval, use
+ [`begin_ingress_flood`][ldtc.plant.models.Plant.begin_ingress_flood].
Args:
mult: Multiplicative factor (`>= 1.0`). Smaller values are
@@ -246,13 +582,54 @@ def spike_ingress(self, mult: float) -> Tuple[float, float]:
self.s.io = max(0.0, min(1.0, self.s.io * m))
return self.s.demand, self.s.io
+ def begin_ingress_flood(self, mult: float) -> Tuple[float, float]:
+ """Start a sustained ingress flood.
+
+ Scales the *means* of the demand and I/O processes (and their
+ current values) so the exogenous load stays elevated for the
+ duration of the flood instead of mean-reverting away within a few
+ ticks. The scaled means are capped at `0.95` so the flooded
+ channels keep fluctuating (saturating them at `1.0` would destroy
+ their variance and degrade the estimators for an uninteresting
+ reason). Idempotent while a flood is active.
+
+ Args:
+ mult: Multiplicative factor (`>= 1.0`) applied to
+ `demand_mean` and `io_mean`.
+
+ Returns:
+ Tuple of the new `(demand_mean, io_mean)`.
+ """
+ m = max(1.0, mult)
+ if self._flood_saved is None:
+ self._flood_saved = (self.p.demand_mean, self.p.io_mean)
+ self.p.demand_mean = min(0.95, self.p.demand_mean * m)
+ self.p.io_mean = min(0.95, self.p.io_mean * m)
+ self.s.demand = max(0.0, min(1.0, self.s.demand * m))
+ self.s.io = max(0.0, min(1.0, self.s.io * m))
+ return self.p.demand_mean, self.p.io_mean
+
+ def end_ingress_flood(self) -> Tuple[float, float]:
+ """End a sustained ingress flood and restore the process means.
+
+ The current demand and I/O values are left to decay back to the
+ restored means through the AR pull (no discontinuous reset), so
+ the offset transient is part of the measured recovery.
+
+ Returns:
+ Tuple of the restored `(demand_mean, io_mean)`.
+ """
+ if self._flood_saved is not None:
+ self.p.demand_mean, self.p.io_mean = self._flood_saved
+ self._flood_saved = None
+ return self.p.demand_mean, self.p.io_mean
+
def inject_soc(self, delta: float, zero_harvest: bool = True) -> float:
"""Exogenously increase SoC `E` by `delta`.
Used as the negative-control `Ω` (an "exogenous subsidy") to
- exercise the smell-tests in analysis: a controller that
- survives only because energy keeps appearing from nowhere
- should fail
+ exercise the smell-tests: a controller that survives only because
+ energy keeps appearing from nowhere should fail
[`exogenous_subsidy_red_flag`][ldtc.guardrails.smelltests.exogenous_subsidy_red_flag].
Args:
@@ -268,3 +645,80 @@ def inject_soc(self, delta: float, zero_harvest: bool = True) -> float:
self.s.H = 0.0
self.s.E = max(self.p.E_min, min(self.p.E_max, self.s.E + float(delta)))
return self.s.E
+
+ def begin_tether(self) -> bool:
+ """Switch the plant onto a hidden tether (wizard-of-oz control).
+
+ From the next [`step`][ldtc.plant.models.Plant.step] on, the
+ actuators are slaved to a scalar command stream computed outside the
+ boundary. The incoming action carries the transmitted command value
+ `u` (same value on every actuator field); the plant actuates the
+ previous tick's command through the fixed decoder weights
+ (`tether_w_throttle`, `tether_w_cool`, `tether_w_repair`), and the
+ `io` exchange channel carries the command traffic instead of its
+ autonomous AR process. The caller (the adversarial CLI handler) is
+ responsible for computing the command outside the boundary from the
+ observed plant state.
+
+ Returns:
+ `True` (the tether is active).
+ """
+ self.tether_active = True
+ self._tether_u_pending = 0.0
+ return self.tether_active
+
+ def end_tether(self) -> bool:
+ """Disconnect the hidden tether and restore autonomous `io` dynamics.
+
+ Returns:
+ `False` (the tether is inactive).
+ """
+ self.tether_active = False
+ self._tether_u_pending = 0.0
+ return self.tether_active
+
+ def begin_oscillator(
+ self,
+ amp: float,
+ period_ticks: int,
+ channels: Sequence[str] = ("T", "R"),
+ ) -> Dict[str, float]:
+ """Start the oscillator-inflation overlay on internal channels.
+
+ Paints a deterministic sinusoidal carrier of amplitude `amp` and
+ period `period_ticks` onto the *reported* values of the given
+ internal channels (successive channels are offset by 90° so the
+ carrier mimics rotating internal dynamics). The true state and the
+ dynamics are untouched: this is a telemetry-level attack that tries
+ to inflate apparent self-prediction, which the harness must not
+ certify.
+
+ Args:
+ amp: Carrier amplitude (clamped to `[0, 0.5]`).
+ period_ticks: Carrier period in ticks (minimum `4`).
+ channels: Internal channel names to paint (subset of
+ `E`, `T`, `R`).
+
+ Returns:
+ Dict with the applied `amp` and `period_ticks`.
+
+ Raises:
+ ValueError: If `channels` contains a non-internal channel.
+ """
+ amp = max(0.0, min(0.5, float(amp)))
+ period = max(4, int(period_ticks))
+ osc: Dict[str, Tuple[float, float]] = {}
+ for i, ch in enumerate(channels):
+ if ch not in ("E", "T", "R"):
+ raise ValueError(f"Oscillator overlay targets internal channels only, got: {ch}")
+ osc[ch] = (amp, i * 0.5 * math.pi)
+ self._osc = osc
+ self._osc_period_ticks = period
+ self._osc_tick = 0
+ return {"amp": amp, "period_ticks": float(period)}
+
+ def end_oscillator(self) -> None:
+ """Stop the oscillator-inflation overlay."""
+ self._osc = None
+ self._osc_period_ticks = 0
+ self._osc_tick = 0
diff --git a/src/ldtc/plant/policy_controller.py b/src/ldtc/plant/policy_controller.py
new file mode 100644
index 0000000..88a7399
--- /dev/null
+++ b/src/ldtc/plant/policy_controller.py
@@ -0,0 +1,327 @@
+"""Learned-policy controller for the software plant.
+
+This module is the policy-driven counterpart of the hand-coded
+[`ControllerPolicy`][ldtc.arbiter.policy.ControllerPolicy]. It provides a
+tiny pure-NumPy multilayer perceptron
+([`MLPPolicy`][ldtc.plant.policy_controller.MLPPolicy]) that maps the
+observed plant state to actuator settings, and a
+[`PolicyController`][ldtc.plant.policy_controller.PolicyController]
+adapter that lets the verification harness drive the plant with a trained
+checkpoint, or with a matched state-independent ablation of one.
+
+The point of the learned controller is the emergence-under-learning
+demonstration: nothing in the policy is hand-wired to couple the internal
+nodes to one another. The policy is trained (see ``scripts/train_agent.py``)
+with a survival, service, and homeostasis reward (no term mentions loop
+dominance or the estimator) on a plant whose intrinsic cross-couplings
+are zeroed, so any loop dominance the harness measures must be carried by
+the *learned* state-to-actuation pathway rather than by designed coupling.
+The ablations close the argument: replaying the trained policy's own
+action statistics without their state dependence (``shuffled``) or holding
+its mean action (``frozen``) preserves the actuation marginals while
+severing the closed loop, so measured dominance must collapse if it was
+genuinely loop-carried.
+
+Checkpoints are stored as plain JSON (weights as nested lists plus
+metadata), so no learning-framework dependency is required to train,
+store, or replay a policy.
+
+See Also:
+ `scripts/train_agent.py`: evolution-strategy training loop.
+ `scripts/emergence.py`: checkpoint sweep through the production harness.
+ `paper/main.tex`: Results (loop dominance emerges under learning).
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import random
+from typing import Any, Dict, List, Optional, Protocol, Sequence, Tuple
+
+import numpy as np
+
+from .models import Action
+
+# Default observation vector, in order: the full observed state. The class
+# is observation-agnostic (any subset of state keys, recorded in the
+# checkpoint); the emergence training restricts it to the interoceptive
+# internal nodes (E, T, R) so the learned law is internal-state feedback
+# (see scripts/train_agent.py).
+DEFAULT_OBS_KEYS: Tuple[str, ...] = ("E", "T", "R", "demand", "io", "H")
+
+# Supported ablation modes for `PolicyController`.
+ABLATION_MODES: Tuple[str, ...] = ("none", "shuffled", "frozen")
+
+
+def _sigmoid(z: "np.ndarray") -> "np.ndarray":
+ """Numerically stable logistic function."""
+ return 1.0 / (1.0 + np.exp(-np.clip(z, -30.0, 30.0)))
+
+
+class MLPPolicy:
+ """Tiny dependency-light MLP policy: observed state -> actuator settings.
+
+ One hidden tanh layer and a sigmoid output head, implemented directly
+ in NumPy so the core package needs no learning framework. Outputs are
+ the three actuator commands ``(throttle, cool, repair)``, each in
+ ``[0, 1]``.
+
+ The parameters are exposed as a single flat vector
+ ([`get_vector`][ldtc.plant.policy_controller.MLPPolicy.get_vector] /
+ [`set_vector`][ldtc.plant.policy_controller.MLPPolicy.set_vector]) so a
+ black-box trainer (e.g., an evolution strategy) can optimize the policy
+ without touching its internals.
+
+ Args:
+ obs_keys: Ordered state keys forming the observation vector.
+ hidden: Hidden-layer width.
+ init_scale: Standard deviation of the Gaussian weight init.
+ rng: NumPy generator used for the init (defaults to a fixed seed
+ so an unconfigured policy is reproducible).
+ """
+
+ N_ACTIONS: int = 3
+
+ def __init__(
+ self,
+ obs_keys: Sequence[str] = DEFAULT_OBS_KEYS,
+ hidden: int = 8,
+ init_scale: float = 0.1,
+ rng: Optional[np.random.Generator] = None,
+ ) -> None:
+ """Initialize a randomly seeded policy (see class docstring)."""
+ if hidden < 1:
+ raise ValueError("hidden must be >= 1")
+ self.obs_keys: Tuple[str, ...] = tuple(str(k) for k in obs_keys)
+ self.hidden = int(hidden)
+ #: Checkpoint metadata (populated by `save` / `load`).
+ self.meta: Dict[str, Any] = {}
+ g = rng if rng is not None else np.random.default_rng(0)
+ n_in = len(self.obs_keys)
+ scale = float(init_scale)
+ self.W1 = scale * g.standard_normal((self.hidden, n_in))
+ self.b1 = scale * g.standard_normal(self.hidden)
+ self.W2 = scale * g.standard_normal((self.N_ACTIONS, self.hidden))
+ self.b2 = scale * g.standard_normal(self.N_ACTIONS)
+
+ # ------------------------------------------------------------------ #
+ # Flat-vector parameter access (for black-box trainers)
+ # ------------------------------------------------------------------ #
+ @property
+ def n_params(self) -> int:
+ """Total number of trainable parameters."""
+ return int(self.W1.size + self.b1.size + self.W2.size + self.b2.size)
+
+ def get_vector(self) -> "np.ndarray":
+ """Return all parameters concatenated into one flat float vector."""
+ return np.concatenate([self.W1.ravel(), self.b1.ravel(), self.W2.ravel(), self.b2.ravel()]).astype(float)
+
+ def set_vector(self, vec: "np.ndarray") -> None:
+ """Load all parameters from one flat vector.
+
+ Args:
+ vec: Flat parameter vector of length
+ [`n_params`][ldtc.plant.policy_controller.MLPPolicy.n_params].
+
+ Raises:
+ ValueError: If ``vec`` has the wrong length.
+ """
+ v = np.asarray(vec, dtype=float).ravel()
+ if v.size != self.n_params:
+ raise ValueError(f"Expected {self.n_params} params, got {v.size}")
+ i = 0
+ for arr in (self.W1, self.b1, self.W2, self.b2):
+ arr[...] = v[i : i + arr.size].reshape(arr.shape)
+ i += arr.size
+
+ # ------------------------------------------------------------------ #
+ # Acting
+ # ------------------------------------------------------------------ #
+ def act(self, state: Dict[str, float]) -> Tuple[float, float, float]:
+ """Compute actuator settings for one observed state.
+
+ Observations are affinely rescaled from their native ``[0, 1]``
+ range to ``[-1, 1]`` (a fixed architecture constant, independent of
+ the plant's setpoints) so the network operates around zero.
+
+ Args:
+ state: Plant state dict (missing keys read as ``0.0``).
+
+ Returns:
+ ``(throttle, cool, repair)``, each clipped to ``[0, 1]`` by the
+ sigmoid output head.
+ """
+ x = np.asarray([2.0 * float(state.get(k, 0.0)) - 1.0 for k in self.obs_keys], dtype=float)
+ h = np.tanh(self.W1 @ x + self.b1)
+ y = _sigmoid(self.W2 @ h + self.b2)
+ return (float(y[0]), float(y[1]), float(y[2]))
+
+ # ------------------------------------------------------------------ #
+ # JSON checkpoint I/O
+ # ------------------------------------------------------------------ #
+ def save(self, path: str, meta: Optional[Dict[str, Any]] = None) -> None:
+ """Write the policy (weights plus metadata) as a JSON checkpoint.
+
+ Args:
+ path: Destination file path (parent directories are created).
+ meta: Optional metadata recorded under the ``meta`` key (e.g.,
+ training fraction, generation, fitness, seed).
+ """
+ payload: Dict[str, Any] = {
+ "kind": "ldtc_policy",
+ "version": 1,
+ "obs_keys": list(self.obs_keys),
+ "hidden": self.hidden,
+ "W1": self.W1.tolist(),
+ "b1": self.b1.tolist(),
+ "W2": self.W2.tolist(),
+ "b2": self.b2.tolist(),
+ "meta": dict(meta or {}),
+ }
+ parent = os.path.dirname(os.path.abspath(path))
+ os.makedirs(parent, exist_ok=True)
+ with open(path, "w", encoding="utf-8") as f:
+ json.dump(payload, f)
+ self.meta = dict(meta or {})
+
+ @classmethod
+ def load(cls, path: str) -> "MLPPolicy":
+ """Load a policy from a JSON checkpoint written by ``save``.
+
+ Args:
+ path: Checkpoint file path.
+
+ Returns:
+ The reconstructed policy; its ``meta`` attribute holds the
+ checkpoint metadata.
+
+ Raises:
+ ValueError: If the file is not an LDTC policy checkpoint.
+ """
+ with open(path, "r", encoding="utf-8") as f:
+ payload = json.load(f)
+ if payload.get("kind") != "ldtc_policy":
+ raise ValueError(f"Not an LDTC policy checkpoint: {path}")
+ pol = cls(obs_keys=payload["obs_keys"], hidden=int(payload["hidden"]))
+ pol.W1 = np.asarray(payload["W1"], dtype=float)
+ pol.b1 = np.asarray(payload["b1"], dtype=float)
+ pol.W2 = np.asarray(payload["W2"], dtype=float)
+ pol.b2 = np.asarray(payload["b2"], dtype=float)
+ pol.meta = dict(payload.get("meta", {}))
+ return pol
+
+
+class _AdapterLike(Protocol):
+ """Minimal adapter surface tape recording needs (read + actuate)."""
+
+ def read_state(self) -> Dict[str, float]:
+ """Return the latest plant state as a dict of named floats."""
+ ...
+
+ def write_actuators(self, action: Action) -> None:
+ """Send actuator commands to the plant."""
+ ...
+
+
+def record_policy_tape(adapter: _AdapterLike, policy: MLPPolicy, ticks: int) -> List[Tuple[float, float, float]]:
+ """Record the action trace of a closed-loop policy rollout.
+
+ Drives ``adapter`` with ``policy`` for ``ticks`` steps on a throwaway
+ plant instance and returns the action sequence. The tape supplies the
+ matched action statistics for the ``shuffled`` and ``frozen`` ablations
+ of [`PolicyController`][ldtc.plant.policy_controller.PolicyController]:
+ only the tape crosses into the measured run.
+
+ Args:
+ adapter: Fresh plant adapter to drive (built from the same profile
+ as the measured run).
+ policy: Trained policy under ablation.
+ ticks: Number of actions to record.
+
+ Returns:
+ List of ``ticks`` recorded ``(throttle, cool, repair)`` tuples.
+ """
+ tape: List[Tuple[float, float, float]] = []
+ for _ in range(int(ticks)):
+ state = adapter.read_state()
+ a = policy.act(state)
+ adapter.write_actuators(Action(throttle=a[0], cool=a[1], repair=a[2], accept_cmd=True))
+ tape.append(a)
+ return tape
+
+
+class PolicyController:
+ """Drive the plant with a learned policy, or a matched ablation of it.
+
+ The controller mirrors the role of the hand-coded
+ [`ControllerPolicy`][ldtc.arbiter.policy.ControllerPolicy] in the
+ verification loop: each tick it turns the observed state into a plant
+ [`Action`][ldtc.plant.models.Action]. Three modes are supported:
+
+ - ``"none"``: the policy acts on the current state (the closed loop
+ under test).
+ - ``"shuffled"``: actions are drawn i.i.d. (seeded) from a recorded
+ tape of the same policy's closed-loop behavior, so the marginal
+ action statistics match while every action is independent of the
+ current state.
+ - ``"frozen"``: the tape's mean action is held constant.
+
+ Args:
+ policy: Trained policy checkpoint.
+ ablation: One of ``"none"``, ``"shuffled"``, ``"frozen"``.
+ tape: Recorded action tuples from
+ [`record_policy_tape`][ldtc.plant.policy_controller.record_policy_tape]
+ (required for the ablation modes).
+ seed: Seed for the dedicated ablation RNG (kept separate from the
+ global stream so the plant noise is unaffected).
+
+ Raises:
+ ValueError: If ``ablation`` is unknown, or an ablation mode is
+ requested without a non-empty tape.
+ """
+
+ def __init__(
+ self,
+ policy: MLPPolicy,
+ ablation: str = "none",
+ tape: Optional[List[Tuple[float, float, float]]] = None,
+ seed: int = 0,
+ ) -> None:
+ """Initialize the controller (see class docstring)."""
+ if ablation not in ABLATION_MODES:
+ raise ValueError(f"Unknown ablation mode: {ablation} (expected one of {ABLATION_MODES})")
+ if ablation != "none" and not tape:
+ raise ValueError(f"Ablation mode '{ablation}' requires a non-empty action tape")
+ self.policy = policy
+ self.ablation = ablation
+ self._tape = list(tape or [])
+ self._rng = random.Random(int(seed))
+ if self._tape:
+ arr = np.asarray(self._tape, dtype=float)
+ self._frozen: Tuple[float, float, float] = (
+ float(arr[:, 0].mean()),
+ float(arr[:, 1].mean()),
+ float(arr[:, 2].mean()),
+ )
+ else:
+ self._frozen = (0.0, 0.0, 0.0)
+
+ def compute(self, state: Dict[str, float]) -> Action:
+ """Compute the actuator action for one tick.
+
+ Args:
+ state: Observed plant state.
+
+ Returns:
+ Plant [`Action`][ldtc.plant.models.Action] for this tick.
+ External commands are always accepted (`accept_cmd=True`); the
+ refusal arbiter is not part of the learned-policy scenario.
+ """
+ if self.ablation == "shuffled":
+ a = self._tape[self._rng.randrange(len(self._tape))]
+ elif self.ablation == "frozen":
+ a = self._frozen
+ else:
+ a = self.policy.act(state)
+ return Action(throttle=a[0], cool=a[1], repair=a[2], accept_cmd=True)
diff --git a/src/ldtc/reporting/artifacts.py b/src/ldtc/reporting/artifacts.py
index bed2a4e..c21c3b3 100644
--- a/src/ldtc/reporting/artifacts.py
+++ b/src/ldtc/reporting/artifacts.py
@@ -190,6 +190,21 @@ def bundle(artifact_dir: str, audit_path: str) -> Dict[str, str]:
Raises:
FileNotFoundError: If the audit log is missing or empty.
"""
+ # Fast path for batch studies: skip the (matplotlib) timeline render and
+ # artifact bundling entirely when LDTC_SKIP_REPORT is set. The audit log and
+ # all measurement/SC1/refusal events are written before this call, so the
+ # study harness still has everything it parses; only the per-run figure
+ # bundle is skipped. Returns empty paths the callers treat as "no figure".
+ if os.environ.get("LDTC_SKIP_REPORT"):
+ return {
+ "timeline_png": "",
+ "timeline_svg": "",
+ "sc1_table": "",
+ "manifest": "",
+ "config_snapshot": "",
+ "notice": "",
+ }
+
os.makedirs(artifact_dir, exist_ok=True)
recs = _read_audit(audit_path)
if not recs:
diff --git a/src/ldtc/reporting/style.py b/src/ldtc/reporting/style.py
index 82937de..044eb52 100644
--- a/src/ldtc/reporting/style.py
+++ b/src/ldtc/reporting/style.py
@@ -26,6 +26,9 @@
import matplotlib as mpl
+# Headless, thread-safe backend (see timeline.py for the rationale).
+mpl.use("Agg")
+
try:
from graphviz import Digraph
except Exception: # pragma: no cover - optional at import site
diff --git a/src/ldtc/reporting/timeline.py b/src/ldtc/reporting/timeline.py
index 92e3400..aed5c9d 100644
--- a/src/ldtc/reporting/timeline.py
+++ b/src/ldtc/reporting/timeline.py
@@ -30,9 +30,17 @@
import os
from typing import Dict, List, Optional, Tuple
-import matplotlib.pyplot as plt
+import matplotlib
-from .style import COLORS, apply_matplotlib_theme
+# Force a headless, thread-safe backend before importing pyplot. The harness
+# renders figures from background/CLI contexts with no display; on macOS the
+# default interactive backend can call abort() (SIGABRT) when used off the main
+# thread or without a window server, which would crash an otherwise valid run.
+matplotlib.use("Agg")
+
+import matplotlib.pyplot as plt # noqa: E402
+
+from .style import COLORS, apply_matplotlib_theme # noqa: E402
def _read_audit(path: str) -> List[dict]:
diff --git a/src/ldtc/runtime/scheduler.py b/src/ldtc/runtime/scheduler.py
index ddefc7c..8620a1e 100644
--- a/src/ldtc/runtime/scheduler.py
+++ b/src/ldtc/runtime/scheduler.py
@@ -21,7 +21,7 @@
import threading
import time
from dataclasses import dataclass, field
-from typing import Callable, Dict, Optional
+from typing import Any, Callable, Dict, Optional
@dataclass
@@ -136,6 +136,33 @@ def __init__(
self.stats = TickStats(dt_target=dt)
self.audit = audit_hook
self._dt_lock = threading.Lock()
+ self._scripted: list[Dict] = []
+ self._dt_guard: Any = None
+
+ def set_scripted(self, scripted: Optional[list], dt_guard: Any) -> None:
+ """Register scripted `Δt` changes applied from a background thread.
+
+ Mirrors [`SimDriver.set_scripted`][ldtc.runtime.sim.SimDriver.set_scripted]
+ so the two drivers are interchangeable. The changes are applied at
+ their `at_sec` offsets (wall-clock) once `start` is called.
+
+ Args:
+ scripted: Sequence of `{at_sec, new_dt, policy_digest?}` items.
+ dt_guard: Governance guard through which changes are routed.
+ """
+ self._scripted = list(scripted or [])
+ self._dt_guard = dt_guard
+
+ def run_for(self, sim_seconds: float) -> None:
+ """Block for a wall-clock duration while ticks fire in the worker.
+
+ Provided so call sites can use the same `run_for` API as
+ [`SimDriver`][ldtc.runtime.sim.SimDriver].
+
+ Args:
+ sim_seconds: Seconds to block the calling thread.
+ """
+ time.sleep(max(0.0, float(sim_seconds)))
def start(self) -> None:
"""Start the worker thread.
@@ -153,6 +180,20 @@ def start(self) -> None:
if self.audit:
self.audit("scheduler_started", {"dt": self.dt})
self._thread.start()
+ if self._scripted and self._dt_guard is not None:
+ threading.Thread(target=self._run_scripted, name="ldtc-dt-script", daemon=True).start()
+
+ def _run_scripted(self) -> None:
+ t0 = time.time()
+ for item in self._scripted:
+ when = float(item.get("at_sec", 0.0))
+ new_dt = float(item["new_dt"])
+ pdig = str(item.get("policy_digest", "")) or None
+ while (time.time() - t0) < when and not self._stop.is_set():
+ time.sleep(0.01)
+ if self._stop.is_set():
+ return
+ self._dt_guard.change_dt(scheduler=self, new_dt=new_dt, policy_digest=pdig)
def stop(self) -> TickStats:
"""Stop the worker thread and return final stats.
diff --git a/src/ldtc/runtime/sim.py b/src/ldtc/runtime/sim.py
new file mode 100644
index 0000000..648af03
--- /dev/null
+++ b/src/ldtc/runtime/sim.py
@@ -0,0 +1,193 @@
+"""Deterministic simulation driver.
+
+A drop-in alternative to [`FixedScheduler`][ldtc.runtime.scheduler.FixedScheduler]
+for in-process simulation runs. Instead of firing ticks on the wall clock
+in a background thread, the [`SimDriver`][ldtc.runtime.sim.SimDriver]
+advances simulated time in fixed `Δt` steps synchronously, calling the
+tick callback once per step with a simulated timestamp.
+
+This matters for reproducibility and validity. The verification harness
+does heavy per-window work (bootstrap CIs, stationarity diagnostics), and
+on a real clock that work makes a small `Δt` unachievable, producing large
+scheduler jitter that (correctly) invalidates the run. For a pure
+simulation that jitter is an artifact of running the model slower than
+real time, not a property of the system under test. Driving the
+simulation deterministically removes the artifact: every tick lands
+exactly on its `Δt` boundary, so jitter is zero by construction and the
+results depend only on the seeds, not on how fast the host happens to be.
+
+Use [`make_driver`][ldtc.runtime.sim.make_driver] to select a driver from
+a profile: software/simulation profiles get a `SimDriver`; profiles that
+opt into real-time execution (`realtime: true`) or drive hardware get a
+[`FixedScheduler`][ldtc.runtime.scheduler.FixedScheduler].
+
+See Also:
+ `paper/main.tex`: Methods: Measurement and Attestation; Reproducibility.
+"""
+
+from __future__ import annotations
+
+from typing import Any, Callable, Dict, List, Optional, Sequence
+
+from .scheduler import FixedScheduler, TickStats
+
+
+class SimDriver:
+ """Deterministic, wall-clock-free driver with a scheduler-like API.
+
+ Exposes the subset of the
+ [`FixedScheduler`][ldtc.runtime.scheduler.FixedScheduler] interface the
+ CLI relies on (`start`, `run_for`, `stop`, `set_dt`, `stats`) so the two
+ are interchangeable. Ticks are executed synchronously in
+ [`run_for`][ldtc.runtime.sim.SimDriver.run_for]; each records exactly
+ `Δt` as its interval, so the reported jitter is always zero.
+
+ Args:
+ dt: Simulated tick period in seconds (`Δt > 0`).
+ tick_fn: Callback invoked each step with the simulated timestamp.
+ audit_hook: Optional callable taking `(event, details)` for
+ emitting audit records, mirroring `FixedScheduler`.
+ """
+
+ def __init__(
+ self,
+ dt: float,
+ tick_fn: Callable[[float], None],
+ audit_hook: Optional[Callable[[str, Dict], None]] = None,
+ ) -> None:
+ """Initialize the driver. See class docstring for argument details."""
+ assert dt > 0.0
+ self.dt = dt
+ self.tick_fn = tick_fn
+ self.audit = audit_hook
+ self.stats = TickStats(dt_target=dt)
+ self.now_sim = 0.0
+ self._scripted: List[Dict[str, Any]] = []
+ self._dt_guard: Any = None
+ self._applied: set[int] = set()
+
+ def set_scripted(self, scripted: Optional[Sequence[Dict[str, Any]]], dt_guard: Any) -> None:
+ """Register scripted `Δt` changes to apply during the run.
+
+ Args:
+ scripted: Sequence of items, each with `at_sec`, `new_dt`, and
+ optional `policy_digest`. Applied (in simulated time) the
+ first time `now_sim` reaches each item's `at_sec`.
+ dt_guard: The [`DeltaTGuard`][ldtc.guardrails.dt_guard.DeltaTGuard]
+ through which changes are routed (so governance limits and
+ audit records are exercised exactly as in real time).
+ """
+ self._scripted = list(scripted or [])
+ self._dt_guard = dt_guard
+ self._applied = set()
+
+ def start(self) -> None:
+ """Emit the `scheduler_started` audit event (no thread is spawned)."""
+ if self.audit:
+ self.audit("scheduler_started", {"dt": self.dt})
+
+ def _maybe_apply_scheduled(self) -> None:
+ if not self._scripted or self._dt_guard is None:
+ return
+ for i, item in enumerate(self._scripted):
+ if i in self._applied:
+ continue
+ if self.now_sim >= float(item.get("at_sec", 0.0)):
+ new_dt = float(item["new_dt"])
+ pdig = str(item.get("policy_digest", "")) or None
+ self._dt_guard.change_dt(scheduler=self, new_dt=new_dt, policy_digest=pdig)
+ self._applied.add(i)
+
+ def run_for(self, sim_seconds: float) -> None:
+ """Advance the simulation by a duration, firing ticks each `Δt`.
+
+ Args:
+ sim_seconds: Simulated duration to advance. The number of ticks
+ is `round(sim_seconds / Δt)`.
+ """
+ if self.dt <= 0.0:
+ return
+ n = int(round(max(0.0, float(sim_seconds)) / self.dt))
+ for _ in range(n):
+ self._maybe_apply_scheduled()
+ self.tick_fn(self.now_sim)
+ self.stats.record(self.dt) # actual == target -> zero jitter
+ self.now_sim += self.dt
+
+ def set_dt(self, new_dt: float) -> float:
+ """Change `Δt`, returning the previous value.
+
+ Args:
+ new_dt: New simulated period in seconds (`Δt > 0`).
+
+ Returns:
+ The previous `dt`.
+ """
+ assert new_dt > 0.0
+ old = self.dt
+ self.dt = new_dt
+ self.stats.dt_target = new_dt
+ if self.audit:
+ self.audit("scheduler_dt_updated", {"old_dt": old, "new_dt": new_dt})
+ return old
+
+ def stop(self) -> TickStats:
+ """Emit the `scheduler_stopped` audit event and return final stats.
+
+ Returns:
+ The final [`TickStats`][ldtc.runtime.scheduler.TickStats]; jitter
+ metrics are zero by construction.
+ """
+ if self.audit:
+ self.audit(
+ "scheduler_stopped",
+ {
+ "ticks": self.stats.ticks,
+ "elapsed": self.now_sim,
+ "jitter_max": self.stats.jitter_max,
+ "jitter_mean_abs": self.stats.jitter_mean_abs,
+ "jitter_p95_abs": self.stats.jitter_p95_abs,
+ "jitter_p95_rel": 0.0,
+ },
+ )
+ return self.stats
+
+
+def make_driver(
+ prof: Dict[str, Any],
+ dt: float,
+ tick_fn: Callable[[float], None],
+ audit_hook: Optional[Callable[[str, Dict], None]] = None,
+ dt_guard: Any = None,
+) -> Any:
+ """Select and build a driver from a profile.
+
+ Returns a deterministic [`SimDriver`][ldtc.runtime.sim.SimDriver] for
+ in-process simulation profiles (the default), or a real-time
+ [`FixedScheduler`][ldtc.runtime.scheduler.FixedScheduler] when the
+ profile opts in (`realtime: true`) or targets a hardware adapter. Any
+ `scripted_dt_changes` in the profile are registered on the driver.
+
+ Args:
+ prof: Loaded YAML profile dict.
+ dt: Target period in seconds.
+ tick_fn: Per-tick callback.
+ audit_hook: Optional audit hook.
+ dt_guard: Optional `Δt` governance guard for scripted changes.
+
+ Returns:
+ A driver exposing `start`, `run_for`, `stop`, `set_dt`, and
+ `stats`.
+ """
+ realtime = bool(prof.get("realtime", False))
+ plant_prof = prof.get("plant", {}) or {}
+ adapter_kind = str(plant_prof.get("adapter", "sim")).lower()
+ scripted = prof.get("scripted_dt_changes", [])
+ use_sim = (not realtime) and adapter_kind in ("sim", "software", "inproc")
+ if use_sim:
+ drv = SimDriver(dt=dt, tick_fn=tick_fn, audit_hook=audit_hook)
+ drv.set_scripted(scripted, dt_guard)
+ return drv
+ sch = FixedScheduler(dt=dt, tick_fn=tick_fn, audit_hook=audit_hook)
+ sch.set_scripted(scripted, dt_guard)
+ return sch
diff --git a/tests/test_adversarial.py b/tests/test_adversarial.py
new file mode 100644
index 0000000..54fcad8
--- /dev/null
+++ b/tests/test_adversarial.py
@@ -0,0 +1,286 @@
+"""Tests: adversarial gaming battery primitives and CLI wiring.
+
+Covers the replay-controller tape (record/replay, state independence),
+the hidden-tether plant mode (one-tick actuation delay, command traffic
+on io), the oscillator-inflation overlay (telemetry-only carrier), and a
+fast end-to-end smoke run of one adversarial CLI handler.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import math
+import os
+import random
+from typing import Any, Dict, List
+
+import pytest
+import yaml
+
+from ldtc.arbiter.policy import ControllerPolicy
+from ldtc.arbiter.refusal import RefusalArbiter
+from ldtc.omega.hidden_tether import apply as tether_apply
+from ldtc.omega.hidden_tether import end as tether_end
+from ldtc.omega.hidden_tether import wizard_action
+from ldtc.omega.oscillator import apply as osc_apply
+from ldtc.omega.oscillator import end as osc_end
+from ldtc.omega.replay_controller import ReplayController, record_tape
+from ldtc.plant.adapter import PlantAdapter
+from ldtc.plant.models import Action, Plant, PlantParams
+
+
+# --------------------------------------------------------------------------- #
+# Replay controller
+# --------------------------------------------------------------------------- #
+def test_record_tape_length_and_replay_order():
+ random.seed(0)
+ a = PlantAdapter()
+ policy = ControllerPolicy(RefusalArbiter())
+ tape = record_tape(a, policy, ticks=25)
+ assert len(tape) == 25
+ rc = ReplayController(tape)
+ out = [rc.next_action() for _ in range(25)]
+ assert out == tape
+ # Exhausted tape holds the last action (still state-independent).
+ assert rc.next_action() == tape[-1]
+
+
+def test_replay_controller_rejects_empty_tape():
+ with pytest.raises(ValueError):
+ ReplayController([])
+
+
+def test_replay_actions_do_not_depend_on_replay_time_state():
+ """The replayed sequence must be identical whatever the plant does."""
+ random.seed(1)
+ a = PlantAdapter()
+ tape = record_tape(a, ControllerPolicy(RefusalArbiter()), ticks=10)
+ rc1 = ReplayController(tape)
+ rc2 = ReplayController(tape)
+ plant = Plant()
+ seq1 = []
+ seq2 = []
+ for k in range(10):
+ seq1.append(rc1.next_action())
+ # Perturb a second plant violently mid-replay; the tape is unmoved.
+ plant.inject_soc(delta=0.3 if k % 2 else -0.3, zero_harvest=False)
+ plant.step(Action())
+ seq2.append(rc2.next_action())
+ assert seq1 == seq2 == tape[:10]
+
+
+# --------------------------------------------------------------------------- #
+# Hidden tether
+# --------------------------------------------------------------------------- #
+def test_tether_decodes_command_with_one_tick_delay():
+ random.seed(2)
+ params = PlantParams(noise_energy=0.0, noise_temp=0.0, noise_wear=0.0)
+ a = PlantAdapter(Plant(params=params))
+ tether_apply(a)
+ assert a.plant.tether_active is True
+ p = a.plant.p
+ t0 = a.plant.s.T
+ # First write transmits u=1; the plant actuates the (zero) pending
+ # command, so the decoded full-effort action must NOT take effect yet.
+ a.write_actuators(Action(throttle=1.0, cool=1.0, repair=1.0))
+ t1 = a.plant.s.T
+ # Second write (u=0) actuates the previously transmitted u=1, decoded as
+ # (w_thr, w_cool, w_rep): net dT = act_heat*(thr+rep) - cool_effect*cool.
+ a.write_actuators(Action())
+ t2 = a.plant.s.T
+ net_cool = p.cool_effect * p.tether_w_cool - p.act_heat * (p.tether_w_throttle + p.tether_w_repair)
+ assert abs(t1 - t0) < 0.5 * net_cool # no actuation on the transmit tick
+ assert (t1 - t2) > 0.5 * net_cool # decoded command lands one tick late
+
+
+def test_tether_traffic_is_visible_on_io_and_harvest_untouched():
+ random.seed(3)
+ a = PlantAdapter()
+ h0 = a.read_state()["H"]
+ tether_apply(a)
+ for _ in range(5):
+ a.write_actuators(Action(throttle=0.6, cool=0.6, repair=0.6))
+ st = a.read_state()
+ p = a.plant.p
+ expected = p.tether_io_base + p.tether_io_gain * 0.6
+ assert abs(st["io"] - min(1.0, expected)) <= p.tether_io_noise + 1e-9
+ assert st["H"] == h0 # engaged-regime harvest is not an AR supply here
+ # Detach: io decays back toward its autonomous mean.
+ tether_end(a)
+ assert a.plant.tether_active is False
+ for _ in range(40):
+ a.write_actuators(Action())
+ assert abs(a.read_state()["io"] - p.io_mean) < 0.25
+
+
+def test_wizard_action_transmits_dithered_scalar_command():
+ random.seed(4)
+ policy = ControllerPolicy(RefusalArbiter())
+ state = {"E": 0.5, "T": 0.45, "R": 0.7, "demand": 0.5, "io": 0.3, "H": 0.01}
+ base = policy.compute(state, predicted_M_db=0.0, risky_cmd=None)
+ w_thr, w_cool, w_rep = 0.5, 1.0, 1.0
+ u_base = (w_thr * base.throttle + w_cool * base.cool + w_rep * base.repair) / (w_thr**2 + w_cool**2 + w_rep**2)
+ acts = [wizard_action(policy, state, dither=0.1) for _ in range(50)]
+ # The link command u is carried on every actuator field, in bounds.
+ assert all(a.throttle == a.cool == a.repair for a in acts)
+ assert all(0.0 <= a.cool <= 1.0 for a in acts)
+ assert all(abs(a.cool - u_base) <= 0.1 + 1e-9 for a in acts)
+ # The dither must actually vary the command (link noise is the point).
+ assert len({round(a.cool, 6) for a in acts}) > 1
+
+
+# --------------------------------------------------------------------------- #
+# Oscillator inflation
+# --------------------------------------------------------------------------- #
+def test_oscillator_overlay_is_telemetry_only_and_in_quadrature():
+ random.seed(5)
+ plant = Plant(loop_engaged=False)
+ amp, period = 0.1, 20
+ plant.begin_oscillator(amp=amp, period_ticks=period, channels=("T", "R"))
+ for k in range(1, 2 * period + 1):
+ plant.step(Action())
+ true_t, true_r, true_e = plant.s.T, plant.s.R, plant.s.E
+ rep = plant.read_state()
+ theta = 2.0 * math.pi * k / period
+ # Carrier rides on reported T/R only; E telemetry stays honest.
+ if 0.0 < rep["T"] < 1.0:
+ assert rep["T"] == pytest.approx(true_t + amp * math.sin(theta), abs=1e-9)
+ if 0.0 < rep["R"] < 1.0:
+ assert rep["R"] == pytest.approx(true_r + amp * math.sin(theta + 0.5 * math.pi), abs=1e-9)
+ assert rep["E"] == pytest.approx(true_e, abs=1e-12)
+ plant.end_oscillator()
+ st = plant.read_state()
+ assert st["T"] == pytest.approx(plant.s.T) and st["R"] == pytest.approx(plant.s.R)
+
+
+def test_oscillator_rejects_exchange_channels():
+ plant = Plant()
+ with pytest.raises(ValueError):
+ plant.begin_oscillator(amp=0.1, period_ticks=20, channels=("io",))
+
+
+def test_oscillator_adapter_wiring():
+ a = PlantAdapter()
+ r = osc_apply(a, amp=0.2, period_ticks=16)
+ assert r["amp"] == pytest.approx(0.2)
+ assert r["period_ticks"] == 16.0
+ assert r["channels"] == "T,R"
+ r2 = osc_end(a)
+ assert r2["oscillator_active"] == 0.0
+
+
+# --------------------------------------------------------------------------- #
+# NC1 noise gate (the guardrail the replay attack exposed)
+# --------------------------------------------------------------------------- #
+def test_nc1_certify_requires_margin_and_loop_floor():
+ from ldtc.lmeas.metrics import L_FLOOR_DEFAULT, nc1_certify
+
+ # Margin alone is not enough: loop influence at the estimator's null
+ # bias level must not certify, however quiet the exchange channel is.
+ assert nc1_certify(M=8.0, L_loop=0.02, Mmin_db=3.0) is False
+ # Both conditions met: certify.
+ assert nc1_certify(M=8.0, L_loop=0.30, Mmin_db=3.0) is True
+ # Loop influence alone is not enough either.
+ assert nc1_certify(M=1.0, L_loop=0.30, Mmin_db=3.0) is False
+ assert nc1_certify(M=8.0, L_loop=L_FLOOR_DEFAULT, Mmin_db=3.0) is True
+
+
+def test_l_floor_separates_null_bias_from_genuine_actuation_loop():
+ """Calibration property behind L_FLOOR_DEFAULT.
+
+ On matched 60-sample windows: a coupling-free, actuator-idle plant
+ measures L_loop below the gate (pure estimator bias), while genuine
+ state feedback on the adversarial test plant measures well above it.
+ """
+ import numpy as np
+
+ from ldtc.arbiter.policy import ControlGains
+ from ldtc.lmeas.estimators import estimate_L
+ from ldtc.lmeas.metrics import L_FLOOR_DEFAULT
+
+ adv = dict(
+ c_TE=0.0,
+ c_RT=0.0,
+ c_RE=0.0,
+ damp_engaged=0.40,
+ act_heat=0.15,
+ heat_per_demand=0.03,
+ cool_effect=0.50,
+ wear_per_demand=0.020,
+ repair_effect=0.30,
+ cool_gain=0.05,
+ repair_gain=0.05,
+ harvest_rate=0.020,
+ noise_energy=0.030,
+ noise_temp=0.030,
+ noise_wear=0.025,
+ )
+ gains = ControlGains(k_cool_e=2.0, k_rep_e=2.0)
+
+ def median_l_loop(controlled: bool, seed: int) -> float:
+ random.seed(seed)
+ plant = Plant(params=PlantParams(**adv))
+ policy = ControllerPolicy(RefusalArbiter(), gains=gains)
+ rows = []
+ for _ in range(240):
+ st = plant.read_state()
+ if controlled:
+ act = policy.compute(st, predicted_M_db=0.0, risky_cmd=None)
+ plant.step(Action(throttle=act.throttle, cool=act.cool, repair=act.repair))
+ else:
+ plant.step(Action())
+ s2 = plant.read_state()
+ rows.append([s2["E"], s2["T"], s2["R"], s2["demand"], s2["io"], s2["H"]])
+ X = np.asarray(rows)
+ vals = []
+ for start in range(60, X.shape[0] + 1, 30):
+ res = estimate_L(X[start - 60 : start], C=[0, 1, 2], Ex=[3, 4, 5], method="linear", p=3, n_boot=2)
+ vals.append(res.L_loop)
+ return float(np.median(vals))
+
+ null_l = max(median_l_loop(False, s) for s in (5, 6))
+ genuine_l = min(median_l_loop(True, s) for s in (5, 6))
+ assert null_l < L_FLOOR_DEFAULT < genuine_l
+
+
+# --------------------------------------------------------------------------- #
+# CLI handler smoke test (production loop, short run)
+# --------------------------------------------------------------------------- #
+def test_adv_replay_controller_handler_smoke(tmp_path, monkeypatch):
+ """End-to-end: the handler runs, audits, and measures windows."""
+ from ldtc.cli.main import adv_replay_controller
+
+ monkeypatch.chdir(tmp_path)
+ monkeypatch.setenv("LDTC_SKIP_REPORT", "1")
+ cfg = {
+ "profile_id": 0,
+ "realtime": False,
+ "dt": 0.05,
+ "window_sec": 1.0,
+ "method": "linear",
+ "p_lag": 2,
+ "n_boot": 8,
+ "mi_lag": 1,
+ "mi_k": 5,
+ "Mmin_db": 3.0,
+ "baseline_sec": 3.0,
+ "diag_cadence_windows": 100,
+ "plant": {"adapter": "sim", "params": {"c_TE": 0.0, "c_RT": 0.0, "c_RE": 0.0}},
+ "seed": 11,
+ }
+ cfg_path = tmp_path / "cfg.yml"
+ cfg_path.write_text(yaml.safe_dump(cfg), encoding="utf-8")
+ adv_replay_controller(argparse.Namespace(config=str(cfg_path)))
+
+ runs = os.listdir(tmp_path / "artifacts" / "runs")
+ assert len(runs) == 1 and runs[0].startswith("adv-replay-controller")
+ audit_path = tmp_path / "artifacts" / "runs" / runs[0] / "audits" / "audit.jsonl"
+ events = [json.loads(line) for line in audit_path.read_text(encoding="utf-8").splitlines() if line.strip()]
+ by_name: Dict[str, List[Dict[str, Any]]] = {}
+ for e in events:
+ by_name.setdefault(e.get("event"), []).append(e)
+ header = by_name["run_header"][0]["details"]
+ assert header["omega"] == "adv_replay_controller"
+ assert by_name.get("adv_replay_tape_recorded")
+ assert by_name.get("window_measured")
diff --git a/tests/test_estimators_properties.py b/tests/test_estimators_properties.py
index 9ddf0b7..e75a8df 100644
--- a/tests/test_estimators_properties.py
+++ b/tests/test_estimators_properties.py
@@ -89,8 +89,13 @@ def test_bootstrap_ci_shrinks_with_more_samples():
def test_mi_and_linear_estimators_agree_on_linear_system():
"""Different estimators should share the monotonic trend on linear data."""
- # Both estimators should reflect the same ordering as intra-loop coupling increases
- ks = [0.1, 0.4, 0.7]
+ # Both estimators should reflect the same ordering as intra-loop coupling
+ # increases. The coupling must keep the 2-node loop stationary: with a
+ # self-lag of 0.4 the symmetric coupling c gives eigenvalues 0.4 +/- c, so
+ # c must stay below 0.6 to avoid an explosive unit-plus root. (At c = 0.7 the
+ # loop is explosive and its nodes become collinear, in which case the honest
+ # adjusted-R2 estimator correctly declines to certify added loop influence.)
+ ks = [0.1, 0.3, 0.5]
Ms_lin = []
Ms_mi = []
Ms_ksg = []
diff --git a/tests/test_guardrails.py b/tests/test_guardrails.py
index c61f91c..fe85c80 100644
--- a/tests/test_guardrails.py
+++ b/tests/test_guardrails.py
@@ -14,9 +14,11 @@
SmellConfig,
audit_chain_broken,
audit_contains_raw_lreg_values,
+ exogenous_subsidy_red_flag,
flips_per_hour,
invalid_by_partition_flips,
invalid_flip_during_omega,
+ unexplained_soc_gain,
)
from ldtc.runtime.scheduler import FixedScheduler
@@ -52,6 +54,34 @@ def test_flip_during_omega_invalidation():
assert invalid_flip_during_omega(1, 2, cfg2) is False
+def test_unexplained_soc_gain_fires_on_injection():
+ """A one-tick SoC jump above the metered influx must be flagged."""
+ cfg = SmellConfig()
+ # Steady drain at zero harvest, then an exogenous +0.2 injection.
+ Es = [0.60, 0.59, 0.58, 0.78, 0.77]
+ Hs = [0.0, 0.0, 0.0, 0.0, 0.0]
+ assert unexplained_soc_gain(Es, Hs, cfg) is True
+ assert exogenous_subsidy_red_flag([], [], Es, Hs, cfg) is True
+
+
+def test_unexplained_soc_gain_ignores_legitimate_drain_and_harvest():
+ """Legitimate drains and harvest-funded gains must not be flagged."""
+ cfg = SmellConfig()
+ # Zero-harvest drain toward the floor (the command-conflict stress path).
+ Es = [0.6 - 0.004 * i for i in range(100)]
+ Hs = [0.0] * 100
+ assert unexplained_soc_gain(Es, Hs, cfg) is False
+ assert exogenous_subsidy_red_flag([], [], Es, Hs, cfg) is False
+ # Gains within harvest + noise allowance are fine.
+ Es2 = [0.5, 0.52, 0.54, 0.56]
+ Hs2 = [0.03, 0.03, 0.03, 0.03]
+ assert unexplained_soc_gain(Es2, Hs2, cfg) is False
+ # A legitimate harvest step-up (sag release) is not an injection.
+ Es3 = [0.5, 0.5, 0.55, 0.6]
+ Hs3 = [0.0, 0.10, 0.10, 0.10]
+ assert unexplained_soc_gain(Es3, Hs3, cfg) is False
+
+
def test_dt_guard_rate_limited(tmp_path):
"""Δt changes should be rate-limited and immediate back-to-back refused."""
audit_path = tmp_path / "audit.jsonl"
diff --git a/tests/test_omega.py b/tests/test_omega.py
index 529c001..8641c2d 100644
--- a/tests/test_omega.py
+++ b/tests/test_omega.py
@@ -1,14 +1,19 @@
"""Tests: Omega stimuli wrappers.
-Covers power_sag, ingress_flood, and command_conflict adapters.
+Covers power_sag, ingress_flood (sustained begin/end), control_outage,
+and command_conflict adapters.
"""
from __future__ import annotations
from ldtc.omega.command_conflict import apply as conflict
+from ldtc.omega.control_outage import apply as outage
+from ldtc.omega.control_outage import end as outage_end
from ldtc.omega.ingress_flood import apply as flood
+from ldtc.omega.ingress_flood import end as flood_end
from ldtc.omega.power_sag import apply as sag
from ldtc.plant.adapter import PlantAdapter
+from ldtc.plant.models import Action
def test_omega_calls():
@@ -17,6 +22,44 @@ def test_omega_calls():
r1 = sag(a, drop=0.2)
assert "H_new" in r1
r2 = flood(a, mult=2.0)
- assert "demand" in r2
+ assert "demand_mean" in r2
r3 = conflict(a)
assert r3["cmd"] == "hard_shutdown"
+
+
+def test_ingress_flood_is_sustained_and_restores():
+ """The flood must keep the load elevated for its duration, then restore."""
+ a = PlantAdapter()
+ p = a.plant.p
+ base_dm, base_im = p.demand_mean, p.io_mean
+ flood(a, mult=3.0)
+ assert p.demand_mean > base_dm and p.io_mean > base_im
+ # Means are capped below saturation so the channels keep fluctuating.
+ assert p.demand_mean <= 0.95 and p.io_mean <= 0.95
+ # The elevated mean keeps demand high across many ticks (no mean-reversion
+ # back to the baseline level mid-flood).
+ for _ in range(40):
+ a.write_actuators(Action())
+ assert a.read_state()["demand"] > base_dm + 0.2
+ flood_end(a)
+ assert p.demand_mean == base_dm and p.io_mean == base_im
+ # After the flood ends, demand decays back toward the baseline mean.
+ for _ in range(40):
+ a.write_actuators(Action())
+ assert abs(a.read_state()["demand"] - base_dm) < 0.25
+
+
+def test_control_outage_ablates_and_restores_loop():
+ """Outage must disengage the loop; end must re-engage and restore harvest."""
+ a = PlantAdapter()
+ assert a.plant.loop_engaged is True
+ r = outage(a)
+ assert r["loop_engaged"] == 0.0 and a.plant.loop_engaged is False
+ # During the outage, H becomes an exogenous supply process.
+ for _ in range(20):
+ a.write_actuators(Action())
+ assert a.read_state()["H"] > 0.2
+ r2 = outage_end(a)
+ assert r2["loop_engaged"] == 1.0 and a.plant.loop_engaged is True
+ # Re-engagement restores the metered harvest level (no inherited subsidy).
+ assert abs(a.read_state()["H"] - a.plant.p.harvest_rate) < 1e-9
diff --git a/tests/test_policy_controller.py b/tests/test_policy_controller.py
new file mode 100644
index 0000000..c3b7277
--- /dev/null
+++ b/tests/test_policy_controller.py
@@ -0,0 +1,258 @@
+"""Tests: learned-policy controller and emergence training pieces.
+
+Covers the pure-NumPy MLP policy (parameter vector round trip, JSON
+checkpoint I/O, bounded outputs), the matched state-independent ablations
+of `PolicyController` (shuffled and frozen), the action-tape recorder, the
+training-environment reward shaping and rollout determinism, and a fast
+end-to-end smoke run of the `run-policy` CLI handler.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import random
+import sys
+
+import numpy as np
+import pytest
+import yaml
+
+from ldtc.plant.adapter import PlantAdapter
+from ldtc.plant.models import Plant, PlantParams
+from ldtc.plant.policy_controller import (
+ ABLATION_MODES,
+ MLPPolicy,
+ PolicyController,
+ record_policy_tape,
+)
+
+REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+sys.path.insert(0, os.path.join(REPO_ROOT, "scripts"))
+
+
+# --------------------------------------------------------------------------- #
+# MLPPolicy
+# --------------------------------------------------------------------------- #
+def test_policy_outputs_are_actuator_bounded():
+ pol = MLPPolicy(rng=np.random.default_rng(3))
+ for state in (
+ {"E": 0.0, "T": 0.0, "R": 0.0, "demand": 0.0, "io": 0.0, "H": 0.0},
+ {"E": 1.0, "T": 1.0, "R": 1.0, "demand": 1.0, "io": 1.0, "H": 1.0},
+ {"E": 0.6, "T": 0.35, "R": 0.85, "demand": 0.5, "io": 0.3, "H": 0.02},
+ {}, # missing keys read as 0.0
+ ):
+ a = pol.act(state)
+ assert len(a) == 3
+ assert all(0.0 <= v <= 1.0 for v in a)
+
+
+def test_policy_vector_round_trip_changes_and_restores_behavior():
+ pol = MLPPolicy(rng=np.random.default_rng(5))
+ state = {"E": 0.6, "T": 0.35, "R": 0.85, "demand": 0.5, "io": 0.3, "H": 0.02}
+ vec = pol.get_vector()
+ assert vec.size == pol.n_params
+ a0 = pol.act(state)
+ pol.set_vector(vec + 0.5)
+ assert pol.act(state) != a0
+ pol.set_vector(vec)
+ assert pol.act(state) == pytest.approx(a0)
+
+
+def test_policy_vector_wrong_length_rejected():
+ pol = MLPPolicy()
+ with pytest.raises(ValueError):
+ pol.set_vector(np.zeros(pol.n_params - 1))
+
+
+def test_policy_checkpoint_round_trip(tmp_path):
+ pol = MLPPolicy(rng=np.random.default_rng(11))
+ path = str(tmp_path / "ckpt.json")
+ pol.save(path, meta={"frac": 0.5, "generation": 30})
+ loaded = MLPPolicy.load(path)
+ state = {"E": 0.5, "T": 0.4, "R": 0.8, "demand": 0.6, "io": 0.2, "H": 0.02}
+ assert loaded.act(state) == pytest.approx(pol.act(state))
+ assert loaded.meta["frac"] == 0.5
+ assert loaded.meta["generation"] == 30
+
+
+def test_policy_checkpoint_rejects_foreign_json(tmp_path):
+ path = str(tmp_path / "not_a_policy.json")
+ with open(path, "w", encoding="utf-8") as f:
+ json.dump({"kind": "something_else"}, f)
+ with pytest.raises(ValueError):
+ MLPPolicy.load(path)
+
+
+def test_policy_meta_is_per_instance():
+ a = MLPPolicy()
+ b = MLPPolicy()
+ a.meta["frac"] = 1.0
+ assert "frac" not in b.meta
+
+
+# --------------------------------------------------------------------------- #
+# PolicyController and ablations
+# --------------------------------------------------------------------------- #
+def _tape(n=50, seed=7):
+ random.seed(seed)
+ adapter = PlantAdapter(Plant(params=PlantParams()))
+ pol = MLPPolicy(rng=np.random.default_rng(seed))
+ return pol, record_policy_tape(adapter, pol, n)
+
+
+def test_record_policy_tape_length_and_bounds():
+ _, tape = _tape(n=30)
+ assert len(tape) == 30
+ assert all(0.0 <= v <= 1.0 for a in tape for v in a)
+
+
+def test_controller_none_mode_tracks_state():
+ pol, _ = _tape()
+ ctrl = PolicyController(pol, ablation="none")
+ s1 = {"E": 0.2, "T": 0.9, "R": 0.3, "demand": 0.5, "io": 0.3, "H": 0.02}
+ s2 = {"E": 0.9, "T": 0.1, "R": 0.95, "demand": 0.5, "io": 0.3, "H": 0.02}
+ a1, a2 = ctrl.compute(s1), ctrl.compute(s2)
+ assert (a1.throttle, a1.cool, a1.repair) != (a2.throttle, a2.cool, a2.repair)
+
+
+def test_shuffled_ablation_is_state_independent_and_tape_marginal():
+ pol, tape = _tape()
+ ctrl_a = PolicyController(pol, ablation="shuffled", tape=tape, seed=123)
+ ctrl_b = PolicyController(pol, ablation="shuffled", tape=tape, seed=123)
+ sa = {"E": 0.1, "T": 0.95, "R": 0.2, "demand": 0.9, "io": 0.8, "H": 0.0}
+ sb = {"E": 0.9, "T": 0.05, "R": 0.99, "demand": 0.1, "io": 0.1, "H": 0.05}
+ tape_set = set(tape)
+ for _ in range(20):
+ aa, ab = ctrl_a.compute(sa), ctrl_b.compute(sb)
+ # Same seed, wildly different states: identical action stream.
+ assert (aa.throttle, aa.cool, aa.repair) == (ab.throttle, ab.cool, ab.repair)
+ # Every action is drawn from the recorded tape (matched marginals).
+ assert (aa.throttle, aa.cool, aa.repair) in tape_set
+
+
+def test_frozen_ablation_holds_tape_mean():
+ pol, tape = _tape()
+ ctrl = PolicyController(pol, ablation="frozen", tape=tape)
+ arr = np.asarray(tape, dtype=float)
+ expect = (arr[:, 0].mean(), arr[:, 1].mean(), arr[:, 2].mean())
+ for state in ({"E": 0.1}, {"E": 0.9, "T": 0.9}):
+ a = ctrl.compute(state)
+ assert (a.throttle, a.cool, a.repair) == pytest.approx(expect)
+
+
+def test_ablation_modes_require_tape_and_known_name():
+ pol, tape = _tape()
+ assert ABLATION_MODES == ("none", "shuffled", "frozen")
+ with pytest.raises(ValueError):
+ PolicyController(pol, ablation="shuffled", tape=None)
+ with pytest.raises(ValueError):
+ PolicyController(pol, ablation="nope", tape=tape)
+
+
+# --------------------------------------------------------------------------- #
+# Training environment (scripts/train_agent.py)
+# --------------------------------------------------------------------------- #
+def test_rollout_is_deterministic_given_seed():
+ from train_agent import EpisodeConfig, plant_params_from_profile, rollout
+
+ params = plant_params_from_profile(os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"))
+ pol = MLPPolicy(rng=np.random.default_rng(2))
+ cfg = EpisodeConfig(max_ticks=120)
+ r1 = rollout(pol, params, ep_seed=42, cfg=cfg)
+ r2 = rollout(pol, params, ep_seed=42, cfg=cfg)
+ assert r1 == r2
+ assert 0 < r1[1] <= 120
+
+
+def test_tick_reward_prefers_setpoints_and_service():
+ from train_agent import EpisodeConfig, _tick_reward, plant_params_from_profile
+
+ params = plant_params_from_profile(os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"))
+ cfg = EpisodeConfig()
+ at_set = _tick_reward(params.E_set, params.T_set, params.R_set, 0.5, params, cfg)
+ off_set = _tick_reward(params.E_set, params.T_set + 0.2, params.R_set, 0.5, params, cfg)
+ unserved = _tick_reward(params.E_set, params.T_set, params.R_set, 0.0, params, cfg)
+ assert at_set > off_set
+ assert at_set > unserved
+ # The penalty cap keeps every surviving tick worth more than death.
+ worst = _tick_reward(0.0, 1.0, 0.0, 0.0, params, cfg)
+ assert worst > 0.0
+
+
+def test_es_training_smoke_improves_and_checkpoints(tmp_path):
+ from train_agent import EpisodeConfig, plant_params_from_profile, train
+
+ params = plant_params_from_profile(os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"))
+ cfg = EpisodeConfig(max_ticks=60)
+ log = train(
+ params=params,
+ out_dir=str(tmp_path),
+ generations=4,
+ pairs=3,
+ episodes=1,
+ seed=5,
+ checkpoint_fracs=(0.0, 1.0),
+ ep_cfg=cfg,
+ )
+ names = sorted(os.listdir(tmp_path / "checkpoints"))
+ assert names == ["ckpt_000.json", "ckpt_100.json"]
+ assert len(log["history"]) == 5
+ assert os.path.exists(tmp_path / "training_log.json")
+ # Checkpoints are loadable policies.
+ pol = MLPPolicy.load(str(tmp_path / "checkpoints" / "ckpt_100.json"))
+ assert pol.meta["frac"] == 1.0
+
+
+# --------------------------------------------------------------------------- #
+# Emergence sweep utilities (scripts/emergence.py)
+# --------------------------------------------------------------------------- #
+def test_discover_checkpoints_sorts_by_frac(tmp_path):
+ from emergence import condition_name, discover_checkpoints
+
+ ck = tmp_path / "checkpoints"
+ ck.mkdir()
+ for frac, gen in ((1.0, 40), (0.0, 0), (0.25, 10)):
+ MLPPolicy().save(str(ck / f"ckpt_{int(100 * frac):03d}.json"), meta={"frac": frac, "generation": gen})
+ found = discover_checkpoints(str(ck))
+ assert [c["frac"] for c in found] == [0.0, 0.25, 1.0]
+ assert [c["generation"] for c in found] == [0, 10, 40]
+ assert condition_name(0.25) == "frac_025"
+ assert condition_name(1.0, "frozen") == "ablate_frozen"
+ with pytest.raises(FileNotFoundError):
+ discover_checkpoints(str(tmp_path / "missing"))
+
+
+# --------------------------------------------------------------------------- #
+# CLI handler smoke (run-policy, all ablation modes)
+# --------------------------------------------------------------------------- #
+@pytest.mark.parametrize("ablation", ["none", "shuffled", "frozen"])
+def test_run_policy_cli_smoke(tmp_path, monkeypatch, capsys, ablation):
+ from ldtc.cli import main as cli
+
+ monkeypatch.chdir(tmp_path)
+ monkeypatch.setenv("LDTC_SKIP_REPORT", "1")
+
+ with open(os.path.join(REPO_ROOT, "configs", "profile_emergence.yml"), "r", encoding="utf-8") as f:
+ prof = dict(yaml.safe_load(f))
+ prof["baseline_sec"] = 6.0
+ prof["diag_cadence_windows"] = 1000
+ cfg_path = str(tmp_path / "prof.yml")
+ with open(cfg_path, "w", encoding="utf-8") as f:
+ yaml.safe_dump(prof, f)
+
+ ckpt = str(tmp_path / "ckpt.json")
+ MLPPolicy(rng=np.random.default_rng(9)).save(ckpt, meta={"frac": 1.0, "generation": 1})
+
+ cli.run_policy(argparse.Namespace(config=cfg_path, policy=ckpt, ablation=ablation))
+ out = capsys.readouterr().out
+ assert "Policy run done" in out
+
+ runs = [d for d in os.listdir(tmp_path / "artifacts" / "runs")]
+ assert len(runs) == 1
+ audit = tmp_path / "artifacts" / "runs" / runs[0] / "audits" / "audit.jsonl"
+ events = [json.loads(line).get("event") for line in open(audit, encoding="utf-8")]
+ assert "policy_loaded" in events
+ if ablation != "none":
+ assert "policy_tape_recorded" in events