From 29ac96bc7fc56a08b0a167302aa5f140bf311d97 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:23:48 -0700 Subject: [PATCH 1/7] Admit one train_schedule variable and an active reserve. A preregistered train_schedule declaration compares the four schedule fields as a single scientific variable. Holdout counts use only current EVAL_RESERVE identities bound to the experiment, leaving spent evaluation_reserved flags and the append-only ledger unchanged. --- .../test_hyperlexical_admission_contract.py | 372 ++++++++++++++++++ 1 file changed, 372 insertions(+) create mode 100644 tests/shadow/test_hyperlexical_admission_contract.py diff --git a/tests/shadow/test_hyperlexical_admission_contract.py b/tests/shadow/test_hyperlexical_admission_contract.py new file mode 100644 index 00000000..3a608497 --- /dev/null +++ b/tests/shadow/test_hyperlexical_admission_contract.py @@ -0,0 +1,372 @@ +"""Admission contract: one train_schedule variable, active reserve only.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +from admission_fixtures import arm_controlled, classify_row, sha256_bytes # noqa: E402 +from hyperlexical.admission import ( # noqa: E402 + AdmissionError, + admit_training_run, + schedule_bundle, +) +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import IdentityLedger, derived_state # noqa: E402 +from hyperlexical.loop import _EARLY_STOP_OFF, _EARLY_STOP_ON # noqa: E402 +from hyperlexical.selection_surface import row_id # noqa: E402 + +EXPERIMENT = "HLX-EXP-TEST" +OTHER = "HLX-EXP-OTHER" +SELECT = "classify_macro_f1_nonnone" +CONTROL_SCHEDULE = { + "HYPERLEX_TRAIN_EPOCHS": "40", + "HYPERLEX_EARLY_STOP": "0", +} +CANDIDATE_SCHEDULE = { + "HYPERLEX_TRAIN_EPOCHS": "12", + "HYPERLEX_EARLY_STOP": "1", + "HYPERLEX_EARLY_STOP_PATIENCE": "4", + "HYPERLEX_EARLY_STOP_MIN_EPOCHS": "4", +} + + +def _admit(): + return admit_training_run( + trunk=Path(__import__("os").environ["HYPERLEX_TRUNK_DIR"]), + out_dir=Path(__import__("os").environ["HYPERLEX_TRAIN_OUT"]), + ) + + +def _write_envs(tmp_path: Path, baseline: dict, candidate: dict) -> None: + (tmp_path / "baseline-env.json").write_text( + json.dumps(baseline, sort_keys=True), encoding="utf-8" + ) + (tmp_path / "candidate-env.json").write_text( + json.dumps(candidate, sort_keys=True), encoding="utf-8" + ) + + +def _arm_schedule( + monkeypatch, + tmp_path: Path, + *, + declared: str | None, + baseline_extra: dict | None = None, + candidate_extra: dict | None = None, + metric: str = SELECT, + baseline_metric: str | None = None, + train_text: str = "train row", +): + armed = arm_controlled(monkeypatch, tmp_path, [classify_row(train_text)]) + baseline = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SELECT_METRIC": metric if baseline_metric is None else baseline_metric, + } + candidate = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SELECT_METRIC": metric, + "HLX_EXPERIMENT_ID": EXPERIMENT, + "HYPERLEX_TRAIN_OUT": str(armed["out"]), + } + baseline.update(CONTROL_SCHEDULE) + candidate.update(CANDIDATE_SCHEDULE) + if baseline_extra: + baseline.update(baseline_extra) + if candidate_extra: + candidate.update(candidate_extra) + if declared is not None: + baseline["HLX_SCIENTIFIC_VARIABLE"] = declared + candidate["HLX_SCIENTIFIC_VARIABLE"] = declared + monkeypatch.setenv("HLX_SCIENTIFIC_VARIABLE", declared) + else: + monkeypatch.delenv("HLX_SCIENTIFIC_VARIABLE", raising=False) + _write_envs(tmp_path, baseline, candidate) + monkeypatch.setenv("HLX_SELECT_METRIC", metric) + for key, value in candidate.items(): + if key in { + "HLX_EXPERIMENT_ID", + "HYPERLEX_TRAIN_OUT", + "HLX_SCIENTIFIC_VARIABLE", + "HLX_SELECT_METRIC", + }: + continue + monkeypatch.setenv(key, value) + return armed + + +def _ledger_bytes(directory: Path) -> dict[str, bytes]: + return {path.name: path.read_bytes() for path in sorted(directory.iterdir())} + + +def _bind(root: Path, ledger: IdentityLedger, experiment_id: str) -> Path: + ledger_dir = root / "ledger" + if ledger_dir.exists(): + raise AssertionError("ledger directory already exists") + ledger.save(ledger_dir) + events_sha = sha256_bytes((ledger_dir / "events.jsonl").read_bytes()) + active = ledger.active_reserve_records(experiment_id) + binding = { + "schema": "hyperlex.reserve_binding.v1", + "experiment_id": experiment_id, + "ledger_events_sha256": events_sha, + "counts": ledger.active_reserve_counts(experiment_id), + "identities": len(active), + "lifecycle": "EVAL_RESERVE", + } + path = root / "reserve-binding.json" + path.write_text(json.dumps(binding, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return path + + +def _observe_reserve(ledger: IdentityLedger, text: str, label: dict, experiment_id: str) -> str: + digest = normalized_text_sha256(text) + ledger.observe( + digest, + source_artifact="fixture", + row_ids=[row_id({"text": text, "task": label["task"], "split": "val"})], + labels=[label], + provenance="fixture", + experiment_id=experiment_id, + ) + ledger.transition(digest, "EVAL_RESERVE", source_artifact="fixture", provenance="fixture") + return digest + + +def _fresh_reserve(experiment_id: str) -> IdentityLedger: + ledger = IdentityLedger() + _observe_reserve( + ledger, + "reserve classify surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + experiment_id, + ) + _observe_reserve( + ledger, + "reserve unbind surface", + { + "task": "unbind", + "class": "INFERRED", + "lineage": "none", + "split": "val", + "unbind_clean": True, + }, + experiment_id, + ) + return ledger + + +def _spend(ledger: IdentityLedger, digest: str) -> None: + ledger.transition(digest, "EVAL_BOUND", source_artifact="fixture", provenance="bind") + ledger.transition(digest, "EVAL_SPENT", source_artifact="fixture", provenance="spend") + + +def test_schedule_tokens_match_the_trainer(): + from hyperlexical.admission import _SCHEDULE_OFF, _SCHEDULE_ON + + assert _SCHEDULE_OFF == _EARLY_STOP_OFF + assert _SCHEDULE_ON == _EARLY_STOP_ON + + +def test_select005_schedule_bundle_is_one_variable(monkeypatch, tmp_path): + armed = _arm_schedule(monkeypatch, tmp_path, declared="train_schedule") + before = _ledger_bytes(armed["reserve"]["ledger"]) + base = json.loads((tmp_path / "baseline-env.json").read_text(encoding="utf-8")) + cand = json.loads((tmp_path / "candidate-env.json").read_text(encoding="utf-8")) + assert schedule_bundle(base) != schedule_bundle(cand) + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.status == "PREREGISTERED" + assert result.receipt["optimizer_loaded"] is False + assert result.receipt["epochs"] == 0 + assert result.receipt["gradient_steps"] == 0 + assert _ledger_bytes(armed["reserve"]["ledger"]) == before + assert not armed["out"].exists() + + +def test_extra_scientific_difference_is_a_second_variable(monkeypatch, tmp_path): + _arm_schedule( + monkeypatch, + tmp_path, + declared="train_schedule", + candidate_extra={"HYPERLEX_TRAIN_LR": "1e-4"}, + ) + with pytest.raises(AdmissionError, match="scientific variable count is 2") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + message = str(exc.value) + assert "train_schedule" in message + assert "HYPERLEX_TRAIN_LR" in message + + +def test_selection_metric_difference_is_not_inside_train_schedule(monkeypatch, tmp_path): + _arm_schedule( + monkeypatch, + tmp_path, + declared="train_schedule", + baseline_metric="unbind_exact", + ) + with pytest.raises(AdmissionError, match="not part of train_schedule") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + + +def test_undeclared_schedule_difference_refuses_admission(monkeypatch, tmp_path): + _arm_schedule(monkeypatch, tmp_path, declared=None) + with pytest.raises(AdmissionError, match="undeclared schedule difference") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + + +def test_declared_select_metric_mode_still_passes(monkeypatch, tmp_path): + armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + baseline = {"HYPERLEX_FILLER_FILTER": "off", "HLX_SCIENTIFIC_VARIABLE": "HLX_SELECT_METRIC"} + candidate = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SCIENTIFIC_VARIABLE": "HLX_SELECT_METRIC", + "HLX_SELECT_METRIC": SELECT, + "HLX_EXPERIMENT_ID": EXPERIMENT, + "HYPERLEX_TRAIN_OUT": str(armed["out"]), + } + _write_envs(tmp_path, baseline, candidate) + monkeypatch.setenv("HLX_SCIENTIFIC_VARIABLE", "HLX_SELECT_METRIC") + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.receipt["optimizer_loaded"] is False + + +def test_implicit_select_metric_mode_still_passes(monkeypatch, tmp_path): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.delenv("HLX_SCIENTIFIC_VARIABLE", raising=False) + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.status == "PREREGISTERED" + + +def test_spent_only_ledger_has_zero_active_counts_and_keeps_the_flag(monkeypatch, tmp_path): + ledger = _fresh_reserve(EXPERIMENT) + for record in list(ledger.identities.values()): + assert record["evaluation_reserved"] is True + _spend(ledger, record["normalized_text_sha256"]) + assert record["evaluation_reserved"] is True + assert derived_state(record) == "EVAL_SPENT" + assert ledger.active_reserve_counts(EXPERIMENT) == { + "classify": 0, + "classify_observed": 0, + "classify_non_none": 0, + "unbind_clean": 0, + } + root = tmp_path / "spent" + root.mkdir() + binding = _bind(root, ledger, EXPERIMENT) + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + before = _ledger_bytes(root / "ledger") + with pytest.raises(AdmissionError, match="missing active reserve") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "holdout_reserve" + assert _ledger_bytes(root / "ledger") == before + reloaded = IdentityLedger.load(root / "ledger") + assert all(record["evaluation_reserved"] is True for record in reloaded.identities.values()) + assert reloaded.active_reserve_counts(EXPERIMENT)["classify"] == 0 + + +def test_mixed_ledger_counts_only_active_identities(monkeypatch, tmp_path): + ledger = _fresh_reserve(EXPERIMENT) + spent = _observe_reserve( + ledger, + "already spent surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + EXPERIMENT, + ) + _spend(ledger, spent) + assert ledger.identity(spent)["evaluation_reserved"] is True + active_hashes = {record["normalized_text_sha256"] for record in ledger.active_reserve_records(EXPERIMENT)} + assert spent not in active_hashes + assert len(active_hashes) == 2 + counts = ledger.active_reserve_counts(EXPERIMENT) + assert counts["classify"] == 1 + assert counts["unbind_clean"] == 1 + root = tmp_path / "mixed" + root.mkdir() + binding = _bind(root, ledger, EXPERIMENT) + armed = _arm_schedule(monkeypatch, tmp_path, declared="train_schedule") + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + before = _ledger_bytes(root / "ledger") + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.receipt["reserve_counts"]["classify"] == 1 + assert _ledger_bytes(root / "ledger") == before + assert not armed["out"].exists() + + +def test_wrong_experiment_active_reserve_refuses(monkeypatch, tmp_path): + ledger = _fresh_reserve(OTHER) + assert ledger.active_reserve_counts(EXPERIMENT) == { + "classify": 0, + "classify_observed": 0, + "classify_non_none": 0, + "unbind_clean": 0, + } + assert ledger.active_reserve_counts(OTHER)["classify"] == 1 + root = tmp_path / "foreign" + root.mkdir() + # Binding names the current experiment. The identities do not. + binding = _bind(root, ledger, EXPERIMENT) + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + with pytest.raises(AdmissionError, match="bound to another experiment") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "holdout_reserve" + + +def test_active_reserve_overlap_still_fails(monkeypatch, tmp_path): + arm_controlled(monkeypatch, tmp_path, [classify_row("reserve classify surface")]) + with pytest.raises(AdmissionError, match="overlaps the sealed reserve") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "train_reserve_disjointness" + + +def test_spent_identity_cannot_be_reallocated_or_reused(): + ledger = IdentityLedger() + digest = _observe_reserve( + ledger, + "spent surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + EXPERIMENT, + ) + _spend(ledger, digest) + record = ledger.identity(digest) + assert record is not None + assert record["evaluation_reserved"] is True + assert derived_state(record) == "EVAL_SPENT" + with pytest.raises(SystemExit, match="monotonic"): + ledger.transition(digest, "EVAL_RESERVE", source_artifact="fixture", provenance="reuse") + report = ledger.admit( + [ + { + "text": "spent surface", + "task": "classify", + "split": "val", + "class": "OBSERVED", + "lineage": "gaming-meta", + } + ], + batch_id="reuse-attempt", + source_artifact="fixture", + targets={"classify": 1, "classify_observed": 1, "classify_non_none": 1, "unbind_clean": 1}, + ) + assert report["rejected_existing_identities"]["EVAL_SPENT"] == 1 + assert report["unique_admitted_to_eval_reserve"] == 0 + assert ledger.identity(digest)["evaluation_reserved"] is True + assert derived_state(ledger.identity(digest)) == "EVAL_SPENT" From fcfe6d856a615221b18dc075ccd3e5d787d1f3d9 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:24:45 -0700 Subject: [PATCH 2/7] Admit one train_schedule variable and an active reserve. A preregistered train_schedule declaration compares the four schedule fields as a single scientific variable. Holdout counts use only current EVAL_RESERVE identities bound to the experiment, leaving spent evaluation_reserved flags and the append-only ledger unchanged. --- tests/shadow/test_hyperlexical_admission_parity.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/shadow/test_hyperlexical_admission_parity.py b/tests/shadow/test_hyperlexical_admission_parity.py index f60657d4..e33d0a07 100644 --- a/tests/shadow/test_hyperlexical_admission_parity.py +++ b/tests/shadow/test_hyperlexical_admission_parity.py @@ -359,24 +359,24 @@ def test_historical_spent_identity_is_not_the_controlled_reserve(monkeypatch, tm def test_reserve_identity_that_left_eval_reserve_rejects_both(monkeypatch, tmp_path, capsys): - from hyperlexical.identity_ledger import IdentityLedger, derived_state + from hyperlexical.identity_ledger import IdentityLedger, derived_state, slices_of armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) ledger = IdentityLedger.load(armed["reserve"]["ledger"]) - reserved = [ + classify = [ digest for digest, record in ledger.identities.items() - if derived_state(record) == "EVAL_RESERVE" + if derived_state(record) == "EVAL_RESERVE" and "classify" in slices_of(record) ] ledger.transition( - reserved[0], + classify[0], "EVAL_BOUND", source_artifact="fixture", provenance="fixture", ) _rebind_ledger(monkeypatch, tmp_path, armed, ledger) pre = _assert_same_failure(monkeypatch, capsys) - assert "EVAL_BOUND" in pre["error"] + assert "reserve slice classify is absent" in pre["error"] assert pre["failed_gate"] == "holdout_reserve" From 9f5b723c6f12a2e3060cdd60ef4c0f68598b2f93 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:25:50 -0700 Subject: [PATCH 3/7] Admit one train_schedule variable and an active reserve. A preregistered train_schedule declaration compares the four schedule fields as a single scientific variable. Holdout counts use only current EVAL_RESERVE identities bound to the experiment, leaving spent evaluation_reserved flags and the append-only ledger unchanged. --- scripts/shadow/hyperlexical/admission.py | 180 +++++++++++++++++++---- 1 file changed, 154 insertions(+), 26 deletions(-) diff --git a/scripts/shadow/hyperlexical/admission.py b/scripts/shadow/hyperlexical/admission.py index d7bcc65d..01a4e89e 100644 --- a/scripts/shadow/hyperlexical/admission.py +++ b/scripts/shadow/hyperlexical/admission.py @@ -32,6 +32,7 @@ from pathlib import Path from typing import Any, Callable, Mapping +from .classify_metrics import SELECT_METRIC_CLASSIFY from .export import repo_root from .holdout_guard import ( HoldoutSpec, @@ -65,6 +66,19 @@ TRUNK_SHA_ENV = "HLX_TRUNK_SHA256" TRAIN_OUT_ENV = "HYPERLEX_TRAIN_OUT" SELECT_METRIC_KEY = "HLX_SELECT_METRIC" +SCIENTIFIC_VARIABLE_KEY = "HLX_SCIENTIFIC_VARIABLE" +DECLARED_SELECT_METRIC = SELECT_METRIC_KEY +DECLARED_TRAIN_SCHEDULE = "train_schedule" +# The only composite. These four trainer fields are one variable, not a +# caller-supplied grouping of arbitrary keys. +SCHEDULE_BUNDLE_KEYS = ( + "HYPERLEX_TRAIN_EPOCHS", + "HYPERLEX_EARLY_STOP", + "HYPERLEX_EARLY_STOP_PATIENCE", + "HYPERLEX_EARLY_STOP_MIN_EPOCHS", +) +_SCHEDULE_OFF = frozenset({"0", "false", "no", "off"}) +_SCHEDULE_ON = frozenset({"1", "true", "yes", "on"}) THRESHOLD_AUTHORIZATION_ENV = "HLX_THRESHOLD_AUTHORIZATION" THRESHOLD_SCHEMA = "hyperlex.threshold_authorization.v1" THRESHOLDS_BLOCKED = "BLOCKED_PENDING_OPERATOR_AUTHORIZATION" @@ -115,6 +129,7 @@ "HLX_RESERVE_BINDING", "HLX_BASELINE_ENV", "HLX_CANDIDATE_ENV", + SCIENTIFIC_VARIABLE_KEY, } ) LAUNCH_OVERLAY_KEYS = frozenset({"HYPERLEX_ALLOW_TRAIN", ADMISSION_ONLY_ENV}) @@ -520,7 +535,22 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]: "ADMISSION FAIL: reserve ledger events sha256 does not match the binding", ) ledger = IdentityLedger.load(ledger_dir) - counts = ledger.reserve_counts() + # Digest above covers the full append-only log. Counts below use only the + # active reserve for this experiment. + active = ledger.active_reserve_records(ctx.experiment_id) + counts = ledger.active_reserve_counts(ctx.experiment_id) + if not active: + foreign = [ + record + for record in ledger.identities.values() + if derived_state(record) == "EVAL_RESERVE" + ] + if foreign: + ctx.fail( + "holdout_reserve", + "ADMISSION FAIL: active reserve is bound to another experiment", + ) + ctx.fail("holdout_reserve", "ADMISSION FAIL: missing active reserve") expected_counts = binding.get("counts") if not isinstance(expected_counts, dict): ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding counts are missing") @@ -532,34 +562,13 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]: "holdout_reserve", f"ADMISSION FAIL: reserve slice {key} does not match the binding", ) - # The sealed reserve is EVAL_RESERVE. Historical spent and abandoned - # identities share the ledger and are not that reserve. An identity that - # still carries evaluation_reserved but has moved off EVAL_RESERVE fails. - reserved = [ - record - for record in ledger.identities.values() - if record.get("evaluation_reserved") or derived_state(record) == "EVAL_RESERVE" - ] - if not reserved: - ctx.fail("holdout_reserve", "ADMISSION FAIL: sealed evaluation reserve has no identities") - for record in reserved: - state = derived_state(record) - if state != "EVAL_RESERVE": - ctx.fail( - "holdout_reserve", - f"ADMISSION FAIL: reserve lifecycle {state} is not EVAL_RESERVE", - ) - if "identities" in binding and int(binding["identities"]) != len(reserved): + if "identities" in binding and int(binding["identities"]) != len(active): ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve identity count does not match the binding") return ledger, binding def _gate_disjoint(ctx: _Context, bundle: Mapping[str, Any], ledger: IdentityLedger) -> dict[str, int]: - reserved = [ - record - for record in ledger.identities.values() - if derived_state(record) == "EVAL_RESERVE" - ] + reserved = ledger.active_reserve_records(ctx.experiment_id) hashes = {record["normalized_text_sha256"] for record in reserved} ids: set[str] = set() for record in reserved: @@ -591,6 +600,67 @@ def _scientific_items(payload: Mapping[str, str]) -> dict[str, str]: return {key: value for key, value in payload.items() if key not in METADATA_KEYS and key not in LAUNCH_OVERLAY_KEYS} +def _declaration_value(payload: Mapping[str, str]) -> str: + raw = payload.get(SCIENTIFIC_VARIABLE_KEY) + if raw is None: + return "" + return str(raw).strip() + + +def _canon_schedule_int(raw: str | None) -> str | None: + if raw is None or not str(raw).strip(): + return None + text = str(raw).strip() + sign = "" + digits = text + if text[0] in "+-": + sign, digits = text[0], text[1:] + if not digits or not digits.isdigit(): + raise ValueError(f"schedule field must be an integer, got {raw!r}") + return str(int(sign + digits)) + + +def _canon_early_stop(raw: str | None) -> str: + if raw is None or not str(raw).strip(): + return "0" + token = str(raw).strip().lower() + if token in _SCHEDULE_OFF: + return "0" + if token in _SCHEDULE_ON: + return "1" + raise ValueError( + f"HYPERLEX_EARLY_STOP must be off or on, got {raw!r}" + ) + + +def schedule_bundle(payload: Mapping[str, str]) -> tuple[str | None, str, str | None, str | None]: + """One normalized schedule. Absent early-stop is off. Absent integers stay absent.""" + return ( + _canon_schedule_int(payload.get("HYPERLEX_TRAIN_EPOCHS")), + _canon_early_stop(payload.get("HYPERLEX_EARLY_STOP")), + _canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_PATIENCE")), + _canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_MIN_EPOCHS")), + ) + + +def _require_schedule_shape(side: str, bundle: tuple[str | None, str, str | None, str | None]) -> None: + epochs, stop, patience, minimum = bundle + if epochs is None or int(epochs) < 1: + raise ValueError(f"{side} HYPERLEX_TRAIN_EPOCHS must be an integer >= 1") + if stop == "0": + return + if patience is None or int(patience) < 0: + raise ValueError(f"{side} HYPERLEX_EARLY_STOP_PATIENCE must be an integer >= 0") + if minimum is None or int(minimum) < 1 or int(minimum) > int(epochs): + raise ValueError( + f"{side} HYPERLEX_EARLY_STOP_MIN_EPOCHS must be in 1..HYPERLEX_TRAIN_EPOCHS" + ) + + +def _metric_value(payload: Mapping[str, str]) -> str: + return str(payload.get(SELECT_METRIC_KEY) or "").strip() + + def _gate_single_variable(ctx: _Context) -> None: baseline_path = os.environ.get(BASELINE_ENV, "").strip() candidate_path = os.environ.get(CANDIDATE_ENV, "").strip() @@ -607,16 +677,74 @@ def _gate_single_variable(ctx: _Context) -> None: "single_variable", "ADMISSION FAIL: launch overlay is persisted in the sealed environment", ) + declared_baseline = _declaration_value(baseline) + declared_candidate = _declaration_value(candidate) + if declared_baseline != declared_candidate: + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable declaration does not match", + ) + if declared_baseline not in ("", DECLARED_SELECT_METRIC, DECLARED_TRAIN_SCHEDULE): + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable declaration is not " + f"{DECLARED_SELECT_METRIC} or {DECLARED_TRAIN_SCHEDULE}", + ) + process_declaration = str(os.environ.get(SCIENTIFIC_VARIABLE_KEY) or "").strip() + if process_declaration != declared_baseline: + ctx.fail( + "single_variable", + "ADMISSION FAIL: process environment " + f"{SCIENTIFIC_VARIABLE_KEY} does not match the sealed declaration", + ) + try: + base_bundle = schedule_bundle(baseline) + cand_bundle = schedule_bundle(candidate) + except ValueError as exc: + ctx.fail("single_variable", f"ADMISSION FAIL: {exc}") + bundle_differs = base_bundle != cand_bundle + declared = declared_baseline or DECLARED_SELECT_METRIC + if declared == DECLARED_TRAIN_SCHEDULE: + try: + _require_schedule_shape("baseline", base_bundle) + _require_schedule_shape("candidate", cand_bundle) + except ValueError as exc: + ctx.fail("single_variable", f"ADMISSION FAIL: {exc}") + if _metric_value(baseline) != _metric_value(candidate) or _metric_value(candidate) != SELECT_METRIC_CLASSIFY: + ctx.fail( + "single_variable", + "ADMISSION FAIL: selection metric difference is not part of train_schedule; " + f"both arms must set {SELECT_METRIC_KEY}={SELECT_METRIC_CLASSIFY}", + ) + elif bundle_differs: + ctx.fail( + "single_variable", + "ADMISSION FAIL: undeclared schedule difference", + ) base_sci = _scientific_items(baseline) cand_sci = _scientific_items(candidate) changed = sorted(set(base_sci) | set(cand_sci)) changed = [key for key in changed if base_sci.get(key) != cand_sci.get(key)] - if changed != [SELECT_METRIC_KEY]: + if not bundle_differs: + changed = [key for key in changed if key not in SCHEDULE_BUNDLE_KEYS] + if declared == DECLARED_TRAIN_SCHEDULE: + parts = ["train_schedule"] if bundle_differs else [] + parts.extend(key for key in changed if key not in SCHEDULE_BUNDLE_KEYS) + if parts != ["train_schedule"]: + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable count is " + f"{len(parts)}: {','.join(parts) or 'none'}", + ) + elif changed != [SELECT_METRIC_KEY]: ctx.fail( "single_variable", "ADMISSION FAIL: scientific variable count is " f"{len(changed)}: {','.join(changed) or 'none'}", ) + skip_on_baseline = {SELECT_METRIC_KEY} + if declared == DECLARED_TRAIN_SCHEDULE: + skip_on_baseline.update(SCHEDULE_BUNDLE_KEYS) for key, value in cand_sci.items(): if os.environ.get(key) != value: ctx.fail( @@ -624,7 +752,7 @@ def _gate_single_variable(ctx: _Context) -> None: f"ADMISSION FAIL: process environment {key} does not match the sealed candidate", ) for key, value in base_sci.items(): - if key == SELECT_METRIC_KEY: + if key in skip_on_baseline: continue if os.environ.get(key) != value: ctx.fail( From cd193ec7438c5a47c3e0efc46ed5c352533790b1 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:27:13 -0700 Subject: [PATCH 4/7] Admit one train_schedule variable and an active reserve. A preregistered train_schedule declaration compares the four schedule fields as a single scientific variable. Holdout counts use only current EVAL_RESERVE identities bound to the experiment, leaving spent evaluation_reserved flags and the append-only ledger unchanged. --- .../shadow/hyperlexical/identity_ledger.py | 28 +++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/scripts/shadow/hyperlexical/identity_ledger.py b/scripts/shadow/hyperlexical/identity_ledger.py index 9e9968ab..231e9861 100644 --- a/scripts/shadow/hyperlexical/identity_ledger.py +++ b/scripts/shadow/hyperlexical/identity_ledger.py @@ -380,6 +380,34 @@ def reserve_counts(self) -> dict[str, int]: counts[key] += 1 return counts + def active_reserve_records(self, experiment_id: str) -> list[dict[str, Any]]: + """Current EVAL_RESERVE identities bound only to this experiment. + + ``evaluation_reserved`` is historical evidence. Spent and abandoned + identities keep that flag and are not members. ``EVAL_BOUND`` is not + an active reserve. A binding list that names another experiment does + not satisfy this experiment. + """ + wanted = str(experiment_id or "") + if not wanted: + return [] + selected: list[dict[str, Any]] = [] + for record in self.identities.values(): + if derived_state(record) != "EVAL_RESERVE": + continue + bindings = [str(item) for item in record.get("experiment_bindings") or [] if str(item)] + if bindings == [wanted]: + selected.append(record) + return selected + + def active_reserve_counts(self, experiment_id: str) -> dict[str, int]: + counts = {key: 0 for key in REQUIRED_SLICES} + for record in self.active_reserve_records(experiment_id): + for key in slices_of(record): + if key in counts: + counts[key] += 1 + return counts + def screen( self, rows: Sequence[Mapping[str, Any]], From 79b26068edee30c3b36aea83c59ca78b89df5259 Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:28:41 -0700 Subject: [PATCH 5/7] Admit one train_schedule variable and an active reserve. A preregistered train_schedule declaration compares the four schedule fields as a single scientific variable. Holdout counts use only current EVAL_RESERVE identities bound to the experiment, leaving spent evaluation_reserved flags and the append-only ledger unchanged. --- specs/007-hyperlexical-model/evaluation-reserve.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/specs/007-hyperlexical-model/evaluation-reserve.md b/specs/007-hyperlexical-model/evaluation-reserve.md index a0201568..7f96f856 100644 --- a/specs/007-hyperlexical-model/evaluation-reserve.md +++ b/specs/007-hyperlexical-model/evaluation-reserve.md @@ -295,7 +295,11 @@ BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, sha256 `fc53676bd3 The canonical controlled-experiment holdout contract is `CONTROLLED_RESERVE`. A launch with `HLX_EXPERIMENT_ID` set is admitted by `admit_training_run`, which both preflight and `run_loop` call, in this order: experiment binding, launch gate, sealed reserve, pinned training input, train/reserve disjointness, single scientific variable, BEST and trunk digests, output directory, then ready. -A sealed reserve binding satisfies the holdout requirement when the ledger is `EVAL_RESERVE`, all four slices are present, the binding digest matches `events.jsonl`, and training overlap by row id and by normalized text hash is 0. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate. +A sealed reserve binding satisfies the holdout requirement when the active reserve is `EVAL_RESERVE` for the current experiment, all four slices are present on that active set, the binding digest matches the full `events.jsonl`, and training overlap by row id and by normalized text hash is 0 against that active set. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate. + +Active membership is the derived lifecycle `EVAL_RESERVE` whose experiment binding is only the current experiment. `evaluation_reserved` stays historical evidence and is not cleared when the lifecycle is `EVAL_SPENT`. Spent identities do not count, do not satisfy overlap, and cannot be reserved again. A ledger with no active identities is a missing active reserve. A live `EVAL_RESERVE` bound to another experiment does not satisfy the current experiment. Slice counts and the binding identity count use the active set. The events digest still covers the append-only log. + +The default scientific variable remains `HLX_SELECT_METRIC`: that key is the only allowed difference. `HLX_SCIENTIFIC_VARIABLE=train_schedule`, set to the same value on the baseline, the candidate, and the process, is the only composite. It normalizes `HYPERLEX_TRAIN_EPOCHS`, `HYPERLEX_EARLY_STOP`, `HYPERLEX_EARLY_STOP_PATIENCE`, and `HYPERLEX_EARLY_STOP_MIN_EPOCHS` into one variable. Both arms must set `HLX_SELECT_METRIC=classify_macro_f1_nonnone`. Any other difference fails. An undeclared schedule difference fails. A selection-metric difference is not part of `train_schedule`. Declaring the variable does not register or authorize an experiment. `HYPERLEX_ALLOW_TRAIN=1` is part of the effective environment hash. `HLX_ADMISSION_ONLY=1` is not. With that flag, `python -m hyperlexical.train --run` returns after admission and does not construct an optimizer, enter epoch 0, or take a gradient step. `TRAINING_READY` means that same admission returned `ADMISSION_PASS`. From 6f8d116e41468ad0cc0470abf72d21a3d94c5eeb Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:51:09 -0700 Subject: [PATCH 6/7] Record the four raw schedule keys as one variable. The intended arm leaves patience and minimum epochs unset on the control and sets both on the candidate. Main would count those four sealed keys separately. Declared train_schedule still admits them as one variable. --- .../test_hyperlexical_admission_contract.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/tests/shadow/test_hyperlexical_admission_contract.py b/tests/shadow/test_hyperlexical_admission_contract.py index 3a608497..05312885 100644 --- a/tests/shadow/test_hyperlexical_admission_contract.py +++ b/tests/shadow/test_hyperlexical_admission_contract.py @@ -14,6 +14,8 @@ from admission_fixtures import arm_controlled, classify_row, sha256_bytes # noqa: E402 from hyperlexical.admission import ( # noqa: E402 + LAUNCH_OVERLAY_KEYS, + METADATA_KEYS, AdmissionError, admit_training_run, schedule_bundle, @@ -168,6 +170,13 @@ def _spend(ledger: IdentityLedger, digest: str) -> None: ledger.transition(digest, "EVAL_SPENT", source_artifact="fixture", provenance="spend") +def _raw_scientific_diff(baseline: dict, candidate: dict) -> list[str]: + """Main's single-variable diff: one key per sealed field, metadata excluded.""" + skip = set(METADATA_KEYS) | set(LAUNCH_OVERLAY_KEYS) + keys = (set(baseline) | set(candidate)) - skip + return sorted(key for key in keys if baseline.get(key) != candidate.get(key)) + + def test_schedule_tokens_match_the_trainer(): from hyperlexical.admission import _SCHEDULE_OFF, _SCHEDULE_ON @@ -181,6 +190,14 @@ def test_select005_schedule_bundle_is_one_variable(monkeypatch, tmp_path): base = json.loads((tmp_path / "baseline-env.json").read_text(encoding="utf-8")) cand = json.loads((tmp_path / "candidate-env.json").read_text(encoding="utf-8")) assert schedule_bundle(base) != schedule_bundle(cand) + assert _raw_scientific_diff(base, cand) == [ + "HYPERLEX_EARLY_STOP", + "HYPERLEX_EARLY_STOP_MIN_EPOCHS", + "HYPERLEX_EARLY_STOP_PATIENCE", + "HYPERLEX_TRAIN_EPOCHS", + ] + assert "HYPERLEX_EARLY_STOP_PATIENCE" not in base + assert "HYPERLEX_EARLY_STOP_MIN_EPOCHS" not in base result = _admit() assert result.admission_result == "ADMISSION_PASS" assert result.status == "PREREGISTERED" From 3d9118ea9b489badf4216143a93befa1059a805f Mon Sep 17 00:00:00 2001 From: Daniel Meyer Date: Sat, 26 Sep 2026 22:55:54 -0700 Subject: [PATCH 7/7] Stop listing schedule keys as the scientific diff. The declared train_schedule arm is one variable. The test checks that the normalized bundles differ and that admission passes. --- .../test_hyperlexical_admission_contract.py | 17 ----------------- 1 file changed, 17 deletions(-) diff --git a/tests/shadow/test_hyperlexical_admission_contract.py b/tests/shadow/test_hyperlexical_admission_contract.py index 05312885..3a608497 100644 --- a/tests/shadow/test_hyperlexical_admission_contract.py +++ b/tests/shadow/test_hyperlexical_admission_contract.py @@ -14,8 +14,6 @@ from admission_fixtures import arm_controlled, classify_row, sha256_bytes # noqa: E402 from hyperlexical.admission import ( # noqa: E402 - LAUNCH_OVERLAY_KEYS, - METADATA_KEYS, AdmissionError, admit_training_run, schedule_bundle, @@ -170,13 +168,6 @@ def _spend(ledger: IdentityLedger, digest: str) -> None: ledger.transition(digest, "EVAL_SPENT", source_artifact="fixture", provenance="spend") -def _raw_scientific_diff(baseline: dict, candidate: dict) -> list[str]: - """Main's single-variable diff: one key per sealed field, metadata excluded.""" - skip = set(METADATA_KEYS) | set(LAUNCH_OVERLAY_KEYS) - keys = (set(baseline) | set(candidate)) - skip - return sorted(key for key in keys if baseline.get(key) != candidate.get(key)) - - def test_schedule_tokens_match_the_trainer(): from hyperlexical.admission import _SCHEDULE_OFF, _SCHEDULE_ON @@ -190,14 +181,6 @@ def test_select005_schedule_bundle_is_one_variable(monkeypatch, tmp_path): base = json.loads((tmp_path / "baseline-env.json").read_text(encoding="utf-8")) cand = json.loads((tmp_path / "candidate-env.json").read_text(encoding="utf-8")) assert schedule_bundle(base) != schedule_bundle(cand) - assert _raw_scientific_diff(base, cand) == [ - "HYPERLEX_EARLY_STOP", - "HYPERLEX_EARLY_STOP_MIN_EPOCHS", - "HYPERLEX_EARLY_STOP_PATIENCE", - "HYPERLEX_TRAIN_EPOCHS", - ] - assert "HYPERLEX_EARLY_STOP_PATIENCE" not in base - assert "HYPERLEX_EARLY_STOP_MIN_EPOCHS" not in base result = _admit() assert result.admission_result == "ADMISSION_PASS" assert result.status == "PREREGISTERED"