diff --git a/scripts/shadow/hyperlexical/admission.py b/scripts/shadow/hyperlexical/admission.py index d7bcc65d..01a4e89e 100644 --- a/scripts/shadow/hyperlexical/admission.py +++ b/scripts/shadow/hyperlexical/admission.py @@ -32,6 +32,7 @@ from pathlib import Path from typing import Any, Callable, Mapping +from .classify_metrics import SELECT_METRIC_CLASSIFY from .export import repo_root from .holdout_guard import ( HoldoutSpec, @@ -65,6 +66,19 @@ TRUNK_SHA_ENV = "HLX_TRUNK_SHA256" TRAIN_OUT_ENV = "HYPERLEX_TRAIN_OUT" SELECT_METRIC_KEY = "HLX_SELECT_METRIC" +SCIENTIFIC_VARIABLE_KEY = "HLX_SCIENTIFIC_VARIABLE" +DECLARED_SELECT_METRIC = SELECT_METRIC_KEY +DECLARED_TRAIN_SCHEDULE = "train_schedule" +# The only composite. These four trainer fields are one variable, not a +# caller-supplied grouping of arbitrary keys. +SCHEDULE_BUNDLE_KEYS = ( + "HYPERLEX_TRAIN_EPOCHS", + "HYPERLEX_EARLY_STOP", + "HYPERLEX_EARLY_STOP_PATIENCE", + "HYPERLEX_EARLY_STOP_MIN_EPOCHS", +) +_SCHEDULE_OFF = frozenset({"0", "false", "no", "off"}) +_SCHEDULE_ON = frozenset({"1", "true", "yes", "on"}) THRESHOLD_AUTHORIZATION_ENV = "HLX_THRESHOLD_AUTHORIZATION" THRESHOLD_SCHEMA = "hyperlex.threshold_authorization.v1" THRESHOLDS_BLOCKED = "BLOCKED_PENDING_OPERATOR_AUTHORIZATION" @@ -115,6 +129,7 @@ "HLX_RESERVE_BINDING", "HLX_BASELINE_ENV", "HLX_CANDIDATE_ENV", + SCIENTIFIC_VARIABLE_KEY, } ) LAUNCH_OVERLAY_KEYS = frozenset({"HYPERLEX_ALLOW_TRAIN", ADMISSION_ONLY_ENV}) @@ -520,7 +535,22 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]: "ADMISSION FAIL: reserve ledger events sha256 does not match the binding", ) ledger = IdentityLedger.load(ledger_dir) - counts = ledger.reserve_counts() + # Digest above covers the full append-only log. Counts below use only the + # active reserve for this experiment. + active = ledger.active_reserve_records(ctx.experiment_id) + counts = ledger.active_reserve_counts(ctx.experiment_id) + if not active: + foreign = [ + record + for record in ledger.identities.values() + if derived_state(record) == "EVAL_RESERVE" + ] + if foreign: + ctx.fail( + "holdout_reserve", + "ADMISSION FAIL: active reserve is bound to another experiment", + ) + ctx.fail("holdout_reserve", "ADMISSION FAIL: missing active reserve") expected_counts = binding.get("counts") if not isinstance(expected_counts, dict): ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding counts are missing") @@ -532,34 +562,13 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]: "holdout_reserve", f"ADMISSION FAIL: reserve slice {key} does not match the binding", ) - # The sealed reserve is EVAL_RESERVE. Historical spent and abandoned - # identities share the ledger and are not that reserve. An identity that - # still carries evaluation_reserved but has moved off EVAL_RESERVE fails. - reserved = [ - record - for record in ledger.identities.values() - if record.get("evaluation_reserved") or derived_state(record) == "EVAL_RESERVE" - ] - if not reserved: - ctx.fail("holdout_reserve", "ADMISSION FAIL: sealed evaluation reserve has no identities") - for record in reserved: - state = derived_state(record) - if state != "EVAL_RESERVE": - ctx.fail( - "holdout_reserve", - f"ADMISSION FAIL: reserve lifecycle {state} is not EVAL_RESERVE", - ) - if "identities" in binding and int(binding["identities"]) != len(reserved): + if "identities" in binding and int(binding["identities"]) != len(active): ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve identity count does not match the binding") return ledger, binding def _gate_disjoint(ctx: _Context, bundle: Mapping[str, Any], ledger: IdentityLedger) -> dict[str, int]: - reserved = [ - record - for record in ledger.identities.values() - if derived_state(record) == "EVAL_RESERVE" - ] + reserved = ledger.active_reserve_records(ctx.experiment_id) hashes = {record["normalized_text_sha256"] for record in reserved} ids: set[str] = set() for record in reserved: @@ -591,6 +600,67 @@ def _scientific_items(payload: Mapping[str, str]) -> dict[str, str]: return {key: value for key, value in payload.items() if key not in METADATA_KEYS and key not in LAUNCH_OVERLAY_KEYS} +def _declaration_value(payload: Mapping[str, str]) -> str: + raw = payload.get(SCIENTIFIC_VARIABLE_KEY) + if raw is None: + return "" + return str(raw).strip() + + +def _canon_schedule_int(raw: str | None) -> str | None: + if raw is None or not str(raw).strip(): + return None + text = str(raw).strip() + sign = "" + digits = text + if text[0] in "+-": + sign, digits = text[0], text[1:] + if not digits or not digits.isdigit(): + raise ValueError(f"schedule field must be an integer, got {raw!r}") + return str(int(sign + digits)) + + +def _canon_early_stop(raw: str | None) -> str: + if raw is None or not str(raw).strip(): + return "0" + token = str(raw).strip().lower() + if token in _SCHEDULE_OFF: + return "0" + if token in _SCHEDULE_ON: + return "1" + raise ValueError( + f"HYPERLEX_EARLY_STOP must be off or on, got {raw!r}" + ) + + +def schedule_bundle(payload: Mapping[str, str]) -> tuple[str | None, str, str | None, str | None]: + """One normalized schedule. Absent early-stop is off. Absent integers stay absent.""" + return ( + _canon_schedule_int(payload.get("HYPERLEX_TRAIN_EPOCHS")), + _canon_early_stop(payload.get("HYPERLEX_EARLY_STOP")), + _canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_PATIENCE")), + _canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_MIN_EPOCHS")), + ) + + +def _require_schedule_shape(side: str, bundle: tuple[str | None, str, str | None, str | None]) -> None: + epochs, stop, patience, minimum = bundle + if epochs is None or int(epochs) < 1: + raise ValueError(f"{side} HYPERLEX_TRAIN_EPOCHS must be an integer >= 1") + if stop == "0": + return + if patience is None or int(patience) < 0: + raise ValueError(f"{side} HYPERLEX_EARLY_STOP_PATIENCE must be an integer >= 0") + if minimum is None or int(minimum) < 1 or int(minimum) > int(epochs): + raise ValueError( + f"{side} HYPERLEX_EARLY_STOP_MIN_EPOCHS must be in 1..HYPERLEX_TRAIN_EPOCHS" + ) + + +def _metric_value(payload: Mapping[str, str]) -> str: + return str(payload.get(SELECT_METRIC_KEY) or "").strip() + + def _gate_single_variable(ctx: _Context) -> None: baseline_path = os.environ.get(BASELINE_ENV, "").strip() candidate_path = os.environ.get(CANDIDATE_ENV, "").strip() @@ -607,16 +677,74 @@ def _gate_single_variable(ctx: _Context) -> None: "single_variable", "ADMISSION FAIL: launch overlay is persisted in the sealed environment", ) + declared_baseline = _declaration_value(baseline) + declared_candidate = _declaration_value(candidate) + if declared_baseline != declared_candidate: + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable declaration does not match", + ) + if declared_baseline not in ("", DECLARED_SELECT_METRIC, DECLARED_TRAIN_SCHEDULE): + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable declaration is not " + f"{DECLARED_SELECT_METRIC} or {DECLARED_TRAIN_SCHEDULE}", + ) + process_declaration = str(os.environ.get(SCIENTIFIC_VARIABLE_KEY) or "").strip() + if process_declaration != declared_baseline: + ctx.fail( + "single_variable", + "ADMISSION FAIL: process environment " + f"{SCIENTIFIC_VARIABLE_KEY} does not match the sealed declaration", + ) + try: + base_bundle = schedule_bundle(baseline) + cand_bundle = schedule_bundle(candidate) + except ValueError as exc: + ctx.fail("single_variable", f"ADMISSION FAIL: {exc}") + bundle_differs = base_bundle != cand_bundle + declared = declared_baseline or DECLARED_SELECT_METRIC + if declared == DECLARED_TRAIN_SCHEDULE: + try: + _require_schedule_shape("baseline", base_bundle) + _require_schedule_shape("candidate", cand_bundle) + except ValueError as exc: + ctx.fail("single_variable", f"ADMISSION FAIL: {exc}") + if _metric_value(baseline) != _metric_value(candidate) or _metric_value(candidate) != SELECT_METRIC_CLASSIFY: + ctx.fail( + "single_variable", + "ADMISSION FAIL: selection metric difference is not part of train_schedule; " + f"both arms must set {SELECT_METRIC_KEY}={SELECT_METRIC_CLASSIFY}", + ) + elif bundle_differs: + ctx.fail( + "single_variable", + "ADMISSION FAIL: undeclared schedule difference", + ) base_sci = _scientific_items(baseline) cand_sci = _scientific_items(candidate) changed = sorted(set(base_sci) | set(cand_sci)) changed = [key for key in changed if base_sci.get(key) != cand_sci.get(key)] - if changed != [SELECT_METRIC_KEY]: + if not bundle_differs: + changed = [key for key in changed if key not in SCHEDULE_BUNDLE_KEYS] + if declared == DECLARED_TRAIN_SCHEDULE: + parts = ["train_schedule"] if bundle_differs else [] + parts.extend(key for key in changed if key not in SCHEDULE_BUNDLE_KEYS) + if parts != ["train_schedule"]: + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable count is " + f"{len(parts)}: {','.join(parts) or 'none'}", + ) + elif changed != [SELECT_METRIC_KEY]: ctx.fail( "single_variable", "ADMISSION FAIL: scientific variable count is " f"{len(changed)}: {','.join(changed) or 'none'}", ) + skip_on_baseline = {SELECT_METRIC_KEY} + if declared == DECLARED_TRAIN_SCHEDULE: + skip_on_baseline.update(SCHEDULE_BUNDLE_KEYS) for key, value in cand_sci.items(): if os.environ.get(key) != value: ctx.fail( @@ -624,7 +752,7 @@ def _gate_single_variable(ctx: _Context) -> None: f"ADMISSION FAIL: process environment {key} does not match the sealed candidate", ) for key, value in base_sci.items(): - if key == SELECT_METRIC_KEY: + if key in skip_on_baseline: continue if os.environ.get(key) != value: ctx.fail( diff --git a/scripts/shadow/hyperlexical/identity_ledger.py b/scripts/shadow/hyperlexical/identity_ledger.py index 9e9968ab..231e9861 100644 --- a/scripts/shadow/hyperlexical/identity_ledger.py +++ b/scripts/shadow/hyperlexical/identity_ledger.py @@ -380,6 +380,34 @@ def reserve_counts(self) -> dict[str, int]: counts[key] += 1 return counts + def active_reserve_records(self, experiment_id: str) -> list[dict[str, Any]]: + """Current EVAL_RESERVE identities bound only to this experiment. + + ``evaluation_reserved`` is historical evidence. Spent and abandoned + identities keep that flag and are not members. ``EVAL_BOUND`` is not + an active reserve. A binding list that names another experiment does + not satisfy this experiment. + """ + wanted = str(experiment_id or "") + if not wanted: + return [] + selected: list[dict[str, Any]] = [] + for record in self.identities.values(): + if derived_state(record) != "EVAL_RESERVE": + continue + bindings = [str(item) for item in record.get("experiment_bindings") or [] if str(item)] + if bindings == [wanted]: + selected.append(record) + return selected + + def active_reserve_counts(self, experiment_id: str) -> dict[str, int]: + counts = {key: 0 for key in REQUIRED_SLICES} + for record in self.active_reserve_records(experiment_id): + for key in slices_of(record): + if key in counts: + counts[key] += 1 + return counts + def screen( self, rows: Sequence[Mapping[str, Any]], diff --git a/specs/007-hyperlexical-model/evaluation-reserve.md b/specs/007-hyperlexical-model/evaluation-reserve.md index a0201568..7f96f856 100644 --- a/specs/007-hyperlexical-model/evaluation-reserve.md +++ b/specs/007-hyperlexical-model/evaluation-reserve.md @@ -295,7 +295,11 @@ BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, sha256 `fc53676bd3 The canonical controlled-experiment holdout contract is `CONTROLLED_RESERVE`. A launch with `HLX_EXPERIMENT_ID` set is admitted by `admit_training_run`, which both preflight and `run_loop` call, in this order: experiment binding, launch gate, sealed reserve, pinned training input, train/reserve disjointness, single scientific variable, BEST and trunk digests, output directory, then ready. -A sealed reserve binding satisfies the holdout requirement when the ledger is `EVAL_RESERVE`, all four slices are present, the binding digest matches `events.jsonl`, and training overlap by row id and by normalized text hash is 0. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate. +A sealed reserve binding satisfies the holdout requirement when the active reserve is `EVAL_RESERVE` for the current experiment, all four slices are present on that active set, the binding digest matches the full `events.jsonl`, and training overlap by row id and by normalized text hash is 0 against that active set. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate. + +Active membership is the derived lifecycle `EVAL_RESERVE` whose experiment binding is only the current experiment. `evaluation_reserved` stays historical evidence and is not cleared when the lifecycle is `EVAL_SPENT`. Spent identities do not count, do not satisfy overlap, and cannot be reserved again. A ledger with no active identities is a missing active reserve. A live `EVAL_RESERVE` bound to another experiment does not satisfy the current experiment. Slice counts and the binding identity count use the active set. The events digest still covers the append-only log. + +The default scientific variable remains `HLX_SELECT_METRIC`: that key is the only allowed difference. `HLX_SCIENTIFIC_VARIABLE=train_schedule`, set to the same value on the baseline, the candidate, and the process, is the only composite. It normalizes `HYPERLEX_TRAIN_EPOCHS`, `HYPERLEX_EARLY_STOP`, `HYPERLEX_EARLY_STOP_PATIENCE`, and `HYPERLEX_EARLY_STOP_MIN_EPOCHS` into one variable. Both arms must set `HLX_SELECT_METRIC=classify_macro_f1_nonnone`. Any other difference fails. An undeclared schedule difference fails. A selection-metric difference is not part of `train_schedule`. Declaring the variable does not register or authorize an experiment. `HYPERLEX_ALLOW_TRAIN=1` is part of the effective environment hash. `HLX_ADMISSION_ONLY=1` is not. With that flag, `python -m hyperlexical.train --run` returns after admission and does not construct an optimizer, enter epoch 0, or take a gradient step. `TRAINING_READY` means that same admission returned `ADMISSION_PASS`. diff --git a/tests/shadow/test_hyperlexical_admission_contract.py b/tests/shadow/test_hyperlexical_admission_contract.py new file mode 100644 index 00000000..3a608497 --- /dev/null +++ b/tests/shadow/test_hyperlexical_admission_contract.py @@ -0,0 +1,372 @@ +"""Admission contract: one train_schedule variable, active reserve only.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +from admission_fixtures import arm_controlled, classify_row, sha256_bytes # noqa: E402 +from hyperlexical.admission import ( # noqa: E402 + AdmissionError, + admit_training_run, + schedule_bundle, +) +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import IdentityLedger, derived_state # noqa: E402 +from hyperlexical.loop import _EARLY_STOP_OFF, _EARLY_STOP_ON # noqa: E402 +from hyperlexical.selection_surface import row_id # noqa: E402 + +EXPERIMENT = "HLX-EXP-TEST" +OTHER = "HLX-EXP-OTHER" +SELECT = "classify_macro_f1_nonnone" +CONTROL_SCHEDULE = { + "HYPERLEX_TRAIN_EPOCHS": "40", + "HYPERLEX_EARLY_STOP": "0", +} +CANDIDATE_SCHEDULE = { + "HYPERLEX_TRAIN_EPOCHS": "12", + "HYPERLEX_EARLY_STOP": "1", + "HYPERLEX_EARLY_STOP_PATIENCE": "4", + "HYPERLEX_EARLY_STOP_MIN_EPOCHS": "4", +} + + +def _admit(): + return admit_training_run( + trunk=Path(__import__("os").environ["HYPERLEX_TRUNK_DIR"]), + out_dir=Path(__import__("os").environ["HYPERLEX_TRAIN_OUT"]), + ) + + +def _write_envs(tmp_path: Path, baseline: dict, candidate: dict) -> None: + (tmp_path / "baseline-env.json").write_text( + json.dumps(baseline, sort_keys=True), encoding="utf-8" + ) + (tmp_path / "candidate-env.json").write_text( + json.dumps(candidate, sort_keys=True), encoding="utf-8" + ) + + +def _arm_schedule( + monkeypatch, + tmp_path: Path, + *, + declared: str | None, + baseline_extra: dict | None = None, + candidate_extra: dict | None = None, + metric: str = SELECT, + baseline_metric: str | None = None, + train_text: str = "train row", +): + armed = arm_controlled(monkeypatch, tmp_path, [classify_row(train_text)]) + baseline = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SELECT_METRIC": metric if baseline_metric is None else baseline_metric, + } + candidate = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SELECT_METRIC": metric, + "HLX_EXPERIMENT_ID": EXPERIMENT, + "HYPERLEX_TRAIN_OUT": str(armed["out"]), + } + baseline.update(CONTROL_SCHEDULE) + candidate.update(CANDIDATE_SCHEDULE) + if baseline_extra: + baseline.update(baseline_extra) + if candidate_extra: + candidate.update(candidate_extra) + if declared is not None: + baseline["HLX_SCIENTIFIC_VARIABLE"] = declared + candidate["HLX_SCIENTIFIC_VARIABLE"] = declared + monkeypatch.setenv("HLX_SCIENTIFIC_VARIABLE", declared) + else: + monkeypatch.delenv("HLX_SCIENTIFIC_VARIABLE", raising=False) + _write_envs(tmp_path, baseline, candidate) + monkeypatch.setenv("HLX_SELECT_METRIC", metric) + for key, value in candidate.items(): + if key in { + "HLX_EXPERIMENT_ID", + "HYPERLEX_TRAIN_OUT", + "HLX_SCIENTIFIC_VARIABLE", + "HLX_SELECT_METRIC", + }: + continue + monkeypatch.setenv(key, value) + return armed + + +def _ledger_bytes(directory: Path) -> dict[str, bytes]: + return {path.name: path.read_bytes() for path in sorted(directory.iterdir())} + + +def _bind(root: Path, ledger: IdentityLedger, experiment_id: str) -> Path: + ledger_dir = root / "ledger" + if ledger_dir.exists(): + raise AssertionError("ledger directory already exists") + ledger.save(ledger_dir) + events_sha = sha256_bytes((ledger_dir / "events.jsonl").read_bytes()) + active = ledger.active_reserve_records(experiment_id) + binding = { + "schema": "hyperlex.reserve_binding.v1", + "experiment_id": experiment_id, + "ledger_events_sha256": events_sha, + "counts": ledger.active_reserve_counts(experiment_id), + "identities": len(active), + "lifecycle": "EVAL_RESERVE", + } + path = root / "reserve-binding.json" + path.write_text(json.dumps(binding, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return path + + +def _observe_reserve(ledger: IdentityLedger, text: str, label: dict, experiment_id: str) -> str: + digest = normalized_text_sha256(text) + ledger.observe( + digest, + source_artifact="fixture", + row_ids=[row_id({"text": text, "task": label["task"], "split": "val"})], + labels=[label], + provenance="fixture", + experiment_id=experiment_id, + ) + ledger.transition(digest, "EVAL_RESERVE", source_artifact="fixture", provenance="fixture") + return digest + + +def _fresh_reserve(experiment_id: str) -> IdentityLedger: + ledger = IdentityLedger() + _observe_reserve( + ledger, + "reserve classify surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + experiment_id, + ) + _observe_reserve( + ledger, + "reserve unbind surface", + { + "task": "unbind", + "class": "INFERRED", + "lineage": "none", + "split": "val", + "unbind_clean": True, + }, + experiment_id, + ) + return ledger + + +def _spend(ledger: IdentityLedger, digest: str) -> None: + ledger.transition(digest, "EVAL_BOUND", source_artifact="fixture", provenance="bind") + ledger.transition(digest, "EVAL_SPENT", source_artifact="fixture", provenance="spend") + + +def test_schedule_tokens_match_the_trainer(): + from hyperlexical.admission import _SCHEDULE_OFF, _SCHEDULE_ON + + assert _SCHEDULE_OFF == _EARLY_STOP_OFF + assert _SCHEDULE_ON == _EARLY_STOP_ON + + +def test_select005_schedule_bundle_is_one_variable(monkeypatch, tmp_path): + armed = _arm_schedule(monkeypatch, tmp_path, declared="train_schedule") + before = _ledger_bytes(armed["reserve"]["ledger"]) + base = json.loads((tmp_path / "baseline-env.json").read_text(encoding="utf-8")) + cand = json.loads((tmp_path / "candidate-env.json").read_text(encoding="utf-8")) + assert schedule_bundle(base) != schedule_bundle(cand) + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.status == "PREREGISTERED" + assert result.receipt["optimizer_loaded"] is False + assert result.receipt["epochs"] == 0 + assert result.receipt["gradient_steps"] == 0 + assert _ledger_bytes(armed["reserve"]["ledger"]) == before + assert not armed["out"].exists() + + +def test_extra_scientific_difference_is_a_second_variable(monkeypatch, tmp_path): + _arm_schedule( + monkeypatch, + tmp_path, + declared="train_schedule", + candidate_extra={"HYPERLEX_TRAIN_LR": "1e-4"}, + ) + with pytest.raises(AdmissionError, match="scientific variable count is 2") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + message = str(exc.value) + assert "train_schedule" in message + assert "HYPERLEX_TRAIN_LR" in message + + +def test_selection_metric_difference_is_not_inside_train_schedule(monkeypatch, tmp_path): + _arm_schedule( + monkeypatch, + tmp_path, + declared="train_schedule", + baseline_metric="unbind_exact", + ) + with pytest.raises(AdmissionError, match="not part of train_schedule") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + + +def test_undeclared_schedule_difference_refuses_admission(monkeypatch, tmp_path): + _arm_schedule(monkeypatch, tmp_path, declared=None) + with pytest.raises(AdmissionError, match="undeclared schedule difference") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "single_variable" + + +def test_declared_select_metric_mode_still_passes(monkeypatch, tmp_path): + armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + baseline = {"HYPERLEX_FILLER_FILTER": "off", "HLX_SCIENTIFIC_VARIABLE": "HLX_SELECT_METRIC"} + candidate = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SCIENTIFIC_VARIABLE": "HLX_SELECT_METRIC", + "HLX_SELECT_METRIC": SELECT, + "HLX_EXPERIMENT_ID": EXPERIMENT, + "HYPERLEX_TRAIN_OUT": str(armed["out"]), + } + _write_envs(tmp_path, baseline, candidate) + monkeypatch.setenv("HLX_SCIENTIFIC_VARIABLE", "HLX_SELECT_METRIC") + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.receipt["optimizer_loaded"] is False + + +def test_implicit_select_metric_mode_still_passes(monkeypatch, tmp_path): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.delenv("HLX_SCIENTIFIC_VARIABLE", raising=False) + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.status == "PREREGISTERED" + + +def test_spent_only_ledger_has_zero_active_counts_and_keeps_the_flag(monkeypatch, tmp_path): + ledger = _fresh_reserve(EXPERIMENT) + for record in list(ledger.identities.values()): + assert record["evaluation_reserved"] is True + _spend(ledger, record["normalized_text_sha256"]) + assert record["evaluation_reserved"] is True + assert derived_state(record) == "EVAL_SPENT" + assert ledger.active_reserve_counts(EXPERIMENT) == { + "classify": 0, + "classify_observed": 0, + "classify_non_none": 0, + "unbind_clean": 0, + } + root = tmp_path / "spent" + root.mkdir() + binding = _bind(root, ledger, EXPERIMENT) + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + before = _ledger_bytes(root / "ledger") + with pytest.raises(AdmissionError, match="missing active reserve") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "holdout_reserve" + assert _ledger_bytes(root / "ledger") == before + reloaded = IdentityLedger.load(root / "ledger") + assert all(record["evaluation_reserved"] is True for record in reloaded.identities.values()) + assert reloaded.active_reserve_counts(EXPERIMENT)["classify"] == 0 + + +def test_mixed_ledger_counts_only_active_identities(monkeypatch, tmp_path): + ledger = _fresh_reserve(EXPERIMENT) + spent = _observe_reserve( + ledger, + "already spent surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + EXPERIMENT, + ) + _spend(ledger, spent) + assert ledger.identity(spent)["evaluation_reserved"] is True + active_hashes = {record["normalized_text_sha256"] for record in ledger.active_reserve_records(EXPERIMENT)} + assert spent not in active_hashes + assert len(active_hashes) == 2 + counts = ledger.active_reserve_counts(EXPERIMENT) + assert counts["classify"] == 1 + assert counts["unbind_clean"] == 1 + root = tmp_path / "mixed" + root.mkdir() + binding = _bind(root, ledger, EXPERIMENT) + armed = _arm_schedule(monkeypatch, tmp_path, declared="train_schedule") + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + before = _ledger_bytes(root / "ledger") + result = _admit() + assert result.admission_result == "ADMISSION_PASS" + assert result.receipt["reserve_counts"]["classify"] == 1 + assert _ledger_bytes(root / "ledger") == before + assert not armed["out"].exists() + + +def test_wrong_experiment_active_reserve_refuses(monkeypatch, tmp_path): + ledger = _fresh_reserve(OTHER) + assert ledger.active_reserve_counts(EXPERIMENT) == { + "classify": 0, + "classify_observed": 0, + "classify_non_none": 0, + "unbind_clean": 0, + } + assert ledger.active_reserve_counts(OTHER)["classify"] == 1 + root = tmp_path / "foreign" + root.mkdir() + # Binding names the current experiment. The identities do not. + binding = _bind(root, ledger, EXPERIMENT) + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(root / "ledger")) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(binding)) + with pytest.raises(AdmissionError, match="bound to another experiment") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "holdout_reserve" + + +def test_active_reserve_overlap_still_fails(monkeypatch, tmp_path): + arm_controlled(monkeypatch, tmp_path, [classify_row("reserve classify surface")]) + with pytest.raises(AdmissionError, match="overlaps the sealed reserve") as exc: + _admit() + assert exc.value.receipt["failed_gate"] == "train_reserve_disjointness" + + +def test_spent_identity_cannot_be_reallocated_or_reused(): + ledger = IdentityLedger() + digest = _observe_reserve( + ledger, + "spent surface", + {"task": "classify", "class": "OBSERVED", "lineage": "gaming-meta", "split": "val"}, + EXPERIMENT, + ) + _spend(ledger, digest) + record = ledger.identity(digest) + assert record is not None + assert record["evaluation_reserved"] is True + assert derived_state(record) == "EVAL_SPENT" + with pytest.raises(SystemExit, match="monotonic"): + ledger.transition(digest, "EVAL_RESERVE", source_artifact="fixture", provenance="reuse") + report = ledger.admit( + [ + { + "text": "spent surface", + "task": "classify", + "split": "val", + "class": "OBSERVED", + "lineage": "gaming-meta", + } + ], + batch_id="reuse-attempt", + source_artifact="fixture", + targets={"classify": 1, "classify_observed": 1, "classify_non_none": 1, "unbind_clean": 1}, + ) + assert report["rejected_existing_identities"]["EVAL_SPENT"] == 1 + assert report["unique_admitted_to_eval_reserve"] == 0 + assert ledger.identity(digest)["evaluation_reserved"] is True + assert derived_state(ledger.identity(digest)) == "EVAL_SPENT" diff --git a/tests/shadow/test_hyperlexical_admission_parity.py b/tests/shadow/test_hyperlexical_admission_parity.py index f60657d4..e33d0a07 100644 --- a/tests/shadow/test_hyperlexical_admission_parity.py +++ b/tests/shadow/test_hyperlexical_admission_parity.py @@ -359,24 +359,24 @@ def test_historical_spent_identity_is_not_the_controlled_reserve(monkeypatch, tm def test_reserve_identity_that_left_eval_reserve_rejects_both(monkeypatch, tmp_path, capsys): - from hyperlexical.identity_ledger import IdentityLedger, derived_state + from hyperlexical.identity_ledger import IdentityLedger, derived_state, slices_of armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) ledger = IdentityLedger.load(armed["reserve"]["ledger"]) - reserved = [ + classify = [ digest for digest, record in ledger.identities.items() - if derived_state(record) == "EVAL_RESERVE" + if derived_state(record) == "EVAL_RESERVE" and "classify" in slices_of(record) ] ledger.transition( - reserved[0], + classify[0], "EVAL_BOUND", source_artifact="fixture", provenance="fixture", ) _rebind_ledger(monkeypatch, tmp_path, armed, ledger) pre = _assert_same_failure(monkeypatch, capsys) - assert "EVAL_BOUND" in pre["error"] + assert "reserve slice classify is absent" in pre["error"] assert pre["failed_gate"] == "holdout_reserve"