Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
180 changes: 154 additions & 26 deletions scripts/shadow/hyperlexical/admission.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@
from pathlib import Path
from typing import Any, Callable, Mapping

from .classify_metrics import SELECT_METRIC_CLASSIFY
from .export import repo_root
from .holdout_guard import (
HoldoutSpec,
Expand Down Expand Up @@ -65,6 +66,19 @@
TRUNK_SHA_ENV = "HLX_TRUNK_SHA256"
TRAIN_OUT_ENV = "HYPERLEX_TRAIN_OUT"
SELECT_METRIC_KEY = "HLX_SELECT_METRIC"
SCIENTIFIC_VARIABLE_KEY = "HLX_SCIENTIFIC_VARIABLE"
DECLARED_SELECT_METRIC = SELECT_METRIC_KEY
DECLARED_TRAIN_SCHEDULE = "train_schedule"
# The only composite. These four trainer fields are one variable, not a
# caller-supplied grouping of arbitrary keys.
SCHEDULE_BUNDLE_KEYS = (
"HYPERLEX_TRAIN_EPOCHS",
"HYPERLEX_EARLY_STOP",
"HYPERLEX_EARLY_STOP_PATIENCE",
"HYPERLEX_EARLY_STOP_MIN_EPOCHS",
)
_SCHEDULE_OFF = frozenset({"0", "false", "no", "off"})
_SCHEDULE_ON = frozenset({"1", "true", "yes", "on"})
THRESHOLD_AUTHORIZATION_ENV = "HLX_THRESHOLD_AUTHORIZATION"
THRESHOLD_SCHEMA = "hyperlex.threshold_authorization.v1"
THRESHOLDS_BLOCKED = "BLOCKED_PENDING_OPERATOR_AUTHORIZATION"
Expand Down Expand Up @@ -115,6 +129,7 @@
"HLX_RESERVE_BINDING",
"HLX_BASELINE_ENV",
"HLX_CANDIDATE_ENV",
SCIENTIFIC_VARIABLE_KEY,
}
)
LAUNCH_OVERLAY_KEYS = frozenset({"HYPERLEX_ALLOW_TRAIN", ADMISSION_ONLY_ENV})
Expand Down Expand Up @@ -520,7 +535,22 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]:
"ADMISSION FAIL: reserve ledger events sha256 does not match the binding",
)
ledger = IdentityLedger.load(ledger_dir)
counts = ledger.reserve_counts()
# Digest above covers the full append-only log. Counts below use only the
# active reserve for this experiment.
active = ledger.active_reserve_records(ctx.experiment_id)
counts = ledger.active_reserve_counts(ctx.experiment_id)
if not active:
foreign = [
record
for record in ledger.identities.values()
if derived_state(record) == "EVAL_RESERVE"
]
if foreign:
ctx.fail(
"holdout_reserve",
"ADMISSION FAIL: active reserve is bound to another experiment",
)
ctx.fail("holdout_reserve", "ADMISSION FAIL: missing active reserve")
expected_counts = binding.get("counts")
if not isinstance(expected_counts, dict):
ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding counts are missing")
Expand All @@ -532,34 +562,13 @@ def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]:
"holdout_reserve",
f"ADMISSION FAIL: reserve slice {key} does not match the binding",
)
# The sealed reserve is EVAL_RESERVE. Historical spent and abandoned
# identities share the ledger and are not that reserve. An identity that
# still carries evaluation_reserved but has moved off EVAL_RESERVE fails.
reserved = [
record
for record in ledger.identities.values()
if record.get("evaluation_reserved") or derived_state(record) == "EVAL_RESERVE"
]
if not reserved:
ctx.fail("holdout_reserve", "ADMISSION FAIL: sealed evaluation reserve has no identities")
for record in reserved:
state = derived_state(record)
if state != "EVAL_RESERVE":
ctx.fail(
"holdout_reserve",
f"ADMISSION FAIL: reserve lifecycle {state} is not EVAL_RESERVE",
)
if "identities" in binding and int(binding["identities"]) != len(reserved):
if "identities" in binding and int(binding["identities"]) != len(active):
ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve identity count does not match the binding")
return ledger, binding


def _gate_disjoint(ctx: _Context, bundle: Mapping[str, Any], ledger: IdentityLedger) -> dict[str, int]:
reserved = [
record
for record in ledger.identities.values()
if derived_state(record) == "EVAL_RESERVE"
]
reserved = ledger.active_reserve_records(ctx.experiment_id)
hashes = {record["normalized_text_sha256"] for record in reserved}
Comment on lines 570 to 572

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Badge Keep every live reserve out of training

When a shared ledger contains a current experiment's valid active reserve plus an EVAL_RESERVE belonging to another experiment, training data that overlaps the foreign reserve now passes because this gate checks only active_reserve_records(ctx.experiment_id). The same gap applies to EVAL_BOUND identities once another active set satisfies the slice counts. This silently contaminates evaluation material; keep binding/count validation scoped to the active set, but check training overlap against every EVAL_RESERVE and EVAL_BOUND identity in the ledger.

Useful? React with 👍 / 👎.

ids: set[str] = set()
for record in reserved:
Expand Down Expand Up @@ -591,6 +600,67 @@ def _scientific_items(payload: Mapping[str, str]) -> dict[str, str]:
return {key: value for key, value in payload.items() if key not in METADATA_KEYS and key not in LAUNCH_OVERLAY_KEYS}


def _declaration_value(payload: Mapping[str, str]) -> str:
raw = payload.get(SCIENTIFIC_VARIABLE_KEY)
if raw is None:
return ""
return str(raw).strip()


def _canon_schedule_int(raw: str | None) -> str | None:
if raw is None or not str(raw).strip():
return None
text = str(raw).strip()
sign = ""
digits = text
if text[0] in "+-":
sign, digits = text[0], text[1:]
if not digits or not digits.isdigit():
raise ValueError(f"schedule field must be an integer, got {raw!r}")
return str(int(sign + digits))


def _canon_early_stop(raw: str | None) -> str:
if raw is None or not str(raw).strip():
return "0"
token = str(raw).strip().lower()
if token in _SCHEDULE_OFF:
return "0"
if token in _SCHEDULE_ON:
return "1"
raise ValueError(
f"HYPERLEX_EARLY_STOP must be off or on, got {raw!r}"
)


def schedule_bundle(payload: Mapping[str, str]) -> tuple[str | None, str, str | None, str | None]:
"""One normalized schedule. Absent early-stop is off. Absent integers stay absent."""
return (
_canon_schedule_int(payload.get("HYPERLEX_TRAIN_EPOCHS")),
_canon_early_stop(payload.get("HYPERLEX_EARLY_STOP")),
_canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_PATIENCE")),
_canon_schedule_int(payload.get("HYPERLEX_EARLY_STOP_MIN_EPOCHS")),
)


def _require_schedule_shape(side: str, bundle: tuple[str | None, str, str | None, str | None]) -> None:
epochs, stop, patience, minimum = bundle
if epochs is None or int(epochs) < 1:
raise ValueError(f"{side} HYPERLEX_TRAIN_EPOCHS must be an integer >= 1")
if stop == "0":
return
if patience is None or int(patience) < 0:
raise ValueError(f"{side} HYPERLEX_EARLY_STOP_PATIENCE must be an integer >= 0")
if minimum is None or int(minimum) < 1 or int(minimum) > int(epochs):
raise ValueError(
f"{side} HYPERLEX_EARLY_STOP_MIN_EPOCHS must be in 1..HYPERLEX_TRAIN_EPOCHS"
)


def _metric_value(payload: Mapping[str, str]) -> str:
return str(payload.get(SELECT_METRIC_KEY) or "").strip()


def _gate_single_variable(ctx: _Context) -> None:
baseline_path = os.environ.get(BASELINE_ENV, "").strip()
candidate_path = os.environ.get(CANDIDATE_ENV, "").strip()
Expand All @@ -607,24 +677,82 @@ def _gate_single_variable(ctx: _Context) -> None:
"single_variable",
"ADMISSION FAIL: launch overlay is persisted in the sealed environment",
)
declared_baseline = _declaration_value(baseline)
declared_candidate = _declaration_value(candidate)
if declared_baseline != declared_candidate:
ctx.fail(
"single_variable",
"ADMISSION FAIL: scientific variable declaration does not match",
)
if declared_baseline not in ("", DECLARED_SELECT_METRIC, DECLARED_TRAIN_SCHEDULE):
ctx.fail(
"single_variable",
"ADMISSION FAIL: scientific variable declaration is not "
f"{DECLARED_SELECT_METRIC} or {DECLARED_TRAIN_SCHEDULE}",
)
process_declaration = str(os.environ.get(SCIENTIFIC_VARIABLE_KEY) or "").strip()
if process_declaration != declared_baseline:
ctx.fail(
"single_variable",
"ADMISSION FAIL: process environment "
f"{SCIENTIFIC_VARIABLE_KEY} does not match the sealed declaration",
)
try:
base_bundle = schedule_bundle(baseline)
cand_bundle = schedule_bundle(candidate)
except ValueError as exc:
Comment on lines +700 to +703

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Badge Bind the declared schedule into the environment hash

When preflight and launch reuse the same baseline/candidate JSON paths, changing the four schedule fields in those files and the matching process environment between calls leaves effective_environment_hash() unchanged because neither the schedule variables nor HLX_SCIENTIFIC_VARIABLE are in ENV_KEYS, and the files are represented only by their paths. Both calls can therefore pass these comparisons while the documented parity check falsely attests that the launch uses the preflighted schedule; include these process inputs or sealed-file content digests in the hash.

Useful? React with 👍 / 👎.

ctx.fail("single_variable", f"ADMISSION FAIL: {exc}")
bundle_differs = base_bundle != cand_bundle
declared = declared_baseline or DECLARED_SELECT_METRIC
if declared == DECLARED_TRAIN_SCHEDULE:
try:
_require_schedule_shape("baseline", base_bundle)
_require_schedule_shape("candidate", cand_bundle)
except ValueError as exc:
ctx.fail("single_variable", f"ADMISSION FAIL: {exc}")
if _metric_value(baseline) != _metric_value(candidate) or _metric_value(candidate) != SELECT_METRIC_CLASSIFY:
ctx.fail(
"single_variable",
"ADMISSION FAIL: selection metric difference is not part of train_schedule; "
f"both arms must set {SELECT_METRIC_KEY}={SELECT_METRIC_CLASSIFY}",
)
elif bundle_differs:
ctx.fail(
"single_variable",
"ADMISSION FAIL: undeclared schedule difference",
)
base_sci = _scientific_items(baseline)
cand_sci = _scientific_items(candidate)
changed = sorted(set(base_sci) | set(cand_sci))
changed = [key for key in changed if base_sci.get(key) != cand_sci.get(key)]
if changed != [SELECT_METRIC_KEY]:
if not bundle_differs:
changed = [key for key in changed if key not in SCHEDULE_BUNDLE_KEYS]
if declared == DECLARED_TRAIN_SCHEDULE:
parts = ["train_schedule"] if bundle_differs else []
parts.extend(key for key in changed if key not in SCHEDULE_BUNDLE_KEYS)
if parts != ["train_schedule"]:
ctx.fail(
"single_variable",
"ADMISSION FAIL: scientific variable count is "
f"{len(parts)}: {','.join(parts) or 'none'}",
)
elif changed != [SELECT_METRIC_KEY]:
ctx.fail(
"single_variable",
"ADMISSION FAIL: scientific variable count is "
f"{len(changed)}: {','.join(changed) or 'none'}",
)
skip_on_baseline = {SELECT_METRIC_KEY}
if declared == DECLARED_TRAIN_SCHEDULE:
skip_on_baseline.update(SCHEDULE_BUNDLE_KEYS)
for key, value in cand_sci.items():
if os.environ.get(key) != value:
ctx.fail(
"single_variable",
f"ADMISSION FAIL: process environment {key} does not match the sealed candidate",
)
for key, value in base_sci.items():
if key == SELECT_METRIC_KEY:
if key in skip_on_baseline:
continue
if os.environ.get(key) != value:
ctx.fail(
Expand Down
28 changes: 28 additions & 0 deletions scripts/shadow/hyperlexical/identity_ledger.py
Original file line number Diff line number Diff line change
Expand Up @@ -380,6 +380,34 @@ def reserve_counts(self) -> dict[str, int]:
counts[key] += 1
return counts

def active_reserve_records(self, experiment_id: str) -> list[dict[str, Any]]:
"""Current EVAL_RESERVE identities bound only to this experiment.

``evaluation_reserved`` is historical evidence. Spent and abandoned
identities keep that flag and are not members. ``EVAL_BOUND`` is not
an active reserve. A binding list that names another experiment does
not satisfy this experiment.
"""
wanted = str(experiment_id or "")
if not wanted:
return []
selected: list[dict[str, Any]] = []
for record in self.identities.values():
if derived_state(record) != "EVAL_RESERVE":
continue
bindings = [str(item) for item in record.get("experiment_bindings") or [] if str(item)]
if bindings == [wanted]:
selected.append(record)
return selected

def active_reserve_counts(self, experiment_id: str) -> dict[str, int]:
counts = {key: 0 for key in REQUIRED_SLICES}
for record in self.active_reserve_records(experiment_id):
for key in slices_of(record):
if key in counts:
counts[key] += 1
return counts

def screen(
self,
rows: Sequence[Mapping[str, Any]],
Expand Down
6 changes: 5 additions & 1 deletion specs/007-hyperlexical-model/evaluation-reserve.md
Original file line number Diff line number Diff line change
Expand Up @@ -295,7 +295,11 @@ BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, sha256 `fc53676bd3

The canonical controlled-experiment holdout contract is `CONTROLLED_RESERVE`. A launch with `HLX_EXPERIMENT_ID` set is admitted by `admit_training_run`, which both preflight and `run_loop` call, in this order: experiment binding, launch gate, sealed reserve, pinned training input, train/reserve disjointness, single scientific variable, BEST and trunk digests, output directory, then ready.

A sealed reserve binding satisfies the holdout requirement when the ledger is `EVAL_RESERVE`, all four slices are present, the binding digest matches `events.jsonl`, and training overlap by row id and by normalized text hash is 0. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate.
A sealed reserve binding satisfies the holdout requirement when the active reserve is `EVAL_RESERVE` for the current experiment, all four slices are present on that active set, the binding digest matches the full `events.jsonl`, and training overlap by row id and by normalized text hash is 0 against that active set. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate.

Active membership is the derived lifecycle `EVAL_RESERVE` whose experiment binding is only the current experiment. `evaluation_reserved` stays historical evidence and is not cleared when the lifecycle is `EVAL_SPENT`. Spent identities do not count, do not satisfy overlap, and cannot be reserved again. A ledger with no active identities is a missing active reserve. A live `EVAL_RESERVE` bound to another experiment does not satisfy the current experiment. Slice counts and the binding identity count use the active set. The events digest still covers the append-only log.

The default scientific variable remains `HLX_SELECT_METRIC`: that key is the only allowed difference. `HLX_SCIENTIFIC_VARIABLE=train_schedule`, set to the same value on the baseline, the candidate, and the process, is the only composite. It normalizes `HYPERLEX_TRAIN_EPOCHS`, `HYPERLEX_EARLY_STOP`, `HYPERLEX_EARLY_STOP_PATIENCE`, and `HYPERLEX_EARLY_STOP_MIN_EPOCHS` into one variable. Both arms must set `HLX_SELECT_METRIC=classify_macro_f1_nonnone`. Any other difference fails. An undeclared schedule difference fails. A selection-metric difference is not part of `train_schedule`. Declaring the variable does not register or authorize an experiment.

`HYPERLEX_ALLOW_TRAIN=1` is part of the effective environment hash. `HLX_ADMISSION_ONLY=1` is not. With that flag, `python -m hyperlexical.train --run` returns after admission and does not construct an optimizer, enter epoch 0, or take a gradient step. `TRAINING_READY` means that same admission returned `ADMISSION_PASS`.

Expand Down
Loading
Loading