diff --git a/scripts/shadow/hyperlexical/admission.py b/scripts/shadow/hyperlexical/admission.py new file mode 100644 index 00000000..d7bcc65d --- /dev/null +++ b/scripts/shadow/hyperlexical/admission.py @@ -0,0 +1,670 @@ +"""One admission path for preflight and the trainer. + +RUNE.PREFLIGHT_LAUNCH_PARITY(x) = + effective_environment_hash(preflight) + == effective_environment_hash(launch) + AND admission_gate_sequence(preflight) + == admission_gate_sequence(launch) + AND admission_result(preflight) + == admission_result(launch) + +Controlled experiments (``HLX_EXPERIMENT_ID`` set) use ``CONTROLLED_RESERVE``. +A sealed evaluation reserve with zero training overlap satisfies the holdout +requirement. ``HLX_HOLDOUT_MANIFESTS`` does not. ``HLX_ALLOW_NO_HOLDOUT`` does +not. Legacy launches that are not controlled experiments still use +``require_holdout_for_training``. + +``HLX_ADMISSION_ONLY=1`` is not part of the environment hash. The trainer +returns after these gates and does not construct an optimizer. + +``TRAINING_READY`` exists only when all three are true: these scientific +gates passed, a sealed threshold authorization matches this experiment, and +admission returns ``ADMISSION_PASS``. Without that decision rule the status +stays ``PREREGISTERED`` and ``ready_to_train`` stays false. An admission pass +is not a training launch. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +from typing import Any, Callable, Mapping + +from .export import repo_root +from .holdout_guard import ( + HoldoutSpec, + allow_no_holdout, + load_holdout_spec, + normalized_text_sha256, + require_holdout_for_training, +) +from .identity_ledger import ( + RESERVE_LEDGER_ENV, + IdentityLedger, + derived_state, +) +from .selection_surface import row_id +from .train_input import ( + TrainInputAdmissionError, + load_training_bundle, + train_input_receipt, +) + +CONTRACT_RESERVE = "CONTROLLED_RESERVE" +CONTRACT_MANIFEST = "LEGACY_MANIFEST" +CONTRACT_UNCONTROLLED = "UNCONTROLLED" +BINDING_SCHEMA = "hyperlex.reserve_binding.v1" +ADMISSION_ONLY_ENV = "HLX_ADMISSION_ONLY" +RESERVE_BINDING_ENV = "HLX_RESERVE_BINDING" +BASELINE_ENV = "HLX_BASELINE_ENV" +CANDIDATE_ENV = "HLX_CANDIDATE_ENV" +BEST_SHA_ENV = "HLX_BEST_SHA256" +BEST_WEIGHTS_ENV = "HLX_BEST_WEIGHTS" +TRUNK_SHA_ENV = "HLX_TRUNK_SHA256" +TRAIN_OUT_ENV = "HYPERLEX_TRAIN_OUT" +SELECT_METRIC_KEY = "HLX_SELECT_METRIC" +THRESHOLD_AUTHORIZATION_ENV = "HLX_THRESHOLD_AUTHORIZATION" +THRESHOLD_SCHEMA = "hyperlex.threshold_authorization.v1" +THRESHOLDS_BLOCKED = "BLOCKED_PENDING_OPERATOR_AUTHORIZATION" +REQUIRED_SLICES = ("classify", "classify_observed", "classify_non_none", "unbind_clean") + +GATE_SEQUENCE = ( + "experiment_binding", + "launch_gate", + "holdout_reserve", + "pinned_training_input", + "train_reserve_disjointness", + "single_variable", + "best_trunk", + "output_directory", + "ready", +) + +# Process environment that admission reads. ``HLX_ADMISSION_ONLY`` is omitted +# so a dry launch and preflight hash the same binding. +ENV_KEYS = ( + "HLX_EXPERIMENT_ID", + "HLX_THRESHOLD_AUTHORIZATION", + "HYPERLEX_ALLOW_TRAIN", + "HLX_EVAL_RESERVE_LEDGER", + "HLX_RESERVE_BINDING", + "HLX_HOLDOUT_MANIFESTS", + "HLX_ALLOW_NO_HOLDOUT", + "HLX_TRAIN_EXPORT_PATH", + "HLX_TRAIN_EXPORT_SHA256", + "HLX_TRAIN_EXPORT_ROWS", + "HLX_BASELINE_ENV", + "HLX_CANDIDATE_ENV", + "HLX_SELECT_METRIC", + "HLX_BEST_SHA256", + "HLX_BEST_WEIGHTS", + "HLX_TRUNK_SHA256", + "HYPERLEX_TRUNK_DIR", + "HYPERLEX_TRAIN_OUT", + "HYPERLEX_INCLUDE_LIVE", +) + +METADATA_KEYS = frozenset( + { + "HLX_EXPERIMENT_ID", + "HYPERLEX_TRAIN_OUT", + "HYPERLEX_UNBIND_RESIDUAL_DUMP", + "HLX_EVAL_RESERVE_LEDGER", + "HLX_RESERVE_BINDING", + "HLX_BASELINE_ENV", + "HLX_CANDIDATE_ENV", + } +) +LAUNCH_OVERLAY_KEYS = frozenset({"HYPERLEX_ALLOW_TRAIN", ADMISSION_ONLY_ENV}) + +MISSING_RESERVE_REASON = ( + "ADMISSION FAIL: HYPERLEX_ALLOW_TRAIN=1 but no sealed evaluation reserve. " + "Set HLX_EVAL_RESERVE_LEDGER to the sealed ledger. " + "CONTROLLED_RESERVE does not accept a legacy holdout manifest." +) + + +class AdmissionError(SystemExit): + """Admission refused. ``SystemExit`` so the trainer CLI does not treat it as a crash.""" + + def __init__(self, message: str, receipt: dict[str, Any]) -> None: + super().__init__(message) + self.receipt = receipt + + +class AdmissionResult: + def __init__( + self, + *, + ready: bool, + status: str | None, + admission_result: str, + contract: str, + launch_armed: bool, + bundle: dict[str, Any] | None, + holdout_spec: HoldoutSpec, + input_receipt: dict[str, Any] | None, + disjoint_receipt: dict[str, Any] | None, + reserve_receipt: dict[str, Any] | None, + receipt: dict[str, Any], + error: str | None = None, + ) -> None: + self.ready = ready + self.status = status + self.admission_result = admission_result + self.contract = contract + self.launch_armed = launch_armed + self.bundle = bundle + self.holdout_spec = holdout_spec + self.input_receipt = input_receipt + self.disjoint_receipt = disjoint_receipt + self.reserve_receipt = reserve_receipt + self.receipt = receipt + self.error = error + + +def effective_environment_material() -> dict[str, str | None]: + """Admission environment. Absent variables are null. No timestamps.""" + return {key: os.environ.get(key) for key in ENV_KEYS} + + +def effective_environment_hash() -> str: + payload = json.dumps( + effective_environment_material(), + sort_keys=True, + separators=(",", ":"), + ) + return hashlib.sha256(payload.encode("utf-8")).hexdigest() + + +def _empty_spec() -> HoldoutSpec: + return HoldoutSpec(frozenset(), frozenset(), ()) + + +def _sha256_file(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _read_env_file(path: Path, what: str) -> dict[str, str]: + if not path.is_file(): + raise AdmissionError( + f"ADMISSION FAIL: {what} is missing: {path}", + {}, + ) + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + raise AdmissionError(f"ADMISSION FAIL: {what} is not JSON", {}) from exc + if not isinstance(payload, dict): + raise AdmissionError(f"ADMISSION FAIL: {what} must be a JSON object", {}) + out: dict[str, str] = {} + for key, value in payload.items(): + if not isinstance(key, str) or not isinstance(value, str): + raise AdmissionError( + f"ADMISSION FAIL: {what} values must be strings", + {}, + ) + out[key] = value + return out + + +def admit_training_run( + *, + include_live: bool = False, + live_store: Path | None = None, + export_dataset: Callable[..., dict[str, Any]] | None = None, + trunk: Path | None = None, + out_dir: Path | None = None, +) -> AdmissionResult: + """Run the trainer admission gates. Does not construct an optimizer.""" + if export_dataset is None: + from .export import export_dataset as export_dataset + + experiment_id = os.environ.get("HLX_EXPERIMENT_ID", "").strip() + controlled = bool(experiment_id) + launch = os.environ.get("HYPERLEX_ALLOW_TRAIN") == "1" + contract = CONTRACT_RESERVE if controlled else (CONTRACT_MANIFEST if launch else CONTRACT_UNCONTROLLED) + ctx = _Context( + include_live=include_live, + live_store=live_store, + export_dataset=export_dataset, + trunk=trunk, + out_dir=out_dir, + experiment_id=experiment_id, + controlled=controlled, + launch=launch, + contract=contract, + ) + if not launch: + return _admit_unarmed(ctx) + if not controlled: + return _admit_legacy_launch(ctx) + return _admit_controlled(ctx) + + +class _Context: + def __init__(self, **kwargs: Any) -> None: + self.__dict__.update(kwargs) + self.env_hash = effective_environment_hash() + self.failed_gate: str | None = None + + def receipt(self, **fields: Any) -> dict[str, Any]: + payload = { + "schema": "hyperlex.admission.v1", + "controlled_holdout_contract": self.contract, + "admission_gate_sequence": list(GATE_SEQUENCE), + "failed_gate": self.failed_gate, + "environment_hash": self.env_hash, + "experiment_id": self.experiment_id or None, + "launch_armed": self.launch, + "training_started": False, + "epochs": 0, + "gradient_steps": 0, + "optimizer_loaded": False, + "best_moved": False, + "hlx_allow_no_holdout": allow_no_holdout(), + "holdout_admitted": False, + } + payload.update(fields) + return payload + + def fail(self, gate: str, message: str, **fields: Any) -> None: + self.failed_gate = gate + raise AdmissionError(message, self.receipt(**fields)) + + +def _load_bundle(ctx: _Context) -> dict[str, Any]: + try: + return load_training_bundle( + repo_root(), + include_live=ctx.include_live, + live_store=ctx.live_store, + export_dataset=ctx.export_dataset, + ) + except TrainInputAdmissionError as exc: + ctx.fail("pinned_training_input", str(exc)) + raise AssertionError("unreachable") from exc + + +def _unarmed_result(ctx: _Context, bundle: dict[str, Any], spec: HoldoutSpec) -> AdmissionResult: + proof = train_input_receipt(bundle) + status = "NOT_READY" if ctx.controlled else None + receipt = ctx.receipt( + admission_result="NOT_ARMED", + status=status, + ready_to_train=False, + **proof, + ) + return AdmissionResult( + ready=False, + status=status, + admission_result="NOT_ARMED", + contract=ctx.contract, + launch_armed=False, + bundle=bundle, + holdout_spec=spec, + input_receipt=proof, + disjoint_receipt=None, + reserve_receipt=None, + receipt=receipt, + ) + + +def _admit_unarmed(ctx: _Context) -> AdmissionResult: + """ALLOW_TRAIN is unset. Load a pinned bundle when one is declared. + + This is not launch admission. Callers that still execute ``run_loop`` + keep the previous non-launch behavior. + """ + bundle = _load_bundle(ctx) + spec = load_holdout_spec() + from .holdout_guard import assert_pinned_holdout_disjoint + + if ctx.experiment_id: + assert_pinned_holdout_disjoint(bundle["rows"], spec) + ledger = os.environ.get(RESERVE_LEDGER_ENV, "").strip() + if ledger: + from .identity_ledger import assert_training_disjoint_from_reserve + + assert_training_disjoint_from_reserve(bundle["rows"], IdentityLedger.load(ledger)) + return _unarmed_result(ctx, bundle, spec) + + +def _admit_legacy_launch(ctx: _Context) -> AdmissionResult: + try: + spec = require_holdout_for_training() + except SystemExit as exc: + ctx.fail("holdout_reserve", str(exc)) + raise AssertionError("unreachable") from exc + bundle = _load_bundle(ctx) + proof = train_input_receipt(bundle) + ready = _trunk_config_ok(ctx.trunk) + receipt = ctx.receipt( + admission_result="ADMISSION_PASS" if ready else "NOT_READY", + status=None, + ready_to_train=ready, + holdout_admitted=bool(spec.manifests) or allow_no_holdout(), + **proof, + ) + return AdmissionResult( + ready=ready, + status=None, + admission_result=receipt["admission_result"], + contract=CONTRACT_MANIFEST, + launch_armed=True, + bundle=bundle, + holdout_spec=spec, + input_receipt=proof, + disjoint_receipt=None, + reserve_receipt=None, + receipt=receipt, + ) + + +def _trunk_config_ok(trunk: Path | None) -> bool: + return bool(trunk) and trunk.is_dir() and (trunk / "config.json").is_file() + + +def _admit_controlled(ctx: _Context) -> AdmissionResult: + _gate_experiment(ctx) + _gate_launch(ctx) + ledger, binding = _gate_reserve(ctx) + bundle = _load_bundle(ctx) + proof = train_input_receipt(bundle) + if proof["training_input_mode"] != "PINNED_EXPORT" or proof["live_export_generation_enabled"]: + ctx.fail( + "pinned_training_input", + "ADMISSION FAIL: controlled experiment did not consume a pinned export", + **proof, + ) + overlap = _gate_disjoint(ctx, bundle, ledger) + _gate_single_variable(ctx) + _gate_best_trunk(ctx) + _gate_output(ctx) + decision_sealed, decision_state = _decision_authorization(ctx) + status = "TRAINING_READY" if decision_sealed else "PREREGISTERED" + spec = _empty_spec() + disjoint = { + "holdout_train_row_id_overlap": 0, + "holdout_train_text_hash_overlap": 0, + "holdout_filter_training_rows_removed": 0, + "holdout_training_disjoint": True, + } + receipt = ctx.receipt( + admission_result="ADMISSION_PASS", + status=status, + ready_to_train=decision_sealed, + scientific_contract_sealed=True, + decision_rule_sealed=decision_sealed, + decision_threshold_state=decision_state, + training_launch_authorized=False, + holdout_admitted=True, + holdout_state="EVAL_RESERVE", + holdout_experiment_id=ctx.experiment_id, + holdout_manifest_sha256=None, + reserve_lifecycle="EVAL_RESERVE", + reserve_events_sha256=binding["ledger_events_sha256"], + reserve_counts=binding["counts"], + **proof, + **overlap, + **disjoint, + ) + return AdmissionResult( + ready=True, + status=status, + admission_result="ADMISSION_PASS", + contract=CONTRACT_RESERVE, + launch_armed=True, + bundle=bundle, + holdout_spec=spec, + input_receipt=proof, + disjoint_receipt=disjoint, + reserve_receipt=overlap, + receipt=receipt, + ) + + +def _decision_authorization(ctx: _Context) -> tuple[bool, str]: + """A missing authorization stays blocked. A bad file fails closed.""" + raw = os.environ.get(THRESHOLD_AUTHORIZATION_ENV, "").strip() + if not raw: + return False, THRESHOLDS_BLOCKED + path = Path(raw) + if not path.is_file(): + ctx.fail("ready", "ADMISSION FAIL: threshold authorization path is not a file") + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError: + ctx.fail("ready", "ADMISSION FAIL: threshold authorization is not JSON") + if not isinstance(payload, dict) or payload.get("schema") != THRESHOLD_SCHEMA: + ctx.fail( + "ready", + "ADMISSION FAIL: threshold authorization schema is not " + THRESHOLD_SCHEMA, + ) + if payload.get("experiment_id") != ctx.experiment_id: + ctx.fail( + "ready", + "ADMISSION FAIL: threshold authorization experiment_id does not match", + ) + if payload.get("sealed") is not True: + ctx.fail("ready", "ADMISSION FAIL: threshold authorization is not sealed") + thresholds = payload.get("decision_thresholds") + if not isinstance(thresholds, dict) or not thresholds: + ctx.fail( + "ready", + "ADMISSION FAIL: threshold authorization does not seal numeric decision thresholds", + ) + for key, value in thresholds.items(): + if not isinstance(key, str) or isinstance(value, bool) or not isinstance(value, (int, float)): + ctx.fail( + "ready", + "ADMISSION FAIL: threshold authorization does not seal numeric decision thresholds", + ) + return True, "SEALED" + + +def _gate_experiment(ctx: _Context) -> None: + if os.environ.get("HLX_HOLDOUT_MANIFESTS", "").strip(): + ctx.fail( + "experiment_binding", + "ADMISSION FAIL: legacy holdout manifest is not the CONTROLLED_RESERVE contract", + ) + if allow_no_holdout(): + ctx.fail( + "experiment_binding", + "ADMISSION FAIL: HLX_ALLOW_NO_HOLDOUT does not admit a controlled experiment", + ) + + +def _gate_launch(ctx: _Context) -> None: + if not _trunk_config_ok(ctx.trunk): + ctx.fail( + "launch_gate", + "ADMISSION FAIL: trunk config is missing", + ) + + +def _gate_reserve(ctx: _Context) -> tuple[IdentityLedger, dict[str, Any]]: + raw_ledger = os.environ.get(RESERVE_LEDGER_ENV, "").strip() + raw_binding = os.environ.get(RESERVE_BINDING_ENV, "").strip() + if not raw_ledger or not raw_binding: + ctx.fail("holdout_reserve", MISSING_RESERVE_REASON) + ledger_dir = Path(raw_ledger) + binding_path = Path(raw_binding) + if not (ledger_dir / "events.jsonl").is_file(): + ctx.fail("holdout_reserve", MISSING_RESERVE_REASON) + if not binding_path.is_file(): + ctx.fail("holdout_reserve", MISSING_RESERVE_REASON) + try: + binding = json.loads(binding_path.read_text(encoding="utf-8")) + except json.JSONDecodeError: + ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding is not JSON") + if not isinstance(binding, dict) or binding.get("schema") != BINDING_SCHEMA: + ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding schema is not sealed") + if binding.get("experiment_id") != ctx.experiment_id: + ctx.fail( + "holdout_reserve", + "ADMISSION FAIL: reserve binding experiment_id does not match HLX_EXPERIMENT_ID", + ) + if binding.get("lifecycle") != "EVAL_RESERVE": + ctx.fail( + "holdout_reserve", + "ADMISSION FAIL: reserve lifecycle is not EVAL_RESERVE", + ) + actual_events = _sha256_file(ledger_dir / "events.jsonl") + if binding.get("ledger_events_sha256") != actual_events: + ctx.fail( + "holdout_reserve", + "ADMISSION FAIL: reserve ledger events sha256 does not match the binding", + ) + ledger = IdentityLedger.load(ledger_dir) + counts = ledger.reserve_counts() + expected_counts = binding.get("counts") + if not isinstance(expected_counts, dict): + ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve binding counts are missing") + for key in REQUIRED_SLICES: + if counts.get(key, 0) < 1: + ctx.fail("holdout_reserve", f"ADMISSION FAIL: reserve slice {key} is absent") + if int(expected_counts.get(key, -1)) != int(counts[key]): + ctx.fail( + "holdout_reserve", + f"ADMISSION FAIL: reserve slice {key} does not match the binding", + ) + # The sealed reserve is EVAL_RESERVE. Historical spent and abandoned + # identities share the ledger and are not that reserve. An identity that + # still carries evaluation_reserved but has moved off EVAL_RESERVE fails. + reserved = [ + record + for record in ledger.identities.values() + if record.get("evaluation_reserved") or derived_state(record) == "EVAL_RESERVE" + ] + if not reserved: + ctx.fail("holdout_reserve", "ADMISSION FAIL: sealed evaluation reserve has no identities") + for record in reserved: + state = derived_state(record) + if state != "EVAL_RESERVE": + ctx.fail( + "holdout_reserve", + f"ADMISSION FAIL: reserve lifecycle {state} is not EVAL_RESERVE", + ) + if "identities" in binding and int(binding["identities"]) != len(reserved): + ctx.fail("holdout_reserve", "ADMISSION FAIL: reserve identity count does not match the binding") + return ledger, binding + + +def _gate_disjoint(ctx: _Context, bundle: Mapping[str, Any], ledger: IdentityLedger) -> dict[str, int]: + reserved = [ + record + for record in ledger.identities.values() + if derived_state(record) == "EVAL_RESERVE" + ] + hashes = {record["normalized_text_sha256"] for record in reserved} + ids: set[str] = set() + for record in reserved: + ids.update(str(item) for item in record.get("row_ids") or []) + id_overlap = 0 + text_overlap = 0 + for row in bundle["rows"]: + if row_id(row) in ids: + id_overlap += 1 + if normalized_text_sha256(str(row.get("text") or "")) in hashes: + text_overlap += 1 + if id_overlap or text_overlap: + ctx.fail( + "train_reserve_disjointness", + "ADMISSION FAIL: training input overlaps the sealed reserve " + f"(row_id_overlap={id_overlap}, text_hash_overlap={text_overlap})", + reserve_train_row_id_overlap=id_overlap, + reserve_train_text_hash_overlap=text_overlap, + training_rows_removed=0, + ) + return { + "reserve_train_row_id_overlap": 0, + "reserve_train_text_hash_overlap": 0, + "training_rows_removed": 0, + } + + +def _scientific_items(payload: Mapping[str, str]) -> dict[str, str]: + return {key: value for key, value in payload.items() if key not in METADATA_KEYS and key not in LAUNCH_OVERLAY_KEYS} + + +def _gate_single_variable(ctx: _Context) -> None: + baseline_path = os.environ.get(BASELINE_ENV, "").strip() + candidate_path = os.environ.get(CANDIDATE_ENV, "").strip() + if not baseline_path or not candidate_path: + ctx.fail("single_variable", "ADMISSION FAIL: baseline and candidate environments are not bound") + try: + baseline = _read_env_file(Path(baseline_path), "baseline environment") + candidate = _read_env_file(Path(candidate_path), "candidate environment") + except AdmissionError as exc: + ctx.fail("single_variable", str(exc)) + for key in LAUNCH_OVERLAY_KEYS: + if key in candidate or key in baseline: + ctx.fail( + "single_variable", + "ADMISSION FAIL: launch overlay is persisted in the sealed environment", + ) + base_sci = _scientific_items(baseline) + cand_sci = _scientific_items(candidate) + changed = sorted(set(base_sci) | set(cand_sci)) + changed = [key for key in changed if base_sci.get(key) != cand_sci.get(key)] + if changed != [SELECT_METRIC_KEY]: + ctx.fail( + "single_variable", + "ADMISSION FAIL: scientific variable count is " + f"{len(changed)}: {','.join(changed) or 'none'}", + ) + for key, value in cand_sci.items(): + if os.environ.get(key) != value: + ctx.fail( + "single_variable", + f"ADMISSION FAIL: process environment {key} does not match the sealed candidate", + ) + for key, value in base_sci.items(): + if key == SELECT_METRIC_KEY: + continue + if os.environ.get(key) != value: + ctx.fail( + "single_variable", + f"ADMISSION FAIL: process environment {key} does not match the sealed baseline", + ) + + +def _gate_best_trunk(ctx: _Context) -> None: + best_sha = os.environ.get(BEST_SHA_ENV, "").strip() + best_path = os.environ.get(BEST_WEIGHTS_ENV, "").strip() + trunk_sha = os.environ.get(TRUNK_SHA_ENV, "").strip() + if not best_sha or not best_path or not trunk_sha: + ctx.fail("best_trunk", "ADMISSION FAIL: BEST or trunk digest is not bound") + best = Path(best_path) + if not best.is_file(): + ctx.fail("best_trunk", "ADMISSION FAIL: BEST weights are missing") + actual_best = _sha256_file(best) + if actual_best != best_sha: + ctx.fail("best_trunk", "ADMISSION FAIL: BEST weights sha256 mismatch") + trunk = ctx.trunk + weights = (trunk / "model.safetensors") if trunk else Path() + if trunk is None or not weights.is_file(): + ctx.fail("best_trunk", "ADMISSION FAIL: trunk weights are missing") + if _sha256_file(weights) != trunk_sha: + ctx.fail("best_trunk", "ADMISSION FAIL: trunk weights sha256 mismatch") + + +def _gate_output(ctx: _Context) -> None: + raw = os.environ.get(TRAIN_OUT_ENV, "").strip() + if not raw: + ctx.fail("output_directory", "ADMISSION FAIL: output directory is not bound") + out = Path(raw) + if ctx.out_dir is not None and ctx.out_dir.resolve() != out.resolve(): + ctx.fail( + "output_directory", + "ADMISSION FAIL: output directory does not match HYPERLEX_TRAIN_OUT", + ) + if out.exists(): + ctx.fail("output_directory", "ADMISSION FAIL: output directory collision") + lock = Path(str(out) + ".lock") + if lock.exists(): + ctx.fail("output_directory", "ADMISSION FAIL: conflicting training run") diff --git a/scripts/shadow/hyperlexical/build_identity_ledger.py b/scripts/shadow/hyperlexical/build_identity_ledger.py new file mode 100644 index 00000000..a2b3d57e --- /dev/null +++ b/scripts/shadow/hyperlexical/build_identity_ledger.py @@ -0,0 +1,364 @@ +"""Build the GEN-0 text-identity ledger from receipt-backed artifacts. + +Does not train, score, draw a holdout, or write raw text. Exits 2 when the +eligible-universe census does not reproduce the sealed exhaustion result. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.holdout_eligibility import census +from hyperlexical.holdout_guard import operational_status, _file_sha256, _read_manifest +from hyperlexical.identity_ledger import ( + GENERATION, + POLICY_ID, + IdentityLedger, + acquisition_gap, + normalized_text_sha256, +) +from hyperlexical.selection_surface import row_id +from hyperlexical.soft_ceiling import clean_surface + +EXPECTED = { + "source_rows": 5005, + "excluded_by_split_test": 440, + "excluded_by_training_row_id": 3247, + "excluded_by_training_normalized_text_hash": 608, + "excluded_by_spent_or_abandoned": 710, + "eligible_rows": 0, + "classify_observed_eligible": 0, + "classify_non_none_eligible": 0, +} + + +def _file_sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _jsonl(path: Path) -> list[dict]: + rows = [] + with path.open(encoding="utf-8") as handle: + for line in handle: + if line.strip(): + rows.append(json.loads(line)) + return rows + + +def _require_status(path: Path, expected: str) -> dict: + record, ids, hashes = _read_manifest(str(path)) + status = operational_status(record) + if status != expected: + raise SystemExit(f"STOP: {path} operational status {status} != {expected}") + return {"record": record, "ids": ids, "hashes": hashes, "status": status} + + +def _clean_hashes(rows: list[dict], train_rows: list[dict]) -> set[str]: + unbind = [row for row in rows if row.get("task") == "unbind"] + train = [row for row in train_rows if row.get("split") == "train"] + kept, _account = clean_surface(unbind, train_rows=train) + return {normalized_text_sha256(str(row.get("text") or "")) for row in kept} + + +def _receipt_shas(models: Path) -> list[str]: + found: set[str] = set() + for path in models.rglob("train-receipt.json"): + text = str(path) + if "/src-" in text or "/stepB/" in text or "/wt-" in text: + continue + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + continue + sha = payload.get("data_sha256") + if isinstance(sha, str) and len(sha) == 64: + found.add(sha) + return sorted(found) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--pin", required=True) + parser.add_argument("--pin-sha256", required=True) + parser.add_argument("--pin-rows", type=int, required=True) + parser.add_argument("--store", required=True) + parser.add_argument("--store-sha256", required=True) + parser.add_argument("--store-rows", type=int, required=True) + parser.add_argument("--v2", required=True) + parser.add_argument("--rc1-scored", required=True) + parser.add_argument("--rc1-unbind-rows", required=True) + parser.add_argument("--select-001", required=True) + parser.add_argument("--select-002", required=True) + parser.add_argument("--train-aux", action="append", default=[], help="path|provenance") + parser.add_argument("--models", required=True) + parser.add_argument("--out", required=True) + args = parser.parse_args(argv) + + pin_path = Path(args.pin) + store_path = Path(args.store) + pin_sha = _file_sha(pin_path) + store_sha = _file_sha(store_path) + if pin_sha != args.pin_sha256: + raise SystemExit("STOP: pinned export sha256 mismatch") + if store_sha != args.store_sha256: + raise SystemExit("STOP: live store sha256 mismatch") + pin_rows = _jsonl(pin_path) + store_rows = _jsonl(store_path) + if len(pin_rows) != args.pin_rows or len(store_rows) != args.store_rows: + raise SystemExit("STOP: row count mismatch") + + v2 = _require_status(Path(args.v2), "SCORED_SPENT") + v2_payload = json.loads(Path(args.v2).read_text(encoding="utf-8")) + definition = str(v2_payload.get("text_hash_definition") or "") + if "normalize_group_text" not in definition: + raise SystemExit("STOP: v2 text hash definition is not the canonical normalizer") + select_001 = _require_status(Path(args.select_001), "UNSCORED_ABANDONED") + select_002 = _require_status(Path(args.select_002), "UNSCORED_ABANDONED") + rc1_payload = json.loads(Path(args.rc1_scored).read_text(encoding="utf-8")) + rc1_record = { + "path": args.rc1_scored, + "sha256": _file_sha256(Path(args.rc1_scored)), + "status": rc1_payload.get("status"), + } + if operational_status(rc1_record) != "SCORED_SPENT": + raise SystemExit("STOP: rc1 scored manifest is not SCORED_SPENT") + rc1_rows = _jsonl(Path(args.rc1_unbind_rows)) + + joined = 0 + mismatch = 0 + v2_by_id = dict(zip(v2_payload["row_ids"], v2_payload["normalized_text_sha256"])) + for row in store_rows: + ident = row_id(row) + if ident in v2_by_id: + joined += 1 + if normalized_text_sha256(str(row.get("text") or "")) != v2_by_id[ident]: + mismatch += 1 + if mismatch: + raise SystemExit(f"STOP: v2 manifest hashes diverge from canonical text identity ({mismatch})") + + report = census( + store_rows, + pin_rows, + extra_rows=rc1_rows, + extra_manifests=[args.v2, args.select_001, args.select_002], + ) + discrepancy = {key: [report.get(key), EXPECTED[key]] for key in EXPECTED if report.get(key) != EXPECTED[key]} + if discrepancy or not report.get("waterfall_sums_to_source"): + sys.stdout.write(json.dumps({"STOP": True, "discrepancy": discrepancy}, indent=2) + "\n") + return 2 + + clean = _clean_hashes(store_rows, pin_rows) + clean |= _clean_hashes(rc1_rows, pin_rows) + ledger = IdentityLedger() + pin_artifact = str(pin_path) + consumed: set[str] = set() + for row in pin_rows: + digest = ledger.observe_row( + row, + source_artifact=pin_artifact, + provenance="morph78 train receipt data_sha256 matches this export", + catalogued=True, + ) + if digest not in consumed: + ledger.mark_historical( + digest, + "training_consumed", + source_artifact=pin_artifact, + provenance="pinned morph78 training export consumed by seed-morph78", + ) + consumed.add(digest) + for item in args.train_aux: + path_text, _sep, reason = item.partition("|") + aux = Path(path_text) + for row in _jsonl(aux): + digest = ledger.observe_row( + row, + source_artifact=str(aux), + provenance=reason, + catalogued=True, + ) + if digest not in consumed: + ledger.mark_historical( + digest, + "training_consumed", + source_artifact=str(aux), + provenance=reason, + ) + consumed.add(digest) + v2_artifact = str(Path(args.v2)) + spent: set[str] = set() + for digest, ident in zip(v2_payload["normalized_text_sha256"], v2_payload["row_ids"]): + ledger.observe( + digest, + source_artifact=v2_artifact, + row_ids=[ident], + catalogued=True, + provenance="holdout-manifest-v2 status SCORED_SPENT; hash definition is normalize_group_text", + ) + if digest not in spent: + ledger.mark_historical( + digest, + "evaluation_spent", + source_artifact=v2_artifact, + provenance="v2 holdout scored and spent", + ) + spent.add(digest) + rc1_artifact = str(Path(args.rc1_unbind_rows)) + for row in rc1_rows: + digest = ledger.observe_row( + row, + source_artifact=rc1_artifact, + provenance="rc1 scored manifest SCORED_SPENT; unbind row file matched slice n=312", + catalogued=True, + unbind_clean=normalized_text_sha256(str(row.get("text") or "")) in clean, + ) + if digest not in spent: + ledger.mark_historical( + digest, + "evaluation_spent", + source_artifact=str(Path(args.rc1_scored)), + provenance="rc1 evaluation material scored; test burned for selection", + ) + spent.add(digest) + for manifest, bundle, experiment in ( + (Path(args.select_001), select_001, "HLX-EXP-2026-09-26-SELECT-001"), + (Path(args.select_002), select_002, "HLX-EXP-2026-09-26-SELECT-002"), + ): + payload = json.loads(manifest.read_text(encoding="utf-8")) + abandoned: set[str] = set() + for digest, ident in zip(payload["normalized_text_sha256"], payload["row_ids"]): + ledger.observe( + digest, + source_artifact=str(manifest), + row_ids=[ident], + catalogued=True, + provenance="lifecycle receipt UNSCORED_ABANDONED; manifest bytes left sealed", + experiment_id=experiment, + ) + if digest not in abandoned: + ledger.mark_historical( + digest, + "evaluation_abandoned", + source_artifact=str(manifest), + provenance="holdout closed before a valid scored execution", + experiment_id=experiment, + ) + abandoned.add(digest) + _ = bundle + store_artifact = str(store_path) + for row in store_rows: + digest = normalized_text_sha256(str(row.get("text") or "")) + ledger.observe_row( + row, + source_artifact=store_artifact, + provenance="current live store; labels joined, text not stored", + current_source=True, + catalogued=True, + unbind_clean=digest in clean, + ) + + live_hashes = {normalized_text_sha256(str(row.get("text") or "")) for row in store_rows} + summary = ledger.census(live_hashes) + if summary["live_available_hashes"] != 0 or summary["currently_available_hashes"] != 0: + sys.stdout.write(json.dumps({"STOP": True, "ledger_census": summary}, indent=2) + "\n") + return 2 + screen = ledger.screen(store_rows, batch_id="live-store-rescreen") + if screen["unique_admitted_to_eval_reserve"] != 0: + raise SystemExit("STOP: live store admitted reserve identities") + + train_hashes = {normalized_text_sha256(str(row.get("text") or "")) for row in pin_rows} + gate = ledger.select_003_gate(train_hashes) + if gate["eligible"]: + raise SystemExit("STOP: SELECT-003 gate must stay closed while the reserve is empty") + + test_rows = [row for row in store_rows if str(row.get("split") or "") == "test"] + test_in_train = 0 + for row in test_rows: + digest = normalized_text_sha256(str(row.get("text") or "")) + record = ledger.identity(digest) + if record and record["training_consumed"]: + test_in_train += 1 + receipt_shas = _receipt_shas(Path(args.models)) + out = Path(args.out) + out.mkdir(parents=True, exist_ok=True) + ledger.save(out) + proof = { + "schema": "hyperlex.eval_reserve_proof.v1", + "recorded_at": datetime.now(timezone.utc).isoformat(), + "generation": GENERATION, + "policy_id": POLICY_ID, + "authorizes_training": False, + "authorizes_select_003": False, + "select_003": "NOT_DRAFTED", + "identity_function": "hyperlexical.holdout_guard.normalized_text_sha256", + "pin_sha256": pin_sha, + "pin_rows": len(pin_rows), + "store_sha256": store_sha, + "store_rows": len(store_rows), + "census_eligible_rows": report["eligible_rows"], + "census_reproduced": True, + "waterfall_sums_to_source": True, + "v2_store_joined": joined, + "v2_hash_mismatch": mismatch, + "v2_status": "SCORED_SPENT", + "rc1_scored_status": "SCORED_SPENT", + "rc1_unbind_rows": len(rc1_rows), + "rc1_classify_row_identities": "NOT_IN_MANIFEST", + "select_001_operational_status": "UNSCORED_ABANDONED", + "select_002_operational_status": "UNSCORED_ABANDONED", + "split_test_rows": len(test_rows), + "split_test_training_consumed": test_in_train, + "ledger_census": summary, + "live_store_admission_screen": screen, + "acquisition_gap": acquisition_gap(ledger.reserve_counts()), + "select_003_gate": gate, + "training_receipt_data_sha256_count": len(receipt_shas), + "pinned_export_sha_in_train_receipts": pin_sha in receipt_shas, + "training_receipt_text_extracted_only_for_bytes_read": True, + "unresolved_training_artifact_shas": [sha for sha in receipt_shas if sha != pin_sha], + "model_outcomes_consulted": False, + "generation_analysis": { + "chosen": "GEN-0", + "gen_1_created": False, + "continue_gen_0": "Acquire evaluation-only text disjoint from TRAIN_CONSUMED, EVAL_SPENT, and EVAL_ABANDONED. Keep the pinned morph78 export and seed-morph78.", + "gen_1_not_chosen_because": "convenience is not a reset condition; the baseline is still the system under test", + "gen_0_impractical_if": [ + "new classify evidence cannot represent OBSERVED and non-none", + "new text keeps colliding with the pinned export or spent/abandoned hashes", + "train-candidate growth continues while reserve slices stay empty", + "the reserve cannot hold the four required slices", + "the historical baseline is no longer the system under test", + ], + }, + } + proof_path = out / "exhaustion-proof.json" + proof_path.write_text(json.dumps(proof, indent=2, sort_keys=True) + "\n", encoding="utf-8") + # Projection must not contain raw text keys. + blob = (out / "ledger.json").read_text(encoding="utf-8") + if '"text":' in blob: + raise SystemExit("STOP: ledger projection contains a text field") + sys.stdout.write( + json.dumps( + { + "out": str(out), + "identities": summary["identities"], + "live_available": summary["live_available_hashes"], + "reserve": summary["reserve_counts"], + "screen_raw_rows": screen["raw_rows"], + "screen_unique": screen["unique_canonical_text_identities"], + "screen_reserved": screen["unique_admitted_to_eval_reserve"], + }, + indent=2, + ) + + "\n" + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/shadow/hyperlexical/clean_unbind.py b/scripts/shadow/hyperlexical/clean_unbind.py new file mode 100644 index 00000000..2dc00945 --- /dev/null +++ b/scripts/shadow/hyperlexical/clean_unbind.py @@ -0,0 +1,811 @@ +"""Clean-unbind reserve gate. + +The executed unbind row is ``export._unbind_dual_scheme_rows``: fillers are +the source atom's own tokens, positional roles are ``pos_N``, and type_slot +roles are structural ``TOKEN`` / ``SLOT`` / ``MARKER``. A gloss is not gold. + +``unbind_clean`` is ``soft_ceiling.clean_surface`` against train-split rows, +the call ``holdout_eligibility.census`` makes. That predicate is not +``normalized_text_sha256``. The hash is only the contamination identity. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from pathlib import Path +from typing import Any, Mapping, Sequence + +from .export import ( + COLLISION_HOLD, + LIVE_UNBIND_MAX_LEN, + LIVE_UNBIND_MAX_TOKENS, + TYPE_SLOT_TAGS, + reject_candidate_text, +) +from .holdout_guard import normalized_text_sha256 +from .identity_ledger import PLANNING_TARGETS, IdentityLedger, derived_state +from .soft_ceiling import clean_surface +from .unbind_settlement import ADMIT_DECISIONS + +UNBIND_CLEAN_DEFINITION = "soft_ceiling.clean_surface" +CONTRACT_SHAPE = "hyperlexical.export._unbind_dual_scheme_rows" +BLOCKED_TARGET_ORIGINS = frozenset( + { + "model", + "hyperlex_model", + "jev", + "classification_family", + "source_hint", + } +) +ALLOWED_TARGET_ORIGINS = frozenset({"source_lemma_tokens", "operator_authored"}) +CLASSIFY_SLICES = ("classify", "classify_observed", "classify_non_none") +BLOCKING_STATES = frozenset( + { + "TRAIN_CONSUMED", + "EVAL_SPENT", + "EVAL_ABANDONED", + "EVAL_RESERVE", + "EVAL_BOUND", + "TRAIN_CANDIDATE", + } +) +INDEX_FILES = ( + ("noun", "index.noun"), + ("verb", "index.verb"), + ("adj", "index.adj"), + ("adv", "index.adv"), +) + + +def refuse(message: str) -> None: + raise SystemExit(f"REFUSE: {message}") + + +def target_sha256(fillers: Sequence[str]) -> str: + """Identity of the filler list. Not a surface hash.""" + joined = " ".join(str(item).casefold() for item in fillers) + return hashlib.sha256(joined.encode("utf-8")).hexdigest() + + +def schema_example() -> dict[str, Any]: + """Synthetic placeholders. Not an acquired row.""" + return { + "positional": { + "text": "EXAMPLE_TOKEN_A EXAMPLE_TOKEN_B", + "split": "eval", + "lineage": "none", + "typology": [], + "stage": "noise", + "roles": ["pos_0", "pos_1"], + "fillers": ["EXAMPLE_TOKEN_A", "EXAMPLE_TOKEN_B"], + "role_scheme": "positional", + "task": "unbind", + "provenance": "source:EXAMPLE_LOCATOR", + "class": "INFERRED", + "license": "EXAMPLE_RIGHTS_GRANT", + "target_origin": "source_lemma_tokens", + }, + "type_slot": { + "text": "TOKEN:EXAMPLE_TOKEN_A SLOT:EXAMPLE_TOKEN_B", + "split": "eval", + "lineage": "none", + "typology": [], + "stage": "noise", + "roles": ["TOKEN", "SLOT"], + "fillers": ["EXAMPLE_TOKEN_A", "EXAMPLE_TOKEN_B"], + "role_scheme": "type_slot", + "task": "unbind", + "provenance": "source:EXAMPLE_LOCATOR", + "class": "INFERRED", + "license": "EXAMPLE_RIGHTS_GRANT", + "target_origin": "source_lemma_tokens", + }, + "scorer": { + "module": "hyperlexical.eval_forward.score_unbind_exact", + "gold": "row fillers, one string per role", + "empty_fillers": "skipped", + "metrics": [ + "unbind_exact", + "unbind_token_f1", + "unbind_slot_f1", + "unbind_exact_strict", + "unbind_token_f1_strict", + "unbind_slot_f1_strict", + ], + "by_role_scheme": ["positional", "type_slot"], + "baseline": "unbind_copy_token", + "clean_slice": UNBIND_CLEAN_DEFINITION, + "this_module_scores": False, + }, + } + + +def structural_tags(n: int) -> list[str]: + return [TYPE_SLOT_TAGS[i % len(TYPE_SLOT_TAGS)] for i in range(n)] + + +def dual_scheme_rows( + tokens: Sequence[str], + *, + license: str, + provenance: str, + target_origin: str, + source_pos: str = "", + lineage: str = "none", + stage: str = "noise", + epistemic: str = "INFERRED", +) -> list[dict[str, Any]]: + """Same text/role/filler shape as ``_unbind_dual_scheme_rows``.""" + items = [str(tok) for tok in tokens] + atom = " ".join(items) + tags = structural_tags(len(items)) + common = { + "split": "eval", + "lineage": lineage, + "typology": [], + "stage": stage, + "task": "unbind", + "provenance": provenance, + "class": epistemic, + "license": license, + "target_origin": target_origin, + "source_pos": source_pos, + } + positional = { + **common, + "text": atom, + "roles": [f"pos_{k}" for k in range(len(items))], + "fillers": list(items), + "role_scheme": "positional", + } + typed = { + **common, + "text": " ".join(f"{tag}:{tok}" for tag, tok in zip(tags, items)), + "roles": tags, + "fillers": list(items), + "role_scheme": "type_slot", + } + return [positional, typed] + + +def structural_reason(row: Mapping[str, Any]) -> str | None: + if str(row.get("task") or "") != "unbind": + return "not_unbind" + scheme = row.get("role_scheme") + if scheme not in {"positional", "type_slot"}: + return "role_scheme" + fillers = row.get("fillers") + if not isinstance(fillers, list) or not fillers: + return "missing_target" + if not all(isinstance(item, str) and item.strip() for item in fillers): + return "missing_target" + if not (2 <= len(fillers) <= LIVE_UNBIND_MAX_TOKENS): + return "token_count" + atom = " ".join(str(item) for item in fillers) + if len(atom) > LIVE_UNBIND_MAX_LEN: + return "atom_length" + if atom.lower() in COLLISION_HOLD: + return "collision_hold" + junk = reject_candidate_text(atom) + if junk: + return f"candidate_{junk}" + roles = row.get("roles") + if not isinstance(roles, list) or len(roles) != len(fillers): + return "role_alignment" + text = str(row.get("text") or "") + if scheme == "positional": + if text.split() != list(fillers): + return "surface_filler_mismatch" + if list(roles) != [f"pos_{k}" for k in range(len(fillers))]: + return "role_alignment" + else: + tags = structural_tags(len(fillers)) + expected = " ".join(f"{tag}:{tok}" for tag, tok in zip(tags, fillers)) + if text != expected or list(roles) != tags: + return "surface_filler_mismatch" + if str(row.get("class") or "") not in {"OBSERVED", "INFERRED"}: + return "class" + if "lineage" not in row: + return "lineage" + return None + + +def rights_reason(row: Mapping[str, Any]) -> str | None: + license_text = row.get("license") + if not isinstance(license_text, str) or not license_text.strip(): + return "rights_unresolved" + lowered = license_text.lower() + if "rights_unresolved" in lowered or "not cc" in lowered: + return "rights_unresolved" + return None + + +def target_reason(row: Mapping[str, Any]) -> str | None: + origin = str(row.get("target_origin") or "").strip() + if not origin: + return "unbind_unresolved" + if origin in BLOCKED_TARGET_ORIGINS: + return "model_derived_target" + if origin not in ALLOWED_TARGET_ORIGINS: + return "unbind_unresolved" + provenance = str(row.get("provenance") or "").lower() + if any(mark in provenance for mark in ("jev", "model-derived", "hyperlex-encoder", "source_hint_only")): + return "model_derived_target" + return None + + +def unbind_clean_hashes( + rows: Sequence[Mapping[str, Any]], + train_rows: Sequence[Mapping[str, Any]], +) -> tuple[set[str], dict[str, Any]]: + """Canonical clean predicate. Train split only. No trained-key extra.""" + train_for_clean = [row for row in train_rows if row.get("split") == "train"] + unbind = [row for row in rows if row.get("task") == "unbind"] + kept, account = clean_surface(unbind, train_rows=train_for_clean) + hashes = {normalized_text_sha256(str(row.get("text") or "")) for row in kept} + account = dict(account) + account["definition"] = UNBIND_CLEAN_DEFINITION + account["train_split_rows"] = len(train_for_clean) + return hashes, account + + +def _reject(bucket: list[dict[str, str]], digest: str, reason: str) -> None: + bucket.append({"normalized_text_sha256": digest, "reason": reason}) + + +def gate_rows( + rows: Sequence[Mapping[str, Any]], + *, + train_rows: Sequence[Mapping[str, Any]], + ledger: IdentityLedger, + settlements: Mapping[str, Mapping[str, Any]] | None = None, + require_settlement: bool = True, +) -> tuple[list[dict[str, Any]], list[dict[str, str]], dict[str, Any]]: + """Return admissible rows, hash rejections, and clean accounting. + + One rejection per canonical text hash. The first failing check wins. + """ + rejections: list[dict[str, str]] = [] + survivors: list[dict[str, Any]] = [] + seen: set[str] = set() + for row in rows: + digest = normalized_text_sha256(str(row.get("text") or "")) + if digest in seen: + continue + seen.add(digest) + reason = structural_reason(row) or target_reason(row) or rights_reason(row) + if reason: + _reject(rejections, digest, reason) + continue + record = ledger.identities.get(digest) + state = derived_state(record) if record else "NEW" + if state in BLOCKING_STATES: + _reject(rejections, digest, state) + continue + if state not in {"NEW", "AVAILABLE"}: + _reject(rejections, digest, state) + continue + survivors.append(dict(row)) + clean_hashes, account = unbind_clean_hashes(survivors, train_rows) + admissible: list[dict[str, Any]] = [] + for row in survivors: + digest = normalized_text_sha256(str(row.get("text") or "")) + if digest not in clean_hashes: + _reject(rejections, digest, "not_clean") + continue + if not require_settlement: + admissible.append(row) + continue + settled = settlements or {} + settlement = settled.get(digest) + if settlement is None: + _reject(rejections, digest, "unsettled") + continue + decision = str(settlement.get("decision") or "") + if decision == "REJECT": + _reject(rejections, digest, "settlement_reject") + continue + if decision == "UNRESOLVED": + _reject(rejections, digest, "settlement_unresolved") + continue + if decision not in ADMIT_DECISIONS: + _reject(rejections, digest, "unsettled") + continue + expected = target_sha256(row.get("fillers") or []) + if str(settlement.get("target_sha256") or "") != expected: + _reject(rejections, digest, "target_mismatch") + continue + if str(settlement.get("target_provenance") or "") != str(row.get("target_origin") or ""): + _reject(rejections, digest, "target_mismatch") + continue + admissible.append(row) + return admissible, rejections, account + + +def select_diverse(rows: Sequence[Mapping[str, Any]], limit: int) -> list[dict[str, Any]]: + """Round-robin atoms by filler count and source pos. Both schemes stay paired.""" + if limit <= 0: + return [] + groups: dict[str, list[dict[str, Any]]] = {} + for row in rows: + groups.setdefault(target_sha256(list(row.get("fillers") or [])), []).append(dict(row)) + buckets: dict[tuple[int, str], list[list[dict[str, Any]]]] = {} + for group in groups.values(): + n_fillers = len(group[0].get("fillers") or []) + source_pos = str(group[0].get("source_pos") or "") + buckets.setdefault((n_fillers, source_pos), []).append(group) + for group_list in buckets.values(): + group_list.sort(key=lambda group: normalized_text_sha256(str(group[0].get("text") or ""))) + order = sorted(buckets) + indexes = {key: 0 for key in order} + picked: list[dict[str, Any]] = [] + while len(picked) < limit: + progressed = False + for key in order: + index = indexes[key] + if index >= len(buckets[key]): + continue + group = buckets[key][index] + indexes[key] = index + 1 + ordered = sorted( + group, + key=lambda row: (row.get("role_scheme") != "positional", str(row.get("role_scheme") or "")), + ) + room = limit - len(picked) + if len(ordered) <= room: + picked.extend(ordered) + progressed = True + else: + positional = [row for row in ordered if row.get("role_scheme") == "positional"] + if positional and len(positional) <= room: + picked.extend(positional) + progressed = True + if len(picked) >= limit: + break + if not progressed: + break + return picked + + +def diversity_report(rows: Sequence[Mapping[str, Any]]) -> dict[str, Any]: + """Dimensions the unbind row already has. No new taxonomy.""" + schemes: Counter[str] = Counter() + filler_counts: Counter[str] = Counter() + sources: Counter[str] = Counter() + classes: Counter[str] = Counter() + lineages: Counter[str] = Counter() + surfaces_per_target: Counter[str] = Counter() + for row in rows: + schemes[str(row.get("role_scheme") or "")] += 1 + filler_counts[str(len(row.get("fillers") or []))] += 1 + sources[str(row.get("source_pos") or "unspecified")] += 1 + classes[str(row.get("class") or "")] += 1 + lineages[str(row.get("lineage") or "")] += 1 + surfaces_per_target[target_sha256(list(row.get("fillers") or []))] += 1 + histogram = Counter(surfaces_per_target.values()) + return { + "rows": len(rows), + "role_scheme": dict(schemes), + "filler_count": dict(sorted(filler_counts.items(), key=lambda item: int(item[0]) if item[0].isdigit() else 0)), + "source_pos_metadata": dict(sources), + "source_pos_metadata_note": "WordNet index file. Not a Hyperlex family.", + "class": dict(classes), + "lineage": dict(lineages), + "unique_targets": len(surfaces_per_target), + "max_surfaces_per_target": max(surfaces_per_target.values()) if surfaces_per_target else 0, + "surfaces_per_target_histogram": {str(size): histogram[size] for size in sorted(histogram)}, + } + + +def assert_wordnet_license(text: str) -> None: + if "Permission to use, copy, modify and distribute" not in text: + refuse("WordNet license grant is not in the license file") + if "Copyright 2006 by Princeton University" not in text: + refuse("WordNet copyright notice is not in the license file") + + +def read_wordnet_index(index_dir: str | Path) -> list[dict[str, Any]]: + """Multiword index lemmas only. Does not read glosses in ``data.*``.""" + root = Path(index_dir) + atoms: list[dict[str, Any]] = [] + seen_surface: set[tuple[str, str]] = set() + for source_pos, name in INDEX_FILES: + path = root / name + if not path.is_file(): + refuse(f"WordNet index missing: {name}") + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line or line[0] == " ": + continue + lemma = line.split(" ", 1)[0] + if "_" not in lemma: + continue + tokens = [part for part in lemma.split("_") if part] + if len(tokens) != lemma.count("_") + 1: + continue + surface = " ".join(tokens) + if not (2 <= len(tokens) <= LIVE_UNBIND_MAX_TOKENS): + continue + if len(surface) > LIVE_UNBIND_MAX_LEN: + continue + if surface.lower() in COLLISION_HOLD: + continue + if reject_candidate_text(surface): + continue + key = (source_pos, surface.casefold()) + if key in seen_surface: + continue + seen_surface.add(key) + atoms.append( + { + "source_pos": source_pos, + "tokens": tokens, + "surface": surface, + "lemma_sha256": hashlib.sha256(lemma.encode("utf-8")).hexdigest(), + } + ) + return atoms + + +def rows_from_wordnet_atoms( + atoms: Sequence[Mapping[str, Any]], + *, + license: str, +) -> list[dict[str, Any]]: + rows: list[dict[str, Any]] = [] + for atom in atoms: + tokens = list(atom["tokens"]) + provenance = ( + "wordnet-3.0:index." + + str(atom.get("source_pos") or "") + + ":lemma:" + + str(atom.get("lemma_sha256") or "") + ) + rows.extend( + dual_scheme_rows( + tokens, + license=license, + provenance=provenance, + target_origin="source_lemma_tokens", + source_pos=str(atom.get("source_pos") or ""), + ) + ) + return rows + + +def reason_counts(rejections: Sequence[Mapping[str, str]]) -> dict[str, int]: + counts: Counter[str] = Counter(item["reason"] for item in rejections) + return dict(sorted(counts.items())) + + +def admit_clean_unbind( + ledger: IdentityLedger, + rows: Sequence[Mapping[str, Any]], + *, + batch_id: str, + source_artifact: str, + train_rows: Sequence[Mapping[str, Any]], + settlements: Mapping[str, Mapping[str, Any]], +) -> dict[str, Any]: + """Admit only gated clean-unbind rows. Does not persist.""" + before = ledger.reserve_counts() + admissible, rejections, clean_account = gate_rows( + rows, + train_rows=train_rows, + ledger=ledger, + settlements=settlements, + ) + room = max(int(PLANNING_TARGETS["unbind_clean"]) - int(before.get("unbind_clean", 0)), 0) + if len(admissible) <= room: + selected = list(admissible) + else: + selected = select_diverse(admissible, room) + clean_hashes = {normalized_text_sha256(str(row.get("text") or "")) for row in selected} + report = ledger.admit( + selected, + batch_id=batch_id, + source_artifact=source_artifact, + unbind_clean_hashes=clean_hashes, + ) + after = ledger.reserve_counts() + for key in CLASSIFY_SLICES: + if int(after.get(key, 0)) < int(before.get(key, 0)): + refuse(f"STOP: {key} decreased") + if int(after.get("unbind_clean", 0)) < int(before.get("unbind_clean", 0)): + refuse("STOP: unbind_clean decreased") + if report["unique_routed_to_train_candidate"]: + refuse("clean-unbind admission routed rows to TRAIN_CANDIDATE") + report = dict(report) + report["reserve_counts_before"] = before + report["reserve_counts_after"] = after + report["rejection_counts"] = reason_counts(rejections) + report["rejected_hashes"] = len(rejections) + report["admissible_before_cap"] = len(admissible) + report["selected_rows"] = len(selected) + report["clean_account"] = clean_account + report["diversity"] = diversity_report(selected) + report["unbind_clean_definition"] = UNBIND_CLEAN_DEFINITION + report["contract_shape"] = CONTRACT_SHAPE + report["vendor_calls"] = 0 + return report + + +def wordnet_license_text(index_dir: str | Path) -> str: + path = Path(index_dir) / "LICENSE" + if not path.is_file(): + path = Path(index_dir).parent / "LICENSE" + if not path.is_file(): + refuse("WordNet LICENSE file is missing") + text = path.read_text(encoding="utf-8", errors="replace") + assert_wordnet_license(text) + return text + + +WORDNET_LICENSE = ( + "WordNet 3.0 Copyright 2006 by Princeton University. " + "Permission to use, copy, modify and distribute this software and " + "database and its documentation for any purpose and without fee or " + "royalty is granted, provided the copyright notice and statements " + "appear on all copies." +) + + +def file_sha256(path: str | Path) -> str: + digest = hashlib.sha256() + with Path(path).open("rb") as handle: + for chunk in iter(lambda: handle.read(1 << 20), b""): + digest.update(chunk) + return digest.hexdigest() + + +def load_jsonl(path: Path) -> list[dict[str, Any]]: + rows: list[dict[str, Any]] = [] + for index, line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1): + if not line.strip(): + continue + try: + obj = json.loads(line) + except json.JSONDecodeError: + refuse(f"{path} line {index} is not JSON") + if not isinstance(obj, dict): + refuse(f"{path} line {index} must be an object") + rows.append(obj) + return rows + + +def _walk_forbid(value: Any) -> None: + if isinstance(value, dict): + for key, item in value.items(): + if key in {"text", "normalized_text", "raw_text", "surface", "fillers"}: + refuse(f"receipt must not carry {key}") + _walk_forbid(item) + elif isinstance(value, list): + for item in value: + _walk_forbid(item) + + +def planning_progress(counts: Mapping[str, int]) -> list[dict[str, Any]]: + rows = [] + for key, target in PLANNING_TARGETS.items(): + have = int(counts.get(key, 0)) + rows.append( + { + "slice": key, + "current": have, + "planning_target": target, + "planning_target_class": "PLANNING_TARGET", + "progress": have / target if target else 0, + "gap_vs_planning_target": max(target - have, 0), + "minimum_required": "NOT_COMPUTABLE", + } + ) + return rows + + +def admission_result_label(before: Mapping[str, int], after: Mapping[str, int], gate: Mapping[str, Any]) -> str: + for key in CLASSIFY_SLICES: + if int(after.get(key, 0)) < int(before.get(key, 0)): + refuse(f"STOP: {key} decreased") + if int(after.get("unbind_clean", 0)) <= int(before.get("unbind_clean", 0)): + return "ACQUISITION_BLOCKED" + if gate.get("eligible") is True: + return "SELECT_003_ELIGIBLE" + if int(after.get("unbind_clean", 0)) > 0: + return "CLEAN_UNBIND_PARTIALLY_FILLED" + return "ACQUISITION_BLOCKED" + + +ZERO_SUPPORT_FAMILIES = ( + "relationship-dating", + "conflict-aggression", + "sports-competition", + "fashion-aesthetic", + "regional-cultural", + "spiritual-mystic", +) + + +def main(argv: Sequence[str] | None = None) -> int: + """Acquire clean-unbind reserve rows. Does not train or score.""" + import argparse + import sys + from datetime import datetime, timezone + + from .unbind_settlement import append_events, event_from_decision, latest_by_hash, load_events + + parser = argparse.ArgumentParser(description="Hyperlex clean-unbind reserve acquisition") + parser.add_argument("--ledger", required=True) + parser.add_argument("--train-export", required=True) + parser.add_argument("--index-dir", required=True) + parser.add_argument("--batch-id", required=True) + parser.add_argument("--source-artifact", required=True) + parser.add_argument("--settlement-log", required=True) + parser.add_argument("--operator", required=True) + parser.add_argument("--rows-out", required=True) + parser.add_argument("--receipt", required=True) + args = parser.parse_args(list(argv) if argv is not None else None) + if Path(args.receipt).exists(): + refuse(f"receipt already exists: {args.receipt}") + if Path(args.rows_out).exists(): + refuse(f"rows output already exists: {args.rows_out}") + + index_dir = Path(args.index_dir) + license_body = wordnet_license_text(index_dir) + license_path = index_dir / "LICENSE" + if not license_path.is_file(): + license_path = index_dir.parent / "LICENSE" + atoms = read_wordnet_index(index_dir) + rows = rows_from_wordnet_atoms(atoms, license=WORDNET_LICENSE) + ledger = IdentityLedger.load(args.ledger) + before = ledger.reserve_counts() + train_rows = load_jsonl(Path(args.train_export)) + admissible, rejections, clean_account = gate_rows( + rows, + train_rows=train_rows, + ledger=ledger, + require_settlement=False, + ) + room = int(PLANNING_TARGETS["unbind_clean"]) - int(before.get("unbind_clean", 0)) + selected = select_diverse(admissible, max(room, 0)) + settled_at = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + events = [] + for row in selected: + events.append( + event_from_decision( + normalized_text_sha256=normalized_text_sha256(str(row.get("text") or "")), + decision="ACCEPT", + target_sha256=target_sha256(list(row.get("fillers") or [])), + target_provenance=str(row.get("target_origin") or ""), + operator=args.operator, + settled_at=settled_at, + decision_basis="source_lemma_token_identity", + ) + ) + append_events(args.settlement_log, events) + settlements = latest_by_hash(load_events(args.settlement_log)) + prior_events = len(ledger.events) + report = admit_clean_unbind( + ledger, + selected, + batch_id=args.batch_id, + source_artifact=args.source_artifact, + train_rows=train_rows, + settlements=settlements, + ) + after = ledger.reserve_counts() + train_hashes = { + normalized_text_sha256(str(row.get("text") or "")) + for row in train_rows + } + gate = ledger.select_003_gate(train_hashes) + label = admission_result_label(before, after, gate) + events_path = Path(args.ledger) / "events.jsonl" + events_sha256_before = file_sha256(events_path) + written = ledger.persist_append(args.ledger, prior_events) + events_sha256_after = file_sha256(events_path) + Path(args.settlement_log).chmod(0o600) + rows_out = Path(args.rows_out) + rows_out.parent.mkdir(parents=True, exist_ok=True) + rows_out.write_text( + "".join(json.dumps(row, sort_keys=True) + "\n" for row in selected), + encoding="utf-8", + ) + rows_out.chmod(0o600) + receipt = { + "schema": "hyperlex.eval_unbind_acquisition.v1", + "batch_id": args.batch_id, + "source": "Princeton WordNet 3.0 index lemmas", + "source_artifact": args.source_artifact, + "source_rights": WORDNET_LICENSE, + "license_file_sha256": file_sha256(license_path), + "raw_artifact_sha256": file_sha256(args.source_artifact), + "license_notice_present": "Copyright 2006 by Princeton University" in license_body, + "schema_contract": CONTRACT_SHAPE, + "unbind_clean_definition": UNBIND_CLEAN_DEFINITION, + "timestamp": settled_at, + "collector": args.operator, + "transformation_pipeline": [ + "read index.noun/verb/adj/adv lemmas only", + "do not read data.* glosses", + "underscore to space", + "keep 2..6 tokens and atom length <= 80", + "emit positional and type_slot from export._unbind_dual_scheme_rows shape", + "fillers are the source lemma tokens", + "novelty screen via normalized_text_sha256", + "clean_surface against train split", + "ACCEPT settlement where fillers equal the lemma tokens", + "IdentityLedger.admit with derived unbind_clean hashes", + ], + "index_mwe_atoms": len(atoms), + "dual_scheme_rows": len(rows), + "screen_rejection_counts": reason_counts(rejections), + "screen_clean_account": { + key: clean_account.get(key) + for key in ( + "n_input", + "n_kept", + "n_excluded_trained_key", + "n_excluded_trained_text", + "n_excluded_train_split_text", + "definition", + "train_split_rows", + ) + }, + "admissible_before_planning_cap": len(admissible), + "selected_rows": len(selected), + "settlement": { + "schema": "hyperlex.eval_unbind_settlement.v1", + "log": args.settlement_log, + "events_appended": len(events), + "decisions": {"ACCEPT": len(events), "CORRECT_TARGET": 0, "REJECT": 0, "UNRESOLVED": 0}, + "decision_basis": "source_lemma_token_identity", + }, + "admission": { + "events_sha256_before": events_sha256_before, + "events_sha256_after": events_sha256_after, + "events_appended": written, + "unique_admitted_to_eval_reserve": report["unique_admitted_to_eval_reserve"], + "unique_routed_to_train_candidate": report["unique_routed_to_train_candidate"], + "rejected_existing_identities": report["rejected_existing_identities"], + "model_outcomes_consulted": report["model_outcomes_consulted"], + }, + "diversity": report["diversity"], + "reserve_counts_before": before, + "reserve_counts_after": after, + "planning_progress": planning_progress(after), + "select_003_gate": gate, + "admission_result": label, + "select_003": "NOT_DRAFTED", + "vendor_calls": 0, + "zero_support_families_not_collected": list(ZERO_SUPPORT_FAMILIES), + "evaluation_quality_note": ( + "Representation completeness is separate from evaluation quality. " + "This batch is one lexicon, class INFERRED, lineage none." + ), + } + _walk_forbid(receipt) + receipt_path = Path(args.receipt) + receipt_path.write_text(json.dumps(receipt, indent=2, sort_keys=True) + "\n", encoding="utf-8") + receipt_path.chmod(0o600) + public = { + "admission_result": label, + "reserve_counts_before": before, + "reserve_counts_after": after, + "selected_rows": len(selected), + "admissible_before_planning_cap": len(admissible), + "events_appended": written, + "vendor_calls": 0, + "select_003": "NOT_DRAFTED", + "gate_eligible": gate.get("eligible"), + "diversity": report["diversity"], + "screen_rejection_counts": reason_counts(rejections), + } + _walk_forbid(public) + sys.stdout.write(json.dumps(public, indent=2, sort_keys=True) + "\n") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/shadow/hyperlexical/eval_settlement.py b/scripts/shadow/hyperlexical/eval_settlement.py new file mode 100644 index 00000000..c0c3dcbe --- /dev/null +++ b/scripts/shadow/hyperlexical/eval_settlement.py @@ -0,0 +1,845 @@ +"""Evaluation-settlement apply path for the eighteen-family taxonomy. + +``source_hint``, ``semantic_family``, and ``attest`` are separate fields. +A confirm settlement may store the same family string in both ``source_hint`` +and ``semantic_family``. This module never copies one field into another, +and it never derives ``attest`` from ``decision``. + +The production classify head is not read or written here. ``ACCEPT`` records +the operator-selected attest. It does not promote existing evidence to +``OBSERVED``. This command does not admit rows to ``EVAL_RESERVE``. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from pathlib import Path +from typing import Any, Mapping, Sequence + +from .holdout_guard import normalized_text_sha256 + +SCHEMA = "hyperlex.eval_settlement.v1" +RECEIPT_SCHEMA = "hyperlex.eval_settlement_receipt.v1" +VENDOR_CALLS = 0 + +ACTIVE_FAMILIES: tuple[str, ...] = ( + "gaming-meta", + "betting-sharp", + "crypto-degen", + "internet-slang", + "memetic", + "social-status", + "relationship-dating", + "approval-disapproval", + "conflict-aggression", + "technology-ai", + "workplace-career", + "sports-competition", + "music-entertainment", + "fashion-aesthetic", + "regional-cultural", + "spiritual-mystic", + "identity-affiliation", + "politics-civic", +) +CANDIDATE_FAMILIES: tuple[str, ...] = ( + "finance-retail", + "market-structure", + "sexual-romantic", + "substance-party", + "crime-illicit", + "health-fitness", +) +ABSTAIN = "none" +DECISIONS = ("ACCEPT", "RECLASSIFY", "NONE", "UNRESOLVED") +DECISION_ALIASES = { + "ACCEPT": "ACCEPT", + "ACCEPT FAMILY": "ACCEPT", + "RECLASSIFY": "RECLASSIFY", + "CHOOSE DIFFERENT FAMILY": "RECLASSIFY", + "NONE": "NONE", + "UNRESOLVED": "UNRESOLVED", +} +ATTESTS = ("OBSERVED", "INFERRED") +REGISTERS = ("slang", "domain-specific", "high-register", "general") +FUNCTIONS = ("address", "evaluation", "intensification", "reference", "affiliation") +SURFACES = { + "A": "CONFIRM", + "B": "PROPOSED FAMILY", + "C": "DISAMBIGUATE", + "D": "RIGHTS BLOCKED", + "HELD_OUTSIDE_LANES": "hint-only", +} +SHEET_COLUMNS = ( + "lane", + "row_key", + "text", + "source_hint", + "source_hint_provenance", + "current_label", + "label_source", + "rights", + "proposed_family_evidence", + "disambiguation_options", + "semantic_family", + "attest", + "register", + "function", + "decision", +) +SETTLED_DECISIONS = frozenset({"ACCEPT", "RECLASSIFY", "NONE"}) +_CLEARED_SOURCES = { + "wiktionary_category": "CC-BY-SA", + "wikipedia_prose": "CC-BY-SA", +} +_BLOCKED_SOURCES = { + "reddit_title": "RIGHTS_UNRESOLVED", + "kym_slang_list": "RIGHTS_UNRESOLVED", +} +_SCORE_KEYS = frozenset( + {"candidate_score", "model_error", "model_score", "prediction"} +) +_FORBIDDEN_KEYS = frozenset({"text", "normalized_text", "raw_text"}) | _SCORE_KEYS +_HINT_PROVENANCE = "stream.family_hint_not_a_label" + + +def refuse(message: str) -> None: + raise SystemExit(f"REFUSE: {message}") + + +def family_flags(name: str, *, activated: set[str] | None = None) -> dict[str, bool]: + """Three independent flags. Evaluation settlement does not enable a family.""" + activated = set(activated or ()) + if name in ACTIVE_FAMILIES: + return { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + if name in CANDIDATE_FAMILIES: + return { + "taxonomy.active": name in activated, + "evaluation.enabled": False, + "production.enabled": False, + } + refuse(f"family is not in the evaluation taxonomy: {name}") + raise AssertionError("refuse") + + +def _activated(activated: set[str] | None) -> set[str]: + extra = set(activated or ()) + unknown = sorted(extra - set(CANDIDATE_FAMILIES)) + if unknown: + refuse(f"activation refused for {unknown[0]}") + return extra + + +def _blank(value: Any) -> bool: + return value is None or (isinstance(value, str) and not value.strip()) + + +def _optional_text(value: Any) -> str | None: + if _blank(value): + return None + if not isinstance(value, str): + refuse("settlement field is not text") + return value.strip() + + +def _decision(value: Any) -> str: + token = _optional_text(value) + if token is None: + refuse("decision is empty") + mapped = DECISION_ALIASES.get(token.upper()) + if mapped is None: + if token.upper() == "REJECT": + refuse("reject is a production attest token, not an evaluation decision") + refuse(f"invalid decision: {token}") + return mapped + + +def _attest(value: Any) -> str | None: + token = _optional_text(value) + if token is None: + return None + mapped = token.upper() + if mapped not in ATTESTS: + refuse(f"invalid attest: {token}") + return mapped + + +def _family_token(value: Any) -> str | None: + token = _optional_text(value) + if token is None: + return None + return token.casefold() + + +def _closed_vocab(value: Any, allowed: tuple[str, ...], label: str) -> str | None: + token = _optional_text(value) + if token is None: + return None + mapped = token.casefold() + if mapped not in allowed: + refuse(f"invalid {label}: {token}") + return mapped + + +def _resolve_family(token: str | None, *, activated: set[str], allow_null: bool) -> str | None: + if token is None: + if allow_null: + return None + refuse("explicit semantic_family is required") + if token == ABSTAIN: + return ABSTAIN + if token in CANDIDATE_FAMILIES and token not in activated: + refuse(f"candidate family is inactive: {token}") + if token in ACTIVE_FAMILIES or token in activated: + return token + refuse(f"family is not in the evaluation taxonomy: {token}") + raise AssertionError("refuse") + + +def rights_of(stream_row: Mapping[str, Any]) -> str: + source_type = str(stream_row.get("source_type") or "") + if source_type in _CLEARED_SOURCES: + return _CLEARED_SOURCES[source_type] + if source_type in _BLOCKED_SOURCES: + return _BLOCKED_SOURCES[source_type] + refuse(f"unknown rights source_type: {source_type or 'missing'}") + raise AssertionError("refuse") + + +def evidence_hint(stream_row: Mapping[str, Any]) -> str | None: + hint = stream_row.get("family_hint_not_a_label") + if _blank(hint): + return None + return str(hint) + + +def sheet_hint_token(hint: str | None) -> str: + return "NO_HINT" if hint is None else hint + + +def sheet_label_token(label: Any) -> str: + if label is None: + return "UNLABELLED" + return str(label) + + +def settlement_allows_reserve(record: Mapping[str, Any]) -> bool: + """Rights and decision gates. This predicate does not admit a row.""" + if record.get("decision") not in SETTLED_DECISIONS: + return False + if record.get("rights") == "RIGHTS_UNRESOLVED": + return False + family = record.get("semantic_family") + attest = record.get("attest") + if family is None or attest not in ATTESTS: + return False + if attest == "OBSERVED" and record.get("decision") is None: + return False + return True + + +def _forbid_payload(value: Mapping[str, Any], where: str) -> None: + for key in value: + if key in _FORBIDDEN_KEYS: + refuse(f"{where} must not carry {key}") + + +def _refuse_scores(value: Mapping[str, Any], where: str) -> None: + for key in value: + if key in _SCORE_KEYS: + refuse(f"{where} must not carry {key}") + + +def _canonical_hash(stream_row: Mapping[str, Any]) -> str: + digest = normalized_text_sha256(str(stream_row.get("text") or "")) + stored = stream_row.get("normtext_sha256") + if isinstance(stored, str) and stored and stored != digest: + refuse("stream normtext_sha256 drifted from canonical row identity") + return digest + + +def _coerce_choice( + decision: str, + family: str | None, + attest: str | None, + proposed: str | None, + hint: str | None, +) -> tuple[str | None, str | None]: + if decision == "UNRESOLVED": + if family is not None or attest is not None: + refuse("UNRESOLVED requires null semantic_family and null attest") + return None, None + if decision == "NONE": + if family not in (None, ABSTAIN): + refuse("NONE requires semantic_family none") + if attest is None: + refuse("NONE requires an explicit attest") + return ABSTAIN, attest + if decision == "ACCEPT": + if family is None: + refuse("ACCEPT requires an explicit semantic_family") + if attest is None: + refuse("ACCEPT requires an explicit attest") + if proposed is not None and family != proposed: + refuse("ACCEPT does not match the proposed family") + return family, attest + if decision == "RECLASSIFY": + if family is None: + refuse("RECLASSIFY requires an explicit family") + if attest is None: + refuse("RECLASSIFY requires an explicit attest") + evidence = proposed if proposed is not None else hint + if evidence is not None and family == evidence: + refuse("RECLASSIFY family must differ from the proposed evidence family") + return family, attest + refuse(f"invalid decision: {decision}") + raise AssertionError("refuse") + + +def validate_settlement( + incoming: Mapping[str, Any], + stream_row: Mapping[str, Any], + *, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> dict[str, Any]: + """Fail closed. Does not read ``label_source`` as ``attest``.""" + _forbid_payload(incoming, "settlement") + extra = _activated(activated) + row_id = _optional_text(incoming.get("row_id")) or _optional_text(stream_row.get("row_key")) + if not row_id or row_id != stream_row.get("row_key"): + refuse("row_id is not in the held-out stream") + if row_id in set(prior_ids or ()): + refuse(f"duplicate settlement for {row_id}") + digest = _canonical_hash(stream_row) + claimed = _optional_text(incoming.get("text_hash")) + if claimed is None or claimed != digest: + refuse(f"text_hash does not match canonical row identity for {row_id}") + hint = evidence_hint(stream_row) + supplied_hint = incoming.get("source_hint") + if _blank(supplied_hint): + supplied_hint = None + elif isinstance(supplied_hint, str) and supplied_hint.strip() == "NO_HINT": + supplied_hint = None + elif isinstance(supplied_hint, str): + supplied_hint = supplied_hint.strip() + else: + refuse("source_hint is not text") + if supplied_hint != hint: + refuse(f"source_hint does not match held-out evidence for {row_id}") + rights = rights_of(stream_row) + sheet_rights = _optional_text(incoming.get("rights")) + if sheet_rights is not None and sheet_rights != rights: + refuse(f"rights state does not match held-out evidence for {row_id}") + decision = _decision(incoming.get("decision")) + proposed = _family_token(incoming.get("proposed_family")) + register = _closed_vocab(incoming.get("register"), REGISTERS, "register") + function = _closed_vocab(incoming.get("function"), FUNCTIONS, "function") + attest = _attest(incoming.get("attest")) + family = _family_token(incoming.get("semantic_family")) + if decision == "UNRESOLVED" and (register is not None or function is not None): + refuse("UNRESOLVED cannot carry register or function") + family, attest = _coerce_choice(decision, family, attest, proposed, hint) + family = _resolve_family(family, activated=extra, allow_null=decision == "UNRESOLVED") + operator = _optional_text(incoming.get("operator")) + settled_at = _optional_text(incoming.get("settled_at")) + provenance = _optional_text(incoming.get("provenance")) + if not operator: + refuse("operator is required") + if not settled_at: + refuse("settled_at is required") + if not provenance: + refuse("provenance is required") + lane = _optional_text(incoming.get("lane")) + if lane is not None and lane not in SURFACES: + refuse(f"unknown settlement surface: {lane}") + record = { + "schema": SCHEMA, + "row_id": row_id, + "text_hash": digest, + "decision": decision, + "semantic_family": family, + "attest": attest, + "register": register, + "function": function, + "source_hint": hint, + "operator": operator, + "settled_at": settled_at, + "provenance": provenance, + "rights": rights, + "lane": lane, + } + _forbid_payload(record, "settlement record") + return record + + +def _stream_index(rows: Sequence[Mapping[str, Any]]) -> dict[str, Mapping[str, Any]]: + index: dict[str, Mapping[str, Any]] = {} + for row in rows: + _refuse_scores(row, "held-out row") + key = row.get("row_key") + if not isinstance(key, str) or not key: + refuse("held-out row is missing row_key") + if key in index: + refuse(f"duplicate held-out row_id {key}") + index[key] = row + return index + + +def load_stream(path: str | Path) -> dict[str, Mapping[str, Any]]: + rows = [] + with Path(path).open(encoding="utf-8") as handle: + for line in handle: + if line.strip(): + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("held-out stream row is not an object") + rows.append(payload) + return _stream_index(rows) + + +def _check_sheet_integrity( + parts: Sequence[str], + header: Sequence[str], + stream_row: Mapping[str, Any], +) -> str: + cell = dict(zip(header, parts)) + row_id = cell["row_key"].strip() + if row_id != stream_row.get("row_key"): + refuse("row_id is not in the held-out stream") + digest = normalized_text_sha256(cell["text"]) + if digest != _canonical_hash(stream_row): + refuse(f"text_hash does not match canonical row identity for {row_id}") + hint = evidence_hint(stream_row) + if cell["source_hint"] != sheet_hint_token(hint): + refuse(f"source_hint does not match held-out evidence for {row_id}") + if cell["source_hint_provenance"] != _HINT_PROVENANCE: + refuse(f"source_hint provenance does not match for {row_id}") + if cell["current_label"] != sheet_label_token(stream_row.get("label")): + refuse(f"current_label does not match held-out evidence for {row_id}") + if cell["label_source"] != str(stream_row.get("label_source") or ""): + refuse(f"label_source does not match held-out evidence for {row_id}") + if cell["rights"].strip() != rights_of(stream_row): + refuse(f"rights state does not match held-out evidence for {row_id}") + lane = cell["lane"].strip() + if lane not in SURFACES: + refuse(f"unknown settlement surface: {lane}") + return digest + + +def _operator_cells(cell: Mapping[str, str]) -> dict[str, str]: + return { + "semantic_family": cell["semantic_family"], + "attest": cell["attest"], + "register": cell["register"], + "function": cell["function"], + "decision": cell["decision"], + } + + +def parse_sheet( + path: str | Path, + stream: Mapping[str, Mapping[str, Any]], + *, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> dict[str, Any]: + """Validate a private lane sheet. Does not write the sheet or fill decisions.""" + lines = Path(path).read_text(encoding="utf-8").splitlines() + if not lines: + refuse("settlement sheet is empty") + header = tuple(lines[0].split("\t")) + if header != SHEET_COLUMNS: + refuse("sheet header does not match the operator lane contract") + records = [] + unset = 0 + lane_rows: dict[str, int] = {} + seen: set[str] = set() + errors: list[str] = [] + for line in lines[1:]: + if not line.strip(): + continue + parts = line.split("\t") + if len(parts) != len(header): + errors.append("REFUSE: sheet row is not rectangular") + continue + row_id = parts[1].strip() + stream_row = stream.get(row_id) + if stream_row is None: + errors.append(f"REFUSE: row_id is not in the held-out stream: {row_id}") + continue + if row_id in seen or row_id in set(prior_ids or ()): + errors.append(f"REFUSE: duplicate settlement for {row_id}") + continue + seen.add(row_id) + try: + digest = _check_sheet_integrity(parts, header, stream_row) + cell = dict(zip(header, parts)) + lane = cell["lane"].strip() + lane_rows[lane] = lane_rows.get(lane, 0) + 1 + chosen = _operator_cells(cell) + if all(_blank(value) for value in chosen.values()): + unset += 1 + continue + if _blank(chosen["decision"]): + refuse(f"decision is empty for {row_id}; refusing to infer one") + record = validate_settlement( + { + "row_id": row_id, + "text_hash": digest, + "decision": chosen["decision"], + "semantic_family": chosen["semantic_family"], + "attest": chosen["attest"], + "register": chosen["register"], + "function": chosen["function"], + "source_hint": evidence_hint(stream_row), + "proposed_family": cell["proposed_family_evidence"], + "operator": operator, + "settled_at": settled_at, + "provenance": provenance, + "rights": cell["rights"], + "lane": lane, + }, + stream_row, + activated=activated, + prior_ids=prior_ids, + ) + records.append(record) + except SystemExit as exc: + errors.append(str(exc)) + if errors: + refuse("; ".join(message.removeprefix("REFUSE: ") for message in errors)) + return {"records": records, "unset_row_count": unset, "lane_rows": lane_rows, "row_ids": seen} + + +def load_record_file( + path: str | Path, + stream: Mapping[str, Mapping[str, Any]], + *, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, +) -> list[dict[str, Any]]: + records = [] + seen = set(prior_ids or ()) + with Path(path).open(encoding="utf-8") as handle: + for line in handle: + if not line.strip(): + continue + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("settlement record is not an object") + row_id = payload.get("row_id") + if not isinstance(row_id, str) or row_id not in stream: + refuse(f"row_id is not in the held-out stream: {row_id}") + if row_id in seen: + refuse(f"duplicate settlement for {row_id}") + payload = dict(payload) + payload.setdefault("operator", operator) + payload.setdefault("provenance", provenance) + payload.setdefault("settled_at", settled_at) + records.append( + validate_settlement( + payload, + stream[row_id], + activated=activated, + prior_ids=prior_ids, + ) + ) + seen.add(row_id) + return records + + +def _decision_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: + counts = {name: 0 for name in DECISIONS} + for record in records: + counts[str(record["decision"])] += 1 + return counts + + +def _family_counts(records: Sequence[Mapping[str, Any]]) -> dict[str, int]: + counts = {name: 0 for name in (*ACTIVE_FAMILIES, ABSTAIN)} + for record in records: + if record["decision"] not in SETTLED_DECISIONS: + continue + family = record.get("semantic_family") + if family in counts: + counts[str(family)] += 1 + return counts + + +def build_receipt( + records: Sequence[Mapping[str, Any]], + *, + batch_id: str, + operator: str, + provenance: str, + unset_row_count: int, + lane_rows: Mapping[str, int], + stream_run_id: str = "", + settled_at: str = "", + input_sheets: Sequence[Mapping[str, str]] | None = None, +) -> dict[str, Any]: + if not str(batch_id or "").strip(): + refuse("batch_id is required") + if not str(operator or "").strip() or not str(provenance or "").strip(): + refuse("operator and provenance are required") + settled = [record for record in records if record["decision"] in SETTLED_DECISIONS] + unresolved = [record for record in records if record["decision"] == "UNRESOLVED"] + body = { + "schema": RECEIPT_SCHEMA, + "batch_id": batch_id, + "operator": operator, + "provenance": provenance, + "settled_row_count": len(settled), + "unresolved_row_count": len(unresolved), + "unset_row_count": unset_row_count, + "decision_counts": _decision_counts(records), + "family_counts": _family_counts(records), + "observed_count": sum(1 for record in settled if record.get("attest") == "OBSERVED"), + "inferred_count": sum(1 for record in settled if record.get("attest") == "INFERRED"), + "rights_blocked_settled_count": sum( + 1 for record in settled if record.get("rights") == "RIGHTS_UNRESOLVED" + ), + "reserve_eligible_count": sum(1 for record in records if settlement_allows_reserve(record)), + "reserve_added": 0, + "ledger_mutated": False, + "vendor_calls": VENDOR_CALLS, + "evaluation_enabled": False, + "lane_rows": dict(lane_rows), + } + run_id = str(stream_run_id or "").strip() + if run_id: + body["stream_run_id"] = run_id + stamp = str(settled_at or "").strip() + if stamp: + body["settled_at"] = stamp + if input_sheets: + sheets = [] + for item in input_sheets: + identity = str(item.get("identity") or "").strip() + digest = str(item.get("sha256") or "").strip() + if not identity or len(digest) != 64: + refuse("input sheet identity is incomplete") + sheets.append({"identity": identity, "sha256": digest}) + body["input_sheets"] = sheets + _walk_forbid(body) + return seal_receipt(body) + + +def seal_receipt(body: Mapping[str, Any]) -> dict[str, Any]: + payload = {key: value for key, value in body.items() if key != "receipt_sha256"} + raw = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8") + sealed = dict(payload) + sealed["receipt_sha256"] = hashlib.sha256(raw).hexdigest() + return sealed + + +def receipt_sha_matches(receipt: Mapping[str, Any]) -> bool: + sealed = seal_receipt(receipt) + return sealed["receipt_sha256"] == receipt.get("receipt_sha256") + + +def _walk_forbid(value: Any) -> None: + if isinstance(value, dict): + for key, item in value.items(): + if key in _FORBIDDEN_KEYS: + refuse(f"receipt must not carry {key}") + _walk_forbid(item) + elif isinstance(value, list): + for item in value: + _walk_forbid(item) + + +def load_logged_ids(path: str | Path) -> set[str]: + log = Path(path) + if not log.exists(): + return set() + found = set() + for line in log.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + payload = json.loads(line) + if not isinstance(payload, dict): + refuse("settlement log row is not an object") + _forbid_payload(payload, "settlement log") + row_id = payload.get("row_id") + if not isinstance(row_id, str) or not row_id: + refuse("settlement log row is missing row_id") + if row_id in found: + refuse(f"settlement log already contains a duplicate for {row_id}") + found.add(row_id) + return found + + +def _write_exclusive(path: Path, text: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "w", encoding="utf-8") as handle: + handle.write(text) + + +def commit_settlement( + records: Sequence[Mapping[str, Any]], + receipt: Mapping[str, Any], + *, + log_path: str | Path, + receipt_path: str | Path, +) -> dict[str, Any]: + """Append settlement rows, then write a new receipt. Never rewrites either.""" + receipt_file = Path(receipt_path) + if receipt_file.exists(): + refuse(f"receipt already exists: {receipt_file}") + log = Path(log_path) + prior = load_logged_ids(log) + fresh_ids = [] + for record in records: + _forbid_payload(record, "settlement record") + row_id = str(record["row_id"]) + if row_id in prior or row_id in fresh_ids: + refuse(f"duplicate settlement for {row_id}") + fresh_ids.append(row_id) + if not log.exists(): + _write_exclusive(log, "") + if records: + with log.open("a", encoding="utf-8") as handle: + for record in records: + handle.write(json.dumps(record, sort_keys=True) + "\n") + sealed = dict(receipt) + _walk_forbid(sealed) + _write_exclusive(receipt_file, json.dumps(sealed, indent=2, sort_keys=True) + "\n") + os.chmod(log, 0o600) + return sealed + + +def settle( + stream_rows: Sequence[Mapping[str, Any]], + decisions: Sequence[Mapping[str, Any]], + *, + batch_id: str, + operator: str, + provenance: str, + settled_at: str, + activated: set[str] | None = None, + prior_ids: set[str] | None = None, + unset_row_count: int = 0, + lane_rows: Mapping[str, int] | None = None, +) -> dict[str, Any]: + """Validate explicit decisions. Blank callers pass no decisions and settle nothing.""" + stream = _stream_index(stream_rows) + prior = set(prior_ids or ()) + records = [] + seen: set[str] = set() + for incoming in decisions: + row_id = incoming.get("row_id") + stream_row = stream.get(row_id) if isinstance(row_id, str) else None + if stream_row is None: + refuse(f"row_id is not in the held-out stream: {row_id}") + if row_id in seen: + refuse(f"duplicate settlement for {row_id}") + seen.add(str(row_id)) + payload = dict(incoming) + payload.setdefault("operator", operator) + payload.setdefault("provenance", provenance) + payload.setdefault("settled_at", settled_at) + records.append( + validate_settlement(payload, stream_row, activated=activated, prior_ids=prior) + ) + receipt = build_receipt( + records, + batch_id=batch_id, + operator=operator, + provenance=provenance, + unset_row_count=unset_row_count, + lane_rows=dict(lane_rows or {}), + ) + return {"records": records, "receipt": receipt} + + +def run_settlement_apply(args: Any) -> int: + """CLI entry. Reads operator cells. Does not choose them or touch the ledger.""" + import sys + + stream = load_stream(args.stream_rows) + activated = set(args.activated_family or []) + sheets = list(args.sheet or []) + records_path = str(args.records or "").strip() + if not sheets and not records_path: + refuse("no operator surface supplied") + prior = load_logged_ids(args.settlement_log) + records: list[dict[str, Any]] = [] + unset = 0 + lane_rows: dict[str, int] = {} + decided: set[str] = set() + surfaced: set[str] = set() + for sheet in sheets: + parsed = parse_sheet( + sheet, + stream, + operator=args.operator, + provenance=args.provenance, + settled_at=args.settled_at, + activated=activated, + prior_ids=prior | decided, + ) + for row_id in parsed["row_ids"]: + if row_id in surfaced: + refuse(f"duplicate settlement for {row_id}") + surfaced.add(row_id) + for record in parsed["records"]: + decided.add(record["row_id"]) + records.append(record) + unset += int(parsed["unset_row_count"]) + for lane, count in parsed["lane_rows"].items(): + lane_rows[lane] = lane_rows.get(lane, 0) + count + if records_path: + loaded = load_record_file( + records_path, + stream, + operator=args.operator, + provenance=args.provenance, + settled_at=args.settled_at, + activated=activated, + prior_ids=prior | decided, + ) + for record in loaded: + if record["row_id"] in decided or record["row_id"] in surfaced: + refuse(f"duplicate settlement for {record['row_id']}") + decided.add(record["row_id"]) + records.append(record) + sheet_identities = [] + for sheet in sheets: + raw = Path(sheet).read_bytes() + sheet_identities.append( + {"identity": Path(sheet).name, "sha256": hashlib.sha256(raw).hexdigest()} + ) + receipt = build_receipt( + records, + batch_id=args.batch_id, + operator=args.operator, + provenance=args.provenance, + unset_row_count=unset, + lane_rows=lane_rows, + stream_run_id=str(getattr(args, "stream_run_id", "") or ""), + settled_at=str(args.settled_at or ""), + input_sheets=sheet_identities, + ) + sealed = commit_settlement( + records, + receipt, + log_path=args.settlement_log, + receipt_path=args.receipt, + ) + sys.stdout.write(json.dumps(sealed, indent=2, sort_keys=True) + "\n") + return 0 diff --git a/scripts/shadow/hyperlexical/holdout_eligibility.py b/scripts/shadow/hyperlexical/holdout_eligibility.py new file mode 100644 index 00000000..8f3e8898 --- /dev/null +++ b/scripts/shadow/hyperlexical/holdout_eligibility.py @@ -0,0 +1,152 @@ +"""Holdout eligibility against a pinned training export. + +Construction and the runtime filter both call +``holdout_guard.normalized_text_sha256``. This module does not draw a +manifest, score rows, or train. +""" + +from __future__ import annotations + +from collections import Counter +from pathlib import Path +from typing import Any, Mapping, Sequence + +from .holdout_guard import _read_manifest, normalized_text_sha256, operational_status +from .selection_surface import row_id + +ELIGIBLE = "eligible" +SPLIT_TEST = "split_test" +TRAIN_ROW_ID = "train_row_id" +TRAIN_TEXT_HASH = "train_text_hash" +SPENT_OR_ABANDONED = "spent_or_abandoned" + + +def text_hash(row: Mapping[str, Any]) -> str: + return normalized_text_sha256(str(row.get("text") or "")) + + +def identity_index(rows: Sequence[Mapping[str, Any]]) -> tuple[set[str], set[str]]: + """Row ids and canonical text hashes present in ``rows``.""" + ids: set[str] = set() + hashes: set[str] = set() + for row in rows: + ids.add(row_id(row)) + hashes.add(text_hash(row)) + return ids, hashes + + +def manifest_identity(path: str | Path) -> tuple[set[str], set[str], str]: + """Ids and hashes stored on a manifest, plus its operational status.""" + record, ids, hashes = _read_manifest(str(path)) + return ids, hashes, operational_status(record) + + +def exclusion_reason( + row: Mapping[str, Any], + *, + train_ids: set[str], + train_hashes: set[str], + other_ids: set[str], + other_hashes: set[str], +) -> str: + """Waterfall reason. First match wins. ``eligible`` means none matched.""" + if str(row.get("split") or "") == "test": + return SPLIT_TEST + ident = row_id(row) + digest = text_hash(row) + if ident in train_ids: + return TRAIN_ROW_ID + if digest in train_hashes: + return TRAIN_TEXT_HASH + if ident in other_ids or digest in other_hashes: + return SPENT_OR_ABANDONED + return ELIGIBLE + + +def multiplicity(rows: Sequence[Mapping[str, Any]]) -> dict[str, Any]: + """How many rows share one canonical text hash. No hash values are kept.""" + counts = Counter(text_hash(row) for row in rows) + histogram = Counter(counts.values()) + return { + "rows": len(rows), + "unique_hashes": len(counts), + "max_rows_per_hash": max(histogram) if histogram else 0, + "hashes_with_multiple_rows": sum( + count for size, count in histogram.items() if size > 1 + ), + "histogram_rows_per_hash": {str(size): histogram[size] for size in sorted(histogram)}, + } + + +def census( + source_rows: Sequence[Mapping[str, Any]], + train_rows: Sequence[Mapping[str, Any]], + *, + extra_rows: Sequence[Mapping[str, Any]] = (), + extra_manifests: Sequence[str | Path] = (), +) -> dict[str, Any]: + """Eligible universe after row-id and normalized-text exclusion. + + ``extra_rows`` and ``extra_manifests`` are spent or abandoned exclusions. + ``unbind_clean`` uses ``soft_ceiling.clean_surface`` and is not the + canonical text hash. + """ + from .soft_ceiling import clean_surface + + train_ids, train_hashes = identity_index(train_rows) + other_ids, other_hashes = identity_index(extra_rows) + manifest_status: list[dict[str, str]] = [] + for path in extra_manifests: + ids, hashes, status = manifest_identity(path) + other_ids |= ids + other_hashes |= hashes + manifest_status.append({"path": str(path), "operational_status": status}) + reasons: Counter[str] = Counter() + eligible: list[Mapping[str, Any]] = [] + source_hashes: set[str] = set() + for row in source_rows: + source_hashes.add(text_hash(row)) + reason = exclusion_reason( + row, + train_ids=train_ids, + train_hashes=train_hashes, + other_ids=other_ids, + other_hashes=other_hashes, + ) + reasons[reason] += 1 + if reason == ELIGIBLE: + eligible.append(row) + eligible_hashes = {text_hash(row) for row in eligible} + classify = [row for row in eligible if row.get("task") == "classify"] + unbind = [row for row in eligible if row.get("task") == "unbind"] + observed = [row for row in classify if str(row.get("class")) == "OBSERVED"] + non_none = [row for row in classify if str(row.get("lineage")) not in ("", "none")] + train_for_clean = [row for row in train_rows if row.get("split") == "train"] + clean, _account = clean_surface(unbind, train_rows=train_for_clean) + return { + "source_rows": len(source_rows), + "source_unique_text_hashes": len(source_hashes), + "excluded_by_split_test": reasons[SPLIT_TEST], + "excluded_by_training_row_id": reasons[TRAIN_ROW_ID], + "excluded_by_training_normalized_text_hash": reasons[TRAIN_TEXT_HASH], + "excluded_by_spent_or_abandoned": reasons[SPENT_OR_ABANDONED], + "eligible_rows": len(eligible), + "eligible_unique_text_hashes": len(eligible_hashes), + "classify_eligible": len(classify), + "classify_observed_eligible": len(observed), + "classify_non_none_eligible": len(non_none), + "unbind_eligible": len(unbind), + "unbind_clean_eligible": len(clean), + "unbind_clean_definition": "soft_ceiling.clean_surface", + "source_multiplicity": multiplicity(source_rows), + "train_multiplicity": multiplicity(train_rows), + "extra_manifests": manifest_status, + "waterfall_sums_to_source": ( + reasons[SPLIT_TEST] + + reasons[TRAIN_ROW_ID] + + reasons[TRAIN_TEXT_HASH] + + reasons[SPENT_OR_ABANDONED] + + len(eligible) + == len(source_rows) + ), + } diff --git a/scripts/shadow/hyperlexical/holdout_guard.py b/scripts/shadow/hyperlexical/holdout_guard.py index e58e4897..d5cbf19d 100644 --- a/scripts/shadow/hyperlexical/holdout_guard.py +++ b/scripts/shadow/hyperlexical/holdout_guard.py @@ -2,6 +2,17 @@ Manifests are id lists plus sha256 hashes of census-normalized text. This module does not train, score, or read row text out of a manifest. + +Canonical text identity, shared by holdout construction and this filter: + + source text + -> heldout_census.normalize_group_text + -> UTF-8 bytes + -> sha256 hex digest + +``normalize_group_text`` applies NFKC, casefold, replaces URLs and every +non-alphanumeric character with a space, then collapses whitespace. There is +no second normalizer on this path. An empty normalized string still hashes. """ from __future__ import annotations @@ -17,6 +28,10 @@ HOLDOUT_MANIFESTS_ENV = "HLX_HOLDOUT_MANIFESTS" ALLOW_NO_HOLDOUT_ENV = "HLX_ALLOW_NO_HOLDOUT" +EXPERIMENT_ID_ENV = "HLX_EXPERIMENT_ID" +ADMISSIBLE_STATUS = "UNSCORED_SEALED" +ABANDONED_STATUS = "UNSCORED_ABANDONED" +LIFECYCLE_FILENAME = "holdout-lifecycle.json" _ID_KEYS = frozenset({"row_ids", "ids"}) _HASH_KEYS = frozenset({"normalized_text_sha256"}) _REMOVED_KEYS = ("classify_train", "classify_val", "unbind_train", "unbind_val") @@ -45,11 +60,54 @@ def _empty_spec() -> HoldoutSpec: def normalized_text_sha256(text: str) -> str: - """sha256 of ``heldout_census.normalize_group_text`` (UTF-8).""" + """sha256 hex of the canonical normalized text, UTF-8. + + Algorithm: NFKC, casefold, URL strip, non-alphanumeric to space, + whitespace collapse, then SHA-256. Punctuation does not survive. + """ normalized = normalize_group_text(text) return hashlib.sha256(normalized.encode("utf-8")).hexdigest() +def lifecycle_path(manifest: str | Path) -> Path: + return Path(manifest).parent / LIFECYCLE_FILENAME + + +def read_lifecycle(manifest: str | Path) -> dict[str, Any] | None: + """Append-only lifecycle receipt beside a sealed manifest, if present.""" + path = lifecycle_path(manifest) + if not path.is_file(): + return None + try: + payload = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError: + raise SystemExit(f"REFUSE: holdout lifecycle is not JSON: {path}") from None + if not isinstance(payload, dict): + raise SystemExit(f"REFUSE: holdout lifecycle must be a JSON object: {path}") + return payload + + +def operational_status(record: Mapping[str, Any]) -> str: + """Sealed manifest status, unless a lifecycle receipt supersedes it. + + The manifest file stays byte-stable. ``UNSCORED_ABANDONED`` means the + holdout was never scored, is not reusable, and is not ``SCORED_SPENT``. + """ + sealed = record.get("status") + life = read_lifecycle(str(record.get("path") or "")) + if life is None: + return "" if sealed is None else str(sealed) + bound = str(life.get("manifest_sha256") or "") + if bound != record.get("sha256"): + raise SystemExit( + "REFUSE: holdout lifecycle manifest_sha256 does not match the manifest" + ) + status = life.get("status") + if not isinstance(status, str) or not status.strip(): + raise SystemExit("REFUSE: holdout lifecycle status is missing") + return status + + def allow_no_holdout(raw: str | None = None) -> bool: if raw is None: raw = os.environ.get(ALLOW_NO_HOLDOUT_ENV) @@ -151,15 +209,31 @@ def _read_manifest(path: str) -> tuple[dict[str, Any], set[str], set[str]]: raise SystemExit( f"REFUSE: holdout manifest has no row_ids or normalized_text_sha256: {file}" ) + status, experiment_id = _admission_fields(payload) record = { "path": str(file), "sha256": _file_sha256(file), "n_row_ids": len(ids), "n_text_hashes": len(hashes), + "status": status, + "experiment_id": experiment_id, } return record, ids, hashes +def _admission_fields(payload: Any) -> tuple[str | None, str | None]: + """Top-level status and experiment id. Nested copies do not authorize a run.""" + if not isinstance(payload, dict): + return None, None + status = payload.get("status") + experiment_id = payload.get("experiment_id") + if status is not None and not isinstance(status, str): + raise SystemExit("REFUSE: holdout manifest status must be a string") + if experiment_id is not None and not isinstance(experiment_id, str): + raise SystemExit("REFUSE: holdout manifest experiment_id must be a string") + return status, experiment_id + + def load_holdout_spec(raw: str | None = None) -> HoldoutSpec: """Load manifests from ``raw`` or ``HLX_HOLDOUT_MANIFESTS``. Empty env → empty spec.""" paths = resolve_manifest_paths(raw) @@ -177,15 +251,56 @@ def load_holdout_spec(raw: str | None = None) -> HoldoutSpec: def require_holdout_for_training() -> HoldoutSpec: - """Fail when training is armed and no manifest or explicit override is set.""" + """Legacy manifest gate. Not the controlled-experiment contract. + + Controlled experiments admit through ``admit_training_run`` under + ``CONTROLLED_RESERVE``. This function still admits a non-experiment + launch that sets ``HLX_HOLDOUT_MANIFESTS`` or ``HLX_ALLOW_NO_HOLDOUT``. + + Admit training only with a fresh, experiment-bound holdout. + + Row exclusion still uses every id and text hash in the loaded manifests. + A ``SCORED_SPENT`` or ``SCORED`` file does not authorize a run. Missing, + unknown, and unbound statuses fail closed. ``HLX_ALLOW_NO_HOLDOUT=1`` + remains the explicit no-manifest override and is not the admission path. + """ spec = load_holdout_spec() - if os.environ.get("HYPERLEX_ALLOW_TRAIN") == "1" and not spec.manifests and not allow_no_holdout(): + if os.environ.get("HYPERLEX_ALLOW_TRAIN") != "1": + return spec + if not spec.manifests and allow_no_holdout(): + return spec + if not spec.manifests: raise SystemExit( "REFUSE: HYPERLEX_ALLOW_TRAIN=1 but no holdout manifest. " f"Set {HOLDOUT_MANIFESTS_ENV} to comma-separated manifest paths, " "pass --holdout-manifest, or set " f"{ALLOW_NO_HOLDOUT_ENV}=1 to override." ) + expected = os.environ.get(EXPERIMENT_ID_ENV) + for item in spec.manifests: + status = item.get("status") + if status is None: + raise SystemExit( + "REFUSE: holdout manifest status is missing; required " + f"{ADMISSIBLE_STATUS}" + ) + if status != ADMISSIBLE_STATUS: + raise SystemExit( + f"REFUSE: holdout manifest status {status} is not admissible; " + f"required {ADMISSIBLE_STATUS}" + ) + operational = operational_status(item) + if operational != ADMISSIBLE_STATUS: + raise SystemExit( + "REFUSE: holdout operational status " + f"{operational} is not admissible; required {ADMISSIBLE_STATUS}" + ) + bound = item.get("experiment_id") + if not expected or bound != expected: + raise SystemExit( + "REFUSE: holdout experiment binding " + f"{bound!r} does not match {EXPERIMENT_ID_ENV}" + ) return spec @@ -200,6 +315,52 @@ def holdout_match(row: Mapping[str, Any], spec: HoldoutSpec) -> str | None: return None +def disjoint_report(rows: Sequence[Mapping[str, Any]], spec: HoldoutSpec) -> dict[str, int | bool]: + """Count pinned rows the runtime filter would drop. + + Overlap counts are rows, not distinct hashes. A controlled experiment is + disjoint only when that removal count is zero. The filter stays in place + as defense in depth; it is not how equivalence is established. + """ + id_overlap = 0 + text_overlap = 0 + removed = 0 + for row in rows: + ident = row_id(row) in spec.row_ids + digest = normalized_text_sha256(str(row.get("text") or "")) in spec.text_hashes + if ident: + id_overlap += 1 + if digest: + text_overlap += 1 + if ident or digest: + removed += 1 + _kept, filtered = filter_holdout_rows(list(rows), spec) + if filtered != removed: + raise SystemExit("REFUSE: holdout filter count diverged from the disjoint report") + return { + "holdout_train_row_id_overlap": id_overlap, + "holdout_train_text_hash_overlap": text_overlap, + "holdout_filter_training_rows_removed": removed, + "holdout_training_disjoint": removed == 0, + } + + +def assert_pinned_holdout_disjoint( + rows: Sequence[Mapping[str, Any]], + spec: HoldoutSpec, +) -> dict[str, int | bool]: + """Fail closed before training when a pinned export meets the holdout.""" + report = disjoint_report(rows, spec) + if report["holdout_training_disjoint"]: + return report + raise SystemExit( + "ADMISSION FAIL: holdout is not disjoint from the pinned training export " + f"(row_id_overlap={report['holdout_train_row_id_overlap']}, " + f"text_hash_overlap={report['holdout_train_text_hash_overlap']}, " + f"rows_removed={report['holdout_filter_training_rows_removed']})" + ) + + def filter_holdout_rows(rows: list, spec: HoldoutSpec) -> tuple[list, int]: """Drop matching rows. Returns the original list when nothing matches.""" if not spec.active: diff --git a/scripts/shadow/hyperlexical/identity_ledger.py b/scripts/shadow/hyperlexical/identity_ledger.py new file mode 100644 index 00000000..9e9968ab --- /dev/null +++ b/scripts/shadow/hyperlexical/identity_ledger.py @@ -0,0 +1,871 @@ +"""Global text-identity ledger for Hyperlex evaluation reserves. + +Contamination is owned by ``holdout_guard.normalized_text_sha256``. This +module does not train, score, or store raw text. A second normalizer is not +defined here. ``unbind_clean`` on a label is a slice predicate computed by +the caller with ``soft_ceiling.clean_surface``; it is not an identity. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path +from typing import Any, Mapping, Sequence + +from .holdout_guard import normalized_text_sha256 +from .selection_surface import row_id + +SCHEMA = "hyperlex.identity_ledger.v1" +EVENT_SCHEMA = "hyperlex.identity_event.v1" +POLICY_ID = "hyperlex.eval_reserve.v1" +GENERATION = "GEN-0" +RESERVE_LEDGER_ENV = "HLX_EVAL_RESERVE_LEDGER" + +# Sealed planning quotas. These are the SELECT-002 row composition, not a +# statistical minimum. See evaluation-reserve.md. +PLANNING_TARGETS: dict[str, int] = { + "classify": 606, + "classify_observed": 287, + "classify_non_none": 604, + "unbind_clean": 250, +} +REQUIRED_SLICES = tuple(PLANNING_TARGETS) + +FORWARD = frozenset( + { + ("NEW", "TRAIN_CANDIDATE"), + ("NEW", "EVAL_RESERVE"), + ("AVAILABLE", "TRAIN_CANDIDATE"), + ("AVAILABLE", "EVAL_RESERVE"), + ("TRAIN_CANDIDATE", "TRAIN_CONSUMED"), + ("EVAL_RESERVE", "EVAL_BOUND"), + ("EVAL_BOUND", "EVAL_SPENT"), + ("EVAL_BOUND", "EVAL_ABANDONED"), + } +) +HISTORICAL_FLAGS = frozenset( + {"training_consumed", "evaluation_spent", "evaluation_abandoned"} +) +_FORBIDDEN_KEYS = frozenset( + { + "text", + "normalized_text", + "raw_text", + "candidate_score", + "model_error", + "model_score", + "prediction", + } +) +_SCORE_KEYS = frozenset({"candidate_score", "model_error", "model_score", "prediction"}) + + +def _refuse_mapping(value: Mapping[str, Any], where: str) -> None: + for key in value: + if key in _FORBIDDEN_KEYS: + raise SystemExit(f"REFUSE: {where} must not carry {key}") + + +def policy_sealed() -> bool: + """The assignment rule is the code under ``POLICY_ID``. It does not read scores.""" + return True + + +def _empty_identity(digest: str) -> dict[str, Any]: + return { + "normalized_text_sha256": digest, + "row_ids": [], + "source_artifacts": [], + "first_seen": "", + "current_sources": [], + "training_consumed": False, + "training_artifacts": [], + "evaluation_spent": False, + "evaluation_spent_artifacts": [], + "evaluation_abandoned": False, + "evaluation_abandoned_artifacts": [], + "evaluation_reserved": False, + "evaluation_bound": False, + "train_candidate": False, + "catalogued": False, + "experiment_bindings": [], + "labels": [], + "provenance": [], + "generation": GENERATION, + } + + +def derived_state(record: Mapping[str, Any]) -> str: + """Blocking historical flags win. Routing flags apply only when none are set.""" + if record.get("training_consumed"): + return "TRAIN_CONSUMED" + if record.get("evaluation_spent"): + return "EVAL_SPENT" + if record.get("evaluation_abandoned"): + return "EVAL_ABANDONED" + if record.get("evaluation_bound"): + return "EVAL_BOUND" + if record.get("evaluation_reserved"): + return "EVAL_RESERVE" + if record.get("train_candidate"): + return "TRAIN_CANDIDATE" + if record.get("catalogued"): + return "AVAILABLE" + return "NEW" + + +def slices_of(record: Mapping[str, Any]) -> set[str]: + """Metric slices this identity can support. One identity counts once per slice.""" + found: set[str] = set() + classify = [lab for lab in record.get("labels") or [] if lab.get("task") == "classify"] + if classify: + found.add("classify") + if any(lab.get("class") == "OBSERVED" for lab in classify): + found.add("classify_observed") + if any(str(lab.get("lineage") or "") not in ("", "none") for lab in classify): + found.add("classify_non_none") + if any(lab.get("task") == "unbind" and lab.get("unbind_clean") is True for lab in record.get("labels") or []): + found.add("unbind_clean") + return found + + +def _sorted_unique(items: Sequence[str]) -> list[str]: + return sorted({item for item in items if item}) + + +def _label(row: Mapping[str, Any], *, unbind_clean: bool | None) -> dict[str, Any]: + label: dict[str, Any] = { + "task": str(row.get("task") or ""), + "class": str(row.get("class") or ""), + "lineage": str(row.get("lineage") or ""), + "split": str(row.get("split") or ""), + } + if label["task"] == "unbind": + label["unbind_clean"] = bool(unbind_clean) + return label + + +def _merge_label(labels: list[dict[str, Any]], label: Mapping[str, Any]) -> None: + if label not in labels: + labels.append(dict(label)) + labels.sort(key=lambda item: json.dumps(item, sort_keys=True)) + + +class IdentityLedger: + """Append-only events plus a derived identity index. No raw text.""" + + def __init__(self) -> None: + self.events: list[dict[str, Any]] = [] + self.identities: dict[str, dict[str, Any]] = {} + self.row_owner: dict[str, str] = {} + self.row_id_collisions: list[dict[str, str]] = [] + + def identity(self, digest: str) -> dict[str, Any] | None: + return self.identities.get(digest) + + def _append(self, event: dict[str, Any]) -> None: + _refuse_mapping(event, "ledger event") + event = dict(event) + event["schema"] = EVENT_SCHEMA + event["seq"] = len(self.events) + 1 + event["generation"] = GENERATION + self.events.append(event) + + def _touch(self, digest: str) -> dict[str, Any]: + record = self.identities.get(digest) + if record is None: + record = _empty_identity(digest) + self.identities[digest] = record + return record + + def observe( + self, + digest: str, + *, + source_artifact: str, + row_ids: Sequence[str] = (), + labels: Sequence[Mapping[str, Any]] = (), + current_source: bool = False, + catalogued: bool = False, + provenance: str, + experiment_id: str = "", + ) -> dict[str, Any]: + """Attach ids, labels, and provenance. Does not route the identity.""" + if len(digest) != 64: + raise SystemExit("REFUSE: normalized_text_sha256 must be 64 hex characters") + record = self._touch(digest) + if not record["first_seen"]: + record["first_seen"] = source_artifact + record["source_artifacts"] = _sorted_unique([*record["source_artifacts"], source_artifact]) + if current_source: + record["current_sources"] = _sorted_unique([*record["current_sources"], source_artifact]) + if catalogued: + record["catalogued"] = True + for ident in row_ids: + self._claim_row_id(record, ident) + for label in labels: + _merge_label(record["labels"], label) + if experiment_id: + record["experiment_bindings"] = _sorted_unique( + [*record["experiment_bindings"], experiment_id] + ) + if provenance and provenance not in record["provenance"]: + record["provenance"].append(provenance) + self._append( + { + "kind": "observe", + "normalized_text_sha256": digest, + "source_artifact": source_artifact, + "row_ids": list(row_ids), + "current_source": current_source, + "catalogued": catalogued, + "provenance": provenance, + "experiment_id": experiment_id, + "labels": [dict(item) for item in labels], + } + ) + return record + + def _claim_row_id(self, record: dict[str, Any], ident: str) -> None: + if not ident: + return + owner = self.row_owner.get(ident) + digest = record["normalized_text_sha256"] + if owner is None: + self.row_owner[ident] = digest + elif owner != digest: + self.row_id_collisions.append( + {"row_id": ident, "existing_hash": owner, "new_hash": digest} + ) + if ident not in record["row_ids"]: + record["row_ids"] = _sorted_unique([*record["row_ids"], ident]) + + def mark_historical( + self, + digest: str, + flag: str, + *, + source_artifact: str, + provenance: str, + experiment_id: str = "", + ) -> dict[str, Any]: + """Record a receipt-backed past state. Flags only ever turn on.""" + if flag not in HISTORICAL_FLAGS: + raise SystemExit(f"REFUSE: unknown historical flag {flag}") + record = self._touch(digest) + record[flag] = True + artifact_key = { + "training_consumed": "training_artifacts", + "evaluation_spent": "evaluation_spent_artifacts", + "evaluation_abandoned": "evaluation_abandoned_artifacts", + }[flag] + record[artifact_key] = _sorted_unique([*record[artifact_key], source_artifact]) + if experiment_id: + record["experiment_bindings"] = _sorted_unique( + [*record["experiment_bindings"], experiment_id] + ) + if provenance and provenance not in record["provenance"]: + record["provenance"].append(provenance) + self._append( + { + "kind": "historical", + "normalized_text_sha256": digest, + "flag": flag, + "source_artifact": source_artifact, + "provenance": provenance, + "experiment_id": experiment_id, + } + ) + return record + + def transition(self, digest: str, to_state: str, *, source_artifact: str, provenance: str) -> str: + """Monotonic forward route. Historical blocks cannot be cleared.""" + record = self.identities.get(digest) + if record is None: + raise SystemExit("REFUSE: cannot route an unseen text identity") + state = derived_state(record) + if record["training_consumed"] or record["evaluation_spent"] or record["evaluation_abandoned"]: + raise SystemExit( + f"REFUSE: text identity is {state}; that block is monotonic in {GENERATION}" + ) + if (state, to_state) not in FORWARD: + raise SystemExit(f"REFUSE: transition {state} -> {to_state} is not allowed") + if to_state == "TRAIN_CANDIDATE": + record["train_candidate"] = True + elif to_state == "TRAIN_CONSUMED": + record["training_consumed"] = True + record["training_artifacts"] = _sorted_unique( + [*record["training_artifacts"], source_artifact] + ) + elif to_state == "EVAL_RESERVE": + record["evaluation_reserved"] = True + elif to_state == "EVAL_BOUND": + record["evaluation_bound"] = True + elif to_state == "EVAL_SPENT": + record["evaluation_spent"] = True + elif to_state == "EVAL_ABANDONED": + record["evaluation_abandoned"] = True + else: + raise SystemExit(f"REFUSE: unknown route {to_state}") + if provenance and provenance not in record["provenance"]: + record["provenance"].append(provenance) + self._append( + { + "kind": "transition", + "normalized_text_sha256": digest, + "from_state": state, + "to_state": to_state, + "source_artifact": source_artifact, + "provenance": provenance, + } + ) + return to_state + + def request_generation_reset(self, *, governed_receipt: str, confirm: bool) -> dict[str, Any]: + """Record a request. Does not clear flags and does not create GEN-1.""" + if not confirm or not str(governed_receipt or "").strip(): + raise SystemExit( + "REFUSE: generation reset requires confirm=True and a governed receipt" + ) + self._append( + { + "kind": "generation_reset_requested", + "normalized_text_sha256": "", + "source_artifact": governed_receipt, + "provenance": "requested only; flags unchanged; GEN-1 not created", + "applied": False, + } + ) + return {"applied": False, "generation": GENERATION, "flags_cleared": 0} + + def observe_row( + self, + row: Mapping[str, Any], + *, + source_artifact: str, + provenance: str, + current_source: bool = False, + catalogued: bool = False, + unbind_clean: bool | None = None, + experiment_id: str = "", + ) -> str: + """Hash one row and catalogue it. The row text is not retained.""" + if any(key in row for key in _SCORE_KEYS): + raise SystemExit("REFUSE: catalogue input carries a model outcome field") + digest = normalized_text_sha256(str(row.get("text") or "")) + ids = [row_id(row)] + declared = row.get("row_id") + if isinstance(declared, str) and declared and declared not in ids: + ids.append(declared) + self.observe( + digest, + source_artifact=source_artifact, + row_ids=ids, + labels=[_label(row, unbind_clean=unbind_clean)], + current_source=current_source, + catalogued=catalogued, + provenance=provenance, + experiment_id=experiment_id, + ) + return digest + + def reserve_counts(self) -> dict[str, int]: + counts = {key: 0 for key in REQUIRED_SLICES} + for record in self.identities.values(): + if derived_state(record) not in ("EVAL_RESERVE", "EVAL_BOUND"): + continue + for key in slices_of(record): + if key in counts: + counts[key] += 1 + return counts + + def screen( + self, + rows: Sequence[Mapping[str, Any]], + *, + batch_id: str, + targets: Mapping[str, int] | None = None, + unbind_clean_hashes: set[str] | None = None, + ) -> dict[str, Any]: + """Admission decision without mutation. Duplicate text does not raise n.""" + return self._route( + rows, + batch_id=batch_id, + targets=dict(targets or PLANNING_TARGETS), + unbind_clean_hashes=set(unbind_clean_hashes or ()), + apply=False, + ) + + def admit( + self, + rows: Sequence[Mapping[str, Any]], + *, + batch_id: str, + source_artifact: str, + targets: Mapping[str, int] | None = None, + unbind_clean_hashes: set[str] | None = None, + ) -> dict[str, Any]: + """Route genuinely new text. Rejects hashes already consumed, spent, abandoned, or reserved.""" + if not str(batch_id or "").strip(): + raise SystemExit("REFUSE: admission batch_id is required") + report = self._route( + rows, + batch_id=batch_id, + targets=dict(targets or PLANNING_TARGETS), + unbind_clean_hashes=set(unbind_clean_hashes or ()), + apply=True, + source_artifact=source_artifact, + ) + return report + + def _route( + self, + rows: Sequence[Mapping[str, Any]], + *, + batch_id: str, + targets: dict[str, int], + unbind_clean_hashes: set[str], + apply: bool, + source_artifact: str = "", + ) -> dict[str, Any]: + grouped: dict[str, list[Mapping[str, Any]]] = {} + collisions = 0 + batch_owner: dict[str, str] = {} + for row in rows: + if any(key in row for key in _SCORE_KEYS): + raise SystemExit("REFUSE: admission input carries a model outcome field") + digest = normalized_text_sha256(str(row.get("text") or "")) + grouped.setdefault(digest, []).append(row) + declared = row.get("row_id") if isinstance(row.get("row_id"), str) else "" + computed = row_id(row) + idents = [declared] if declared else [] + if computed not in idents: + idents.append(computed) + row_collides = False + for ident in idents: + owner = self.row_owner.get(ident) or batch_owner.get(ident) + if owner is not None and owner != digest: + row_collides = True + elif ident not in self.row_owner: + batch_owner.setdefault(ident, digest) + if row_collides: + collisions += 1 + def sort_key(digest: str) -> str: + return hashlib.sha256(f"{GENERATION}|{batch_id}|{digest}".encode("utf-8")).hexdigest() + + counts = self.reserve_counts() + rejected: dict[str, int] = { + "TRAIN_CONSUMED": 0, + "EVAL_SPENT": 0, + "EVAL_ABANDONED": 0, + "EVAL_RESERVE": 0, + "EVAL_BOUND": 0, + "TRAIN_CANDIDATE": 0, + } + reserved = 0 + candidates = 0 + duplicate_text_rows = 0 + accepted_hashes: list[str] = [] + for digest in sorted(grouped, key=sort_key): + members = grouped[digest] + if len(members) > 1: + duplicate_text_rows += len(members) - 1 + record = self.identities.get(digest) + state = derived_state(record) if record else "NEW" + if state in rejected: + rejected[state] += 1 + continue + if state not in ("NEW", "AVAILABLE"): + raise SystemExit(f"REFUSE: cannot admit text in state {state}") + pending_labels = [] + pending_ids: list[str] = [] + for row in members: + clean = digest in unbind_clean_hashes + pending_labels.append(_label(row, unbind_clean=clean if str(row.get("task") or "") == "unbind" else None)) + pending_ids.append(row_id(row)) + declared = row.get("row_id") + if isinstance(declared, str) and declared: + pending_ids.append(declared) + snapshot = _empty_identity(digest) + if record: + snapshot["labels"] = [dict(item) for item in record["labels"]] + for label in pending_labels: + _merge_label(snapshot["labels"], label) + served = slices_of(snapshot) + opens = [key for key in served if counts.get(key, 0) < int(targets.get(key, 0))] + to_state = "EVAL_RESERVE" if opens else "TRAIN_CANDIDATE" + if apply: + self.observe( + digest, + source_artifact=source_artifact, + row_ids=pending_ids, + labels=pending_labels, + provenance=f"admitted batch {batch_id}", + ) + self.transition( + digest, + to_state, + source_artifact=source_artifact, + provenance=f"assignment {POLICY_ID} batch {batch_id}", + ) + if to_state == "EVAL_RESERVE": + reserved += 1 + accepted_hashes.append(digest) + for key in served: + if key in counts: + counts[key] += 1 + else: + candidates += 1 + return { + "policy_id": POLICY_ID, + "generation": GENERATION, + "batch_id": batch_id, + "raw_rows": len(rows), + "unique_canonical_text_identities": len(grouped), + "unique_admitted_to_eval_reserve": reserved, + "unique_routed_to_train_candidate": candidates, + "duplicate_text_rows_not_counted": duplicate_text_rows, + "row_id_collisions_with_new_text": collisions, + "rejected_existing_identities": rejected, + "rejected_unique_identities": sum(rejected.values()), + "reserve_counts_after": counts if apply else self._projected_counts(counts), + "applied": apply, + "model_outcomes_consulted": False, + } + + def _projected_counts(self, counts: dict[str, int]) -> dict[str, int]: + return dict(counts) + + def select_003_gate(self, train_hashes: set[str]) -> dict[str, Any]: + """Necessary conditions only. Does not draft an experiment.""" + reserved = [ + record + for record in self.identities.values() + if derived_state(record) in ("EVAL_RESERVE", "EVAL_BOUND") + ] + hashes = {record["normalized_text_sha256"] for record in reserved} + counts = self.reserve_counts() + represented = {key: counts[key] >= 1 for key in REQUIRED_SLICES} + blocked = [ + record["normalized_text_sha256"] + for record in reserved + if record["training_consumed"] or record["evaluation_spent"] or record["evaluation_abandoned"] + ] + overlap = sorted(hashes & train_hashes) + eligible = ( + policy_sealed() + and bool(reserved) + and all(represented.values()) + and not blocked + and not overlap + ) + return { + "select_003": "NOT_DRAFTED", + "eligible": eligible, + "policy_sealed": policy_sealed(), + "reserve_identities": len(reserved), + "required_slices_represented": represented, + "text_disjoint_from_training": not overlap, + "training_overlap_identities": len(overlap), + "spent_or_abandoned_in_reserve": len(blocked), + "authorization": "SEPARATE", + "statistical_minimum": "NOT_COMPUTABLE", + } + + def census(self, live_hashes: set[str]) -> dict[str, Any]: + """Aggregate counts. Hash lists stay in the ledger, not in this summary.""" + def bucket(pred) -> set[str]: + return {digest for digest, record in self.identities.items() if pred(record)} + + consumed = bucket(lambda record: record["training_consumed"]) + spent = bucket(lambda record: record["evaluation_spent"]) + abandoned = bucket(lambda record: record["evaluation_abandoned"]) + available = bucket(lambda record: derived_state(record) == "AVAILABLE") + live_available = available & live_hashes + by_slice = {key: 0 for key in (*REQUIRED_SLICES, "unbind", "classify_inferred", "other")} + for digest in live_available: + record = self.identities[digest] + labels = list(record["labels"]) + tasks = {lab.get("task") for lab in labels} + classify = [lab for lab in labels if lab.get("task") == "classify"] + if classify: + by_slice["classify"] += 1 + if any(lab.get("class") == "OBSERVED" for lab in classify): + by_slice["classify_observed"] += 1 + if any(lab.get("class") == "INFERRED" for lab in classify): + by_slice["classify_inferred"] += 1 + if any(str(lab.get("lineage") or "") not in ("", "none") for lab in classify): + by_slice["classify_non_none"] += 1 + if "unbind" in tasks: + by_slice["unbind"] += 1 + if any( + lab.get("task") == "unbind" and lab.get("unbind_clean") is True for lab in labels + ): + by_slice["unbind_clean"] += 1 + if not tasks: + by_slice["other"] += 1 + return { + "generation": GENERATION, + "policy_id": POLICY_ID, + "identities": len(self.identities), + "all_live_text_hashes": len(live_hashes), + "live_hashes_in_ledger": len(live_hashes & set(self.identities)), + "training_consumed_hashes": len(consumed), + "spent_evaluation_hashes": len(spent), + "abandoned_evaluation_hashes": len(abandoned), + "currently_available_hashes": len(available), + "live_training_consumed_hashes": len(consumed & live_hashes), + "live_spent_evaluation_hashes": len(spent & live_hashes), + "live_abandoned_evaluation_hashes": len(abandoned & live_hashes), + "live_available_hashes": len(live_available), + "available_by_slice": by_slice, + "reserve_counts": self.reserve_counts(), + "row_id_collisions": len(self.row_id_collisions), + } + + def project(self) -> dict[str, Any]: + identities = [] + for digest in sorted(self.identities): + record = dict(self.identities[digest]) + record["state"] = derived_state(record) + record["slices"] = sorted(slices_of(record)) + identities.append(record) + payload = { + "schema": SCHEMA, + "policy_id": POLICY_ID, + "generation": GENERATION, + "identity_function": "hyperlexical.holdout_guard.normalized_text_sha256", + "planning_targets": dict(PLANNING_TARGETS), + "planning_target_class": "PLANNING_TARGET", + "statistical_minimum": "NOT_COMPUTABLE", + "identities": identities, + "row_id_collisions": list(self.row_id_collisions), + "n_events": len(self.events), + } + _walk_forbid(payload) + return payload + + def persist_append(self, directory: str | Path, prior_event_count: int) -> int: + """Append events after ``prior_event_count`` and refresh the projection. + + ``events.jsonl`` is append-only. ``ledger.json`` is a derived view. + """ + root = Path(directory) + events_path = root / "events.jsonl" + fresh = self.events[prior_event_count:] + if fresh: + with events_path.open("a", encoding="utf-8") as handle: + for event in fresh: + handle.write(json.dumps(event, sort_keys=True) + "\n") + (root / "ledger.json").write_text( + json.dumps(self.project(), indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return len(fresh) + + def save(self, directory: str | Path) -> None: + root = Path(directory) + root.mkdir(parents=True, exist_ok=True) + events_path = root / "events.jsonl" + if events_path.exists(): + raise SystemExit(f"REFUSE: ledger events already exist: {events_path}") + lines = [json.dumps(event, sort_keys=True) for event in self.events] + events_path.write_text("\n".join(lines) + ("\n" if lines else ""), encoding="utf-8") + (root / "ledger.json").write_text( + json.dumps(self.project(), indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + + @classmethod + def load(cls, directory: str | Path) -> "IdentityLedger": + events_path = Path(directory) / "events.jsonl" + if not events_path.is_file(): + raise SystemExit(f"REFUSE: ledger events are missing: {events_path}") + ledger = cls() + for line in events_path.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + event = json.loads(line) + kind = event.get("kind") + if kind == "observe": + ledger.observe( + event["normalized_text_sha256"], + source_artifact=event.get("source_artifact") or "", + row_ids=event.get("row_ids") or [], + labels=event.get("labels") or [], + current_source=bool(event.get("current_source")), + catalogued=bool(event.get("catalogued")), + provenance=event.get("provenance") or "", + experiment_id=event.get("experiment_id") or "", + ) + elif kind == "historical": + ledger.mark_historical( + event["normalized_text_sha256"], + event["flag"], + source_artifact=event.get("source_artifact") or "", + provenance=event.get("provenance") or "", + experiment_id=event.get("experiment_id") or "", + ) + elif kind == "transition": + ledger.transition( + event["normalized_text_sha256"], + event["to_state"], + source_artifact=event.get("source_artifact") or "", + provenance=event.get("provenance") or "", + ) + elif kind == "generation_reset_requested": + ledger.request_generation_reset( + governed_receipt=event.get("source_artifact") or "", + confirm=True, + ) + else: + raise SystemExit(f"REFUSE: unknown ledger event {kind}") + # Replay appended a second copy. Replace with the file's events. + ledger.events = [] + for line in events_path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + ledger.events.append(json.loads(line)) + return ledger + + +def _walk_forbid(value: Any) -> None: + if isinstance(value, dict): + for key, item in value.items(): + if key in _FORBIDDEN_KEYS: + raise SystemExit(f"REFUSE: ledger projection contains {key}") + _walk_forbid(item) + elif isinstance(value, list): + for item in value: + _walk_forbid(item) + + +def assert_training_disjoint_from_reserve( + rows: Sequence[Mapping[str, Any]], + ledger: IdentityLedger, +) -> dict[str, int | bool]: + """Fail closed when pinned training contains fresh reserve text. + + Spent and abandoned flags do not trip this gate. Those identities may + already sit inside the historical training export. + """ + reserved = 0 + for row in rows: + digest = normalized_text_sha256(str(row.get("text") or "")) + record = ledger.identity(digest) + if record is None: + continue + if derived_state(record) in ("EVAL_RESERVE", "EVAL_BOUND"): + reserved += 1 + if reserved: + raise SystemExit( + "ADMISSION FAIL: training input contains " + f"{reserved} EVAL_RESERVE text identities" + ) + return { + "eval_reserve_training_overlap": 0, + "eval_reserve_training_disjoint": True, + } + + +def _read_jsonl(path: Path) -> list[dict[str, Any]]: + rows = [] + with path.open(encoding="utf-8") as handle: + for line in handle: + if line.strip(): + rows.append(json.loads(line)) + return rows + + +def main(argv: Sequence[str] | None = None) -> int: + """Admission, census, and evaluation settlement. Does not train or score.""" + import argparse + import sys + + parser = argparse.ArgumentParser(description="Hyperlex evaluation-reserve ledger") + sub = parser.add_subparsers(dest="cmd", required=True) + admit = sub.add_parser("admit") + admit.add_argument("--ledger", required=True) + admit.add_argument("--rows", required=True) + admit.add_argument("--batch-id", required=True) + admit.add_argument("--source", required=True) + admit.add_argument( + "--train-export", + default="", + help="JSONL export. unbind_clean uses soft_ceiling.clean_surface on split=train.", + ) + census_cmd = sub.add_parser("census") + census_cmd.add_argument("--ledger", required=True) + census_cmd.add_argument("--live-hashes", help="Optional JSON list of live text hashes") + settle = sub.add_parser("settlement-apply") + settle.add_argument("--stream-rows", required=True) + settle.add_argument("--sheet", action="append", default=[]) + settle.add_argument("--records", default="") + settle.add_argument("--operator", required=True) + settle.add_argument("--provenance", required=True) + settle.add_argument("--batch-id", required=True) + settle.add_argument("--receipt", required=True) + settle.add_argument("--settlement-log", required=True) + settle.add_argument("--settled-at", required=True) + settle.add_argument("--stream-run-id", default="") + settle.add_argument("--activated-family", action="append", default=[]) + args = parser.parse_args(list(argv) if argv is not None else None) + if args.cmd == "settlement-apply": + from .eval_settlement import run_settlement_apply + + return run_settlement_apply(args) + if args.cmd == "census": + ledger = IdentityLedger.load(args.ledger) + live: set[str] = set() + if args.live_hashes: + payload = json.loads(Path(args.live_hashes).read_text(encoding="utf-8")) + live = set(payload) + report = ledger.census(live) + report["acquisition_gap"] = acquisition_gap(ledger.reserve_counts()) + sys.stdout.write(json.dumps(report, indent=2, sort_keys=True) + "\n") + return 0 + ledger = IdentityLedger.load(args.ledger) + prior = len(ledger.events) + rows = _read_jsonl(Path(args.rows)) + clean_hashes: set[str] = set() + if args.train_export: + from .clean_unbind import unbind_clean_hashes + + exported = _read_jsonl(Path(args.train_export)) + clean_hashes, _account = unbind_clean_hashes(rows, exported) + report = ledger.admit( + rows, + batch_id=args.batch_id, + source_artifact=args.source, + unbind_clean_hashes=clean_hashes, + ) + written = ledger.persist_append(args.ledger, prior) + report["events_appended"] = written + report["acquisition_gap"] = acquisition_gap(ledger.reserve_counts()) + _walk_forbid(report) + sys.stdout.write(json.dumps(report, indent=2, sort_keys=True) + "\n") + return 0 + + +def acquisition_gap(reserved: Mapping[str, int] | None = None) -> list[dict[str, Any]]: + """Planning gaps. Statistical minima stay ``NOT_COMPUTABLE``.""" + counts = dict(reserved or {}) + rows = [] + for key, target in PLANNING_TARGETS.items(): + have = int(counts.get(key, 0)) + rows.append( + { + "slice": key, + "currently_reserved": have, + "minimum_required": "NOT_COMPUTABLE", + "planning_target": target, + "planning_target_class": "PLANNING_TARGET", + "gap_vs_minimum": "NOT_COMPUTABLE", + "gap_vs_planning_target": max(target - have, 0), + } + ) + return rows + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/shadow/hyperlexical/loop.py b/scripts/shadow/hyperlexical/loop.py index 1e896bee..fba5fade 100644 --- a/scripts/shadow/hyperlexical/loop.py +++ b/scripts/shadow/hyperlexical/loop.py @@ -8,6 +8,8 @@ from .align import atom_token_index, offsets_from_tokenizer, pool_indices from .export import export_dataset, repo_root, write_export +from .admission import AdmissionError, admit_training_run +from .train_input import train_input_receipt from .classify_metrics import ( NONE_LABEL, SELECT_METRIC_CLASSIFY, @@ -38,7 +40,6 @@ holdout_receipt, load_holdout_spec, log_holdout, - require_holdout_for_training, ) from .provenance import provenance from .release_set import maybe_release @@ -475,6 +476,11 @@ def _offsets(tok, text: str): return None +def _enter_training_execution() -> None: + """Reached only after admission. Admission-only mode returns before this.""" + return None + + def run_loop( trunk: Path, out_dir: Path, @@ -482,9 +488,26 @@ def run_loop( include_live: bool = False, live_store: Path | None = None, ) -> dict: - holdout_spec = require_holdout_for_training() + # Same gates as preflight. Admission-only returns before any optimizer. + admission = admit_training_run( + include_live=include_live, + live_store=live_store, + export_dataset=export_dataset, + trunk=trunk, + out_dir=out_dir, + ) + if os.environ.get("HLX_ADMISSION_ONLY") == "1": + if not admission.ready: + raise AdmissionError(admission.error or "ADMISSION FAIL", admission.receipt) + return admission.receipt + bundle = admission.bundle + holdout_spec = admission.holdout_spec + input_receipt = train_input_receipt(bundle) + disjoint_receipt = admission.disjoint_receipt + reserve_receipt = admission.reserve_receipt + if bundle is None: + raise AdmissionError("ADMISSION FAIL: training bundle was not loaded", admission.receipt) root = repo_root() - bundle = export_dataset(root, include_live=include_live, live_store=live_store) release_rows_, release_stats = maybe_release(bundle["rows"]) if release_stats["release_set"]: bundle = {**bundle, "rows": release_rows_} @@ -538,6 +561,7 @@ def run_loop( selection_rows, ) + _enter_training_execution() import torch from torch import nn from torch.optim import AdamW @@ -986,8 +1010,15 @@ def score(): "epoch_metrics": epoch_metrics, "weight_file": weight_file, "aligner": "char_span + offset_mapping", - "data_sha256": bundle["sha256"], + "data_sha256": input_receipt["data_sha256"], + "training_input_mode": input_receipt["training_input_mode"], + "training_export_path": input_receipt["training_export_path"], + "training_export_sha256_expected": input_receipt["training_export_sha256_expected"], + "training_export_sha256_actual": input_receipt["training_export_sha256_actual"], + "training_export_rows": input_receipt["training_export_rows"], + "live_export_generation_enabled": input_receipt["live_export_generation_enabled"], "holdout": holdout_receipt(holdout_spec, holdout_removed), + "holdout_training_disjoint": disjoint_receipt, "classify_admission": classify_admission_receipt, "include_live": include_live, "live_included": bundle["counts"].get("live_included", 0), @@ -997,6 +1028,8 @@ def score(): "forecast_eligible": False, "note": "HF-shaped dump. Not Hyperlexical until E2.", } + if reserve_receipt is not None: + receipt["eval_reserve_disjoint"] = reserve_receipt if classify_split_receipt is not None: receipt["classify_split"] = classify_split_receipt if seed_receipt is not None: diff --git a/scripts/shadow/hyperlexical/preflight.py b/scripts/shadow/hyperlexical/preflight.py index 0547f1fb..a6d04900 100644 --- a/scripts/shadow/hyperlexical/preflight.py +++ b/scripts/shadow/hyperlexical/preflight.py @@ -1,4 +1,11 @@ -"""U3 preflight. No Hub. No train.""" +"""U3 preflight. No Hub. No train. + +``TRAINING_READY`` exists only when the scientific contract passed, +the decision rule is sealed, and ``admit_training_run`` returns +``ADMISSION_PASS``. An admission pass without a sealed decision rule +stays ``PREREGISTERED``. The trainer entrypoint with +``HLX_ADMISSION_ONLY=1`` stops after those gates. +""" from __future__ import annotations @@ -8,38 +15,87 @@ import sys from pathlib import Path -from .export import export_dataset, repo_root +from .admission import AdmissionError, admit_training_run +from .export import export_dataset TRUNK = "answerdotai/ModernBERT-base" -def main(argv=None) -> int: - root = repo_root() - allow = os.environ.get("HYPERLEX_ALLOW_TRAIN") == "1" - trunk = Path(os.environ.get("HYPERLEX_TRUNK_DIR") or "") - bundle = export_dataset(root) - report = { +def _print(report: dict) -> None: + print(json.dumps(report, indent=2, sort_keys=True)) + + +def _base() -> dict: + trunk_raw = os.environ.get("HYPERLEX_TRUNK_DIR", "").strip() + trunk = Path(trunk_raw) if trunk_raw else None + experiment_id = os.environ.get("HLX_EXPERIMENT_ID", "").strip() + return { "schema": "hyperlex.hyperlexical.preflight.v0.1", "machine": platform.machine(), "python": sys.version.split()[0], - "allow_train": allow, + "allow_train": os.environ.get("HYPERLEX_ALLOW_TRAIN") == "1", "trunk_id": TRUNK, "trunk_dir": str(trunk) if trunk else None, "trunk_dir_exists": trunk.is_dir() if trunk else False, "trunk_config": (trunk / "config.json").is_file() if trunk else False, - "data_sha256": bundle["sha256"], - "data_counts": bundle["counts"], "name_gate": False, "brier": None, - "ready_to_train": bool( - allow and trunk.is_dir() and (trunk / "config.json").is_file() + "experiment_id": experiment_id or None, + "note": ( + "TRAINING_READY requires a sealed scientific contract, a sealed " + "decision rule, and admit_training_run. Admission without a " + "decision rule stays PREREGISTERED. It is not E2 and not a " + "Hyperlexical name." ), - "note": "ready_to_train is a gate check, not E2 and not a Hyperlexical name.", } - print(json.dumps(report, indent=2, sort_keys=True)) - if not report["ready_to_train"]: + + +def _paths() -> tuple[Path | None, Path | None]: + trunk_raw = os.environ.get("HYPERLEX_TRUNK_DIR", "").strip() + out_raw = os.environ.get("HYPERLEX_TRAIN_OUT", "").strip() + trunk = Path(trunk_raw) if trunk_raw else None + out = Path(out_raw) if out_raw else None + return trunk, out + + +def main(argv=None) -> int: + del argv + trunk, out = _paths() + include_live = os.environ.get("HYPERLEX_INCLUDE_LIVE") == "1" + try: + result = admit_training_run( + include_live=include_live, + live_store=None, + export_dataset=export_dataset, + trunk=trunk, + out_dir=out, + ) + except AdmissionError as exc: + report = _base() + report.update(exc.receipt) + report["error"] = str(exc) + report["ready_to_train"] = False + report["status"] = "ADMISSION_FAIL" + report["admission_result"] = "ADMISSION_FAIL" + _print(report) + return 2 + except SystemExit as exc: + report = _base() + report["error"] = str(exc) + report["ready_to_train"] = False + report["status"] = "ADMISSION_FAIL" + report["admission_result"] = "ADMISSION_FAIL" + report["holdout_admitted"] = False + _print(report) return 2 - return 0 + report = _base() + report.update(result.receipt) + if result.contract == "CONTROLLED_RESERVE" and not result.launch_armed: + report["status"] = "NOT_READY" + report["admission_result"] = "NOT_ARMED" + report["ready_to_train"] = False + _print(report) + return 0 if result.ready else 2 if __name__ == "__main__": diff --git a/scripts/shadow/hyperlexical/train_input.py b/scripts/shadow/hyperlexical/train_input.py new file mode 100644 index 00000000..610e36b1 --- /dev/null +++ b/scripts/shadow/hyperlexical/train_input.py @@ -0,0 +1,187 @@ +"""Training-input contract. Live rebuild versus an exact pinned export. + +RUNE.TRAIN_INPUT_BIND(x) = + experiment_id_present + AND export_path_present + AND expected_export_digest_present + AND export_file_exists + AND actual_export_digest == expected_export_digest + AND trainer_source == PINNED_EXPORT + +A controlled experiment (``HLX_EXPERIMENT_ID`` set) that fails any term +stops before training and does not call ``export_dataset``. Outside a +controlled experiment, an absent pin keeps ``LIVE_BUILD``. +""" + +from __future__ import annotations + +import hashlib +import hmac +import json +import os +from pathlib import Path +from typing import Any, Callable + +LIVE_BUILD = "LIVE_BUILD" +PINNED_EXPORT = "PINNED_EXPORT" + +PATH_ENV = "HLX_TRAIN_EXPORT_PATH" +SHA_ENV = "HLX_TRAIN_EXPORT_SHA256" +ROWS_ENV = "HLX_TRAIN_EXPORT_ROWS" +EXPERIMENT_ID_ENV = "HLX_EXPERIMENT_ID" + + +class TrainInputAdmissionError(SystemExit): + """Fail closed before training. ``SystemExit`` so the CLI does not swallow it.""" + + +def _env(name: str) -> str: + return os.environ.get(name, "").strip() + + +def pinned_request() -> tuple[str, str, int | None]: + """Return ``(path, expected_sha256, expected_rows)``. Rows are optional.""" + path = _env(PATH_ENV) + digest = _env(SHA_ENV) + raw_rows = _env(ROWS_ENV) + expected_rows: int | None = None + if raw_rows: + if not raw_rows.isdigit(): + raise TrainInputAdmissionError( + f"ADMISSION FAIL: {ROWS_ENV} must be a non-negative integer" + ) + expected_rows = int(raw_rows) + return path, digest, expected_rows + + +def peek_train_input_mode() -> str: + """Env-only admission. Does not read an export or call ``export_dataset``. + + Incomplete pins fail closed. A controlled experiment with no pin fails + closed. Neither case falls through to ``LIVE_BUILD``. + """ + experiment_id = _env(EXPERIMENT_ID_ENV) + path, digest, _expected_rows = pinned_request() + if experiment_id and (not path or not digest): + raise TrainInputAdmissionError( + "ADMISSION FAIL: experiment id present but no pinned export" + ) + if path or digest: + if not path or not digest: + raise TrainInputAdmissionError( + "ADMISSION FAIL: pinned export requires " + f"{PATH_ENV} and {SHA_ENV}" + ) + return PINNED_EXPORT + return LIVE_BUILD + + +def _digest_matches(actual: str, expected: str) -> bool: + if len(actual) != len(expected): + return False + return hmac.compare_digest(actual, expected) + + +def _load_pinned(path: Path, expected: str, expected_rows: int | None) -> dict[str, Any]: + if not path.is_file(): + raise TrainInputAdmissionError(f"ADMISSION FAIL: pinned export missing: {path}") + raw = path.read_bytes() + actual = hashlib.sha256(raw).hexdigest() + if not _digest_matches(actual, expected): + raise TrainInputAdmissionError( + "ADMISSION FAIL: pinned export digest mismatch: " + f"expected {expected} actual {actual}" + ) + try: + text = raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise TrainInputAdmissionError( + "ADMISSION FAIL: pinned export is not UTF-8 JSONL" + ) from exc + rows: list[Any] = [] + for line in text.splitlines(): + if not line.strip(): + continue + try: + rows.append(json.loads(line)) + except json.JSONDecodeError as exc: + raise TrainInputAdmissionError( + "ADMISSION FAIL: pinned export is not JSONL" + ) from exc + if expected_rows is not None and len(rows) != expected_rows: + raise TrainInputAdmissionError( + "ADMISSION FAIL: pinned export row count mismatch: " + f"expected {expected_rows} actual {len(rows)}" + ) + return { + "rows": rows, + "sha256": actual, + "counts": {"n": len(rows), "live_included": 0}, + "payload": text, + "train_input": { + "mode": PINNED_EXPORT, + "path": str(path.resolve()), + "sha256_expected": expected, + "sha256_actual": actual, + "rows": len(rows), + "live_export_generation_enabled": False, + }, + } + + +def _tag_live(bundle: dict[str, Any]) -> dict[str, Any]: + rows = bundle.get("rows") or [] + digest = bundle.get("sha256") + bundle["train_input"] = { + "mode": LIVE_BUILD, + "path": None, + "sha256_expected": None, + "sha256_actual": digest, + "rows": len(rows), + "live_export_generation_enabled": True, + } + return bundle + + +def load_training_bundle( + root: Path, + *, + include_live: bool, + live_store: Path | None, + export_dataset: Callable[..., dict[str, Any]], +) -> dict[str, Any]: + """Return the bundle ``run_loop`` trains on. + + ``PINNED_EXPORT`` reads the sealed file and does not call ``export_dataset``. + ``LIVE_BUILD`` is the existing generator, including ``include_live``. + """ + mode = peek_train_input_mode() + if mode == PINNED_EXPORT: + path, digest, expected_rows = pinned_request() + return _load_pinned(Path(path), digest, expected_rows) + bundle = export_dataset(root, include_live=include_live, live_store=live_store) + return _tag_live(bundle) + + +def train_input_receipt(bundle: dict[str, Any]) -> dict[str, Any]: + """Fields copied onto the train receipt. Digest is the artifact actually loaded.""" + info = bundle.get("train_input") or {} + mode = info.get("mode") or LIVE_BUILD + actual = info.get("sha256_actual") + if actual is None: + actual = bundle.get("sha256") + rows = info.get("rows") + if rows is None: + rows = len(bundle.get("rows") or []) + generated = info.get("live_export_generation_enabled") + if generated is None: + generated = mode == LIVE_BUILD + return { + "data_sha256": actual if mode == PINNED_EXPORT else bundle.get("sha256", actual), + "training_input_mode": mode, + "training_export_path": info.get("path"), + "training_export_sha256_expected": info.get("sha256_expected"), + "training_export_sha256_actual": actual, + "training_export_rows": rows, + "live_export_generation_enabled": bool(generated), + } diff --git a/scripts/shadow/hyperlexical/unbind_settlement.py b/scripts/shadow/hyperlexical/unbind_settlement.py new file mode 100644 index 00000000..20811076 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_settlement.py @@ -0,0 +1,133 @@ +"""Unbind target settlement. Separate from classify settlement. + +Decisions are about the unbind target only. They do not set +``semantic_family``, ``attest``, or ``evaluation.enabled``. + +A correction is an appended event. Earlier events stay in the log. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Mapping, Sequence + +SCHEMA = "hyperlex.eval_unbind_settlement.v1" +RECEIPT_SCHEMA = "hyperlex.eval_unbind_settlement_receipt.v1" +DECISIONS = ("ACCEPT", "CORRECT_TARGET", "REJECT", "UNRESOLVED") +ADMIT_DECISIONS = frozenset({"ACCEPT", "CORRECT_TARGET"}) +VENDOR_CALLS = 0 + + +def refuse(message: str) -> None: + raise SystemExit(f"REFUSE: {message}") + + +def _hex64(value: Any, field: str) -> str: + text = str(value or "") + if len(text) != 64 or any(ch not in "0123456789abcdef" for ch in text): + refuse(f"{field} must be 64 lowercase hex characters") + return text + + +def _forbid_text(value: Mapping[str, Any]) -> None: + for key in value: + if key in {"text", "normalized_text", "raw_text", "fillers", "surface"}: + refuse(f"unbind settlement must not carry {key}") + + +def event_from_decision( + *, + normalized_text_sha256: str, + decision: str, + target_sha256: str, + target_provenance: str, + operator: str, + settled_at: str, + decision_basis: str, + previous_target_sha256: str = "", +) -> dict[str, Any]: + """Build one append-only settlement event. No surface text.""" + if decision not in DECISIONS: + refuse(f"unbind decision must be one of {DECISIONS}") + if not str(operator or "").strip() or not str(settled_at or "").strip(): + refuse("operator and settled_at are required") + if not str(target_provenance or "").strip(): + refuse("target provenance is required") + if not str(decision_basis or "").strip(): + refuse("decision basis is required") + digest = _hex64(normalized_text_sha256, "normalized_text_sha256") + target = _hex64(target_sha256, "target_sha256") + previous = str(previous_target_sha256 or "") + if decision == "CORRECT_TARGET": + previous = _hex64(previous, "previous_target_sha256") + if previous == target: + refuse("CORRECT_TARGET requires a different target") + elif previous: + refuse("previous_target_sha256 is only valid on CORRECT_TARGET") + body: dict[str, Any] = { + "schema": SCHEMA, + "normalized_text_sha256": digest, + "decision": decision, + "target_sha256": target, + "target_provenance": str(target_provenance), + "operator": str(operator), + "settled_at": str(settled_at), + "decision_basis": str(decision_basis), + "vendor_calls": VENDOR_CALLS, + } + if decision == "CORRECT_TARGET": + body["previous_target_sha256"] = previous + _forbid_text(body) + return body + + +def append_events(path: str | Path, events: Sequence[Mapping[str, Any]]) -> int: + """Append events. Refuse if the existing prefix changes.""" + dest = Path(path) + dest.parent.mkdir(parents=True, exist_ok=True) + prior = dest.read_bytes() if dest.exists() else b"" + encoded: list[bytes] = [] + for event in events: + if not isinstance(event, Mapping): + refuse("settlement event must be an object") + _forbid_text(event) + if event.get("schema") != SCHEMA: + refuse("settlement event schema mismatch") + encoded.append((json.dumps(dict(event), sort_keys=True) + "\n").encode("utf-8")) + with dest.open("ab") as handle: + for line in encoded: + handle.write(line) + current = dest.read_bytes() + if not current.startswith(prior): + refuse("unbind settlement log was not append-only") + return len(encoded) + + +def load_events(path: str | Path) -> list[dict[str, Any]]: + dest = Path(path) + if not dest.is_file(): + return [] + events: list[dict[str, Any]] = [] + for index, line in enumerate(dest.read_text(encoding="utf-8").splitlines(), start=1): + if not line.strip(): + continue + try: + obj = json.loads(line) + except json.JSONDecodeError: + refuse(f"settlement log line {index} is not JSON") + if not isinstance(obj, dict) or obj.get("schema") != SCHEMA: + refuse(f"settlement log line {index} has the wrong schema") + _forbid_text(obj) + events.append(obj) + return events + + +def latest_by_hash(events: Sequence[Mapping[str, Any]]) -> dict[str, dict[str, Any]]: + """Last event for each surface hash wins. Earlier lines stay in the file.""" + latest: dict[str, dict[str, Any]] = {} + for event in events: + digest = str(event.get("normalized_text_sha256") or "") + if digest: + latest[digest] = dict(event) + return latest diff --git a/specs/007-hyperlexical-model/engineering.md b/specs/007-hyperlexical-model/engineering.md index 7139f5d1..c4b48f3e 100644 --- a/specs/007-hyperlexical-model/engineering.md +++ b/specs/007-hyperlexical-model/engineering.md @@ -53,3 +53,8 @@ Any held-out manifest must be frozen and its sha256 recorded before any weights - Box: DGX Spark per `hardware.md`. - Train only after name-gate true on `milestones.md`. - E2 vs 004 probe is the T1 *artifact* name gate, separate from the dataset name-gate. + +## Evaluation reserve (2026-09-26) + +The live evaluation universe is exhausted. Doctrine, the text-identity ledger, and the admission command are `evaluation-reserve.md`. Contamination identity is `normalized_text_sha256`, not row id. SELECT-001 and SELECT-002 holdouts are `UNSCORED_ABANDONED` and are not reopened. No statistical holdout minimum is declared. The SELECT-002 shape (606 / 287 / 604 / 250) is a planning target only. Do not train on `EVAL_RESERVE`. + diff --git a/specs/007-hyperlexical-model/evaluation-reserve.md b/specs/007-hyperlexical-model/evaluation-reserve.md new file mode 100644 index 00000000..a0201568 --- /dev/null +++ b/specs/007-hyperlexical-model/evaluation-reserve.md @@ -0,0 +1,349 @@ +# Evaluation reserve (GEN-0) + +Text identity is `hyperlexical.holdout_guard.normalized_text_sha256`. +That is SHA-256 of `heldout_census.normalize_group_text`: NFKC, casefold, URL strip, non-alphanumeric to space, whitespace collapse. Empty normalized text still hashes. No second normalizer is an identity. `soft_ceiling.clean_surface` remains the unbind-clean slice predicate only. + +Row ids are historical labels. One text identity may carry many row ids. Contamination, spending, abandonment, and reserve membership attach to the text hash. + +## Exhaustion + +On 2026-09-26 the live store (`01c1b38fd6e926962e47a520228e884ae7d99ea0ad6bbe188dc9980064254ac2`, 5005 rows) had **zero** rows eligible for a fresh holdout after split=test, the pinned morph78 export (`64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`, 9150 rows), spent v2, rc1 unbind rows, and the abandoned SELECT-001 / SELECT-002 holdouts. The 440 `split=test` rows are also training-consumed under the canonical hash, so dropping the split gate does not open a fresh classify set. Classify OBSERVED and classify non-none in that eligible universe are 0. + +SELECT-001 is `CLOSED_AT_LAUNCH_GATE`. SELECT-002 is `EXECUTION_INVALID` (0 epochs, 0 gradient steps). Both holdouts are `UNSCORED_ABANDONED`: never scored, not reusable, not `SCORED_SPENT`. Manifest bytes stay as sealed. Do not reopen either experiment. + +## Lifecycle + +New text is routed **before** training exposure. Records are append-only. Flags that mean consumed, spent, or abandoned only turn on. + +```text +NEW + | + +--> TRAIN_CANDIDATE + | | + | +--> TRAIN_CONSUMED + | + +--> EVAL_RESERVE + | + +--> EVAL_BOUND + | + +--> EVAL_SPENT + | + +--> EVAL_ABANDONED +``` + +`AVAILABLE` is catalogued text with none of those flags. It may still be routed to `TRAIN_CANDIDATE` or `EVAL_RESERVE`. + +Forbidden without a governed generation reset: + +- `EVAL_RESERVE` or `EVAL_BOUND` later entering training +- training-consumed text later used as unseen evaluation +- spent or abandoned text returning to the fresh reserve + +A generation-reset **request** can be recorded. It does not clear flags and it does not create a new baseline. GEN-1 is not opened by this policy. + +## Assignment + +Policy id: `hyperlex.eval_reserve.v1`. Fixed before any candidate output is observed. Model scores, errors, and predictions are refused as admission inputs. + +Within a batch, identities are ordered by `sha256(GEN-0 | batch_id | normalized_text_sha256)`. A new identity is `EVAL_RESERVE` when it still fills at least one open required slice. Otherwise it is `TRAIN_CANDIDATE`. Quotas are unique text identities, not raw rows. A second row id on an existing hash does not increase the evaluation sample size. A new hash that reuses an old row id stays a different identity and is reported as a collision. + +Required slices, because checkpoint selection needs each of them: + +- `classify` (classification accuracy) +- `classify_observed` (OBSERVED-label accuracy) +- `classify_non_none` (`classify_macro_f1_nonnone`) +- `unbind_clean` (unbind clean exact) + +## Coverage classes + +| Claim | Class | +| --- | --- | +| Canonical text hash is the contamination identity | CANONICAL_REQUIREMENT | +| Fresh evaluation text is disjoint from training, spent, and abandoned text | CANONICAL_REQUIREMENT | +| Each required slice is represented before SELECT-003 can be drafted | CANONICAL_REQUIREMENT | +| Statistical minimum n for those metrics | NOT_COMPUTABLE | +| v2 `DRAW_READY_MIN` 470 | HISTORICAL_PRECEDENT (unbind census target; PR #113 sets no v2 size) | +| Name-gate 2500 | CANONICAL_REQUIREMENT for the name gate, not for this holdout | +| rc1 classify_clean 444 / OBSERVED 72 / unbind_clean 306 | HISTORICAL_PRECEDENT (spent test) | +| SELECT-002 eligibility 606 classify / 287 OBSERVED / 604 non-none / 250 clean unbind | HISTORICAL_PRECEDENT and the PLANNING_TARGET | + +The planning target restores that SELECT-002 shape. It is not a validity theorem. Gap versus a statistical minimum is `NOT_COMPUTABLE`. + +## Admission + +```sh +PYTHONPATH=scripts/shadow python -m hyperlexical.identity_ledger admit \ + --ledger /path/to/ledger --rows new.jsonl --batch-id BATCH --source acquire +``` + +A row is accepted only when its canonical hash is absent from `TRAIN_CONSUMED`, `EVAL_SPENT`, `EVAL_ABANDONED`, and the existing reserve. The command reports raw rows and unique canonical text identities. Reserve rows must not be used for training, checkpoint selection, hyperparameter tuning, candidate-specific error review, or repeated scoring. Aggregate census counts are allowed. When `HLX_EVAL_RESERVE_LEDGER` is set, the trainer refuses if a loaded training row is still `EVAL_RESERVE` or `EVAL_BOUND`. + +## SELECT-003 + +Do not draft SELECT-003 until a new reserve exists, the four slices are represented, the reserve is text-disjoint from the pinned training export, the reserve is not spent or abandoned, and this policy stays sealed. The unresolved hypothesis, if a later authorization allows it, is still `HLX_SELECT_METRIC`: `unbind_exact` versus `classify_macro_f1_nonnone`. Authorization is a separate act. Meeting a planning target does not grant it. + +## GEN-0 and GEN-1 + +GEN-0 keeps `seed-morph78` and the pinned 9150-row export as the training baseline. New material is evaluation-only until the reserve slices are filled. Overflow may become `TRAIN_CANDIDATE` and must not be pulled back into the reserve. + +GEN-0 becomes impractical only if, after a real acquisition attempt: + +- new classify evidence cannot represent OBSERVED and non-none together +- new text keeps colliding with the pinned export or with spent/abandoned hashes +- accepted training candidates grow while the reserve slices stay empty +- the reserve cannot hold the four slices at once +- the historical baseline is no longer the system under test + +Convenience is not one of those conditions. This document does not create GEN-1. + +## Ledger + +`scripts/shadow/hyperlexical/build_identity_ledger.py` catalogues receipt-backed artifacts. It does not infer lifecycle from filenames. The populated ledger stays on the host store, not in git, and does not contain raw text. + +## Acquisition batch HLX-EVAL-ACQ-2026-09-26-001 + +Screened local rights-cleared and operator-labeled corpora. The ledger was not mutated. Decision `REJECT_BATCH`. The reserve stays empty. + +- 2026 backfill atoms: 67/67 unique hashes already `TRAIN_CONSUMED`. +- Harvest multiword file: 890/890 unique hashes already `TRAIN_CONSUMED`. +- Civilian seed: 17 novel identities, all lineage `ai-native`. A registry export marked `OBSERVED` is not an operator settlement, so those rows were held. Class imbalance is severe. No balance threshold is declared. +- 2026-09-16 structure gold: the authorized train rows are already consumed. Four novel leftovers carry model predictions and were excluded. +- Agent-memetics seeds: rights are unresolved and task labels are absent. + +New `OBSERVED` coverage is `NOT_COMPUTABLE` until an operator harvest settlement exists. This record does not settle phrases. SELECT-003 stays undrafted. GEN-1 was not created. In-repo settled gold is exhausted. That alone does not reset GEN-0. + +## Acquisition batch HLX-EVAL-ACQ-2026-09-26-002 + +The held-out stream (`hs-20260925T211358Z`, 339 rows) is the shelf that sits beside the Jev lane. Every canonical text hash is absent from the identity ledger. Overlap with the box Jev exposure list is 0. The ledger was not mutated. Decision: not admitted. The reserve stays empty. + +Jev in this batch is the exposure fence. `hs_run.py` and each row policy forbid evaluating Jev, the lineage rule, or any other model on these rows. `JEV_API_KEY` is unset on this host. No vendor call was made. A Jev family call is not an operator settlement. + +- Rights-cleared INFERRED labels: gaming-meta 99, betting-sharp 61, crypto-degen 5, plus 40 encyclopedic `none`. Five families have no independent label. +- 134 rows are `UNLABELLED`. The attest column is empty on all 339 rows, so new `OBSERVED` coverage is `NOT_COMPUTABLE`. +- 16 Reddit and Know Your Meme rows have unresolved rights. +- Class imbalance on the rights-cleared non-none subset is severe. No balance threshold is declared. Admitting it would open `classify_macro_f1_nonnone` on three families. + +SELECT-003 stays undrafted. GEN-1 was not created. Novel yield against `TRAIN_CONSUMED` is 339/339, so GEN-0 is not collision-blocked. The next label step is operator `attest-apply` on the queued sheet. + +## Taxonomy proposal HLX-EVAL-TAXON-2026-09-26-001 + +The next label step is no longer `attest-apply`. The operator directed a taxonomy expansion first. Draft: `label-taxonomy-proposal.md`. Private remap receipt has no row text. The ledger was not mutated. Nothing was admitted. Jev was not called. `attest-apply` was not run. + +The production head stays the nine-way `layout.FAMILIES` list. The proposal adds sixteen non-none names as a draft active set, with `attest`, `register`, and `function` as separate surfaces. `evaluation.enabled` is false on every name. Support minimum is `NOT_COMPUTABLE`. + +Of the 339 stream rows, 204 keep the same family as a `PROPOSED_REMAP` (gaming-meta 98, betting-sharp 61, crypto-degen 5, none 40). 135 are `LABEL_UNRESOLVED`. `brainrot-aura` is not split. Kinship hints are not mapped to `relationship-dating`. Eleven of the sixteen names have no row on this shelf. Row settlement is not ready. SELECT-003 stays undrafted. + +## Taxonomy acceptance HLX-EVAL-TAXON-2026-09-26-002 + +The operator accepted the ontology structure and amended it. Active non-none families are eighteen, including `identity-affiliation` and `politics-civic`. `work-hustle` is renamed `workplace-career`. `brainrot-aura` is not a family. Six names stay candidates. `taxonomy.active` is true and `evaluation.enabled` is false on all eighteen. `none` stays abstain. `source_hint` is evidence, not `semantic_family`. + +Lanes are prepared and unsettled: A 204, B 65, C 21, D 16. Thirty-three Wiktionary hint-only rows sit outside those lanes and stay `LABEL_UNRESOLVED`. Decision cells are empty. `attest-apply` was not run. The current command would force `OBSERVED` and would reject the new names, so it must not be used on these sheets. Vendor calls: 0. Reserve stays 0. SELECT-003 stays undrafted. + + +## Settlement tool HLX-EVAL-SETTLE-2026-09-26-001 + +Evaluation settlement is a separate command, `python -m hyperlexical.identity_ledger settlement-apply`. Production `attest-apply` was not modified and was not run. That command still accepts only the production families plus `none` or `reject`, and it still writes `label_source=OBSERVED` for an accepted value. + +`settlement-apply` reads completed operator cells. It does not fill them. `source_hint`, `semantic_family`, and `attest` are separate fields. `ACCEPT` stores the attest the operator entered and does not promote existing evidence to `OBSERVED`. `RECLASSIFY` requires an explicit family that differs from the proposed evidence. `NONE` stores `semantic_family=none` and is not `reject`. `UNRESOLVED` stores null family and null attest. A second decision for the same row is refused. The log and the receipt are append-only and contain no row text. + +`taxonomy.active`, `evaluation.enabled`, and `production.enabled` are three flags. The eighteen families stay taxonomy-active only. Both enable flags stay false. `layout.FAMILIES` is unchanged. Lanes A–D and the hint-only holding sheet were validated with every decision cell empty (204 / 65 / 21 / 16 / 33). Rows settled: 0. Unresolved decisions: 0. Reserve added: 0. Vendor calls: 0. The identity ledger was not mutated. SELECT-003 stays undrafted. + + +## Operator settlement HLX-EVAL-SETTLE-2026-09-26-002 + +The five private sheets for stream `hs-20260925T211358Z` were filled with an explicit decision on every row, then applied with `python -m hyperlexical.identity_ledger settlement-apply`. Production `attest-apply` was not run. `layout.FAMILIES` was not edited. No row text is in this file. + +Blank and `UNRESOLVED` stay distinct. Blank means the operator has not reviewed the row. `UNRESOLVED` means the operator reviewed the row and declined to settle it. This pass left blank at 0 and `UNRESOLVED` at 84. + +Counts: settled 255 (`ACCEPT` 204, `RECLASSIFY` 32, `NONE` 19), `UNRESOLVED` 84, blank 0. `OBSERVED` 128. `INFERRED` 127. Rights-blocked settled 14. Those 14 stay out of `EVAL_RESERVE`. Rights status was not changed by the semantic decision. + +`OBSERVED` was used only when a stored gloss or the row text directly states the settled family, the atom is slangish or multiword jargon, and the primary stored sense is that family. A category, a topic, or `source_hint` did not set family or attest. Accepting a family did not promote an existing `INFERRED` label. Encyclopedic `none` rows stayed `INFERRED`. + +Settled support by unique canonical text, including zeros: gaming-meta 60 (OBSERVED 38, INFERRED 22, cleared 60, blocked 0); betting-sharp 58 (41, 17, 53, 5); crypto-degen 4 (3, 1, 3, 1); internet-slang 19 (10, 9, 19, 0); memetic 3 (0, 3, 2, 1); social-status 6 (3, 3, 6, 0); relationship-dating 0; approval-disapproval 1 (0, 1, 1, 0); conflict-aggression 0; technology-ai 4 (1, 3, 4, 0); workplace-career 10 (9, 1, 10, 0); sports-competition 1 (0, 1, 0, 1); music-entertainment 1 (0, 1, 1, 0); fashion-aesthetic 0; regional-cultural 0; spiritual-mystic 0; identity-affiliation 16 (13, 3, 16, 0); politics-civic 13 (10, 3, 13, 0); none 59 (0, 59, 53, 6). + +Rights-cleared settled non-none identities: 188. Quantitative support thresholds are `NOT_COMPUTABLE`, so no `UNREPRESENTED` / `LOW_SUPPORT` / `REPRESENTED` flag is assigned. `OBSERVED` total 128, all non-none; `OBSERVED` none is 0. + +One workplace-lane row describes retail product reformulation and price-tier inflation. `finance-retail` remains a candidate and was not activated. That row is `UNRESOLVED`. + +Admission used `identity_ledger admit` under `hyperlex.eval_reserve.v1`, batch `HLX-EVAL-ADMIT-2026-09-26-002`. Admitted to `EVAL_RESERVE`: 241. Rejected existing identities: 0. Slice counts after admission: classify 241, classify_observed 123, classify_non_none 188, unbind_clean 0. `clean_unbind_support` is 0. This stream has no unbind target and none was fabricated. SELECT-003 was not drafted. The gate reports `eligible: false` because `unbind_clean` is unrepresented. Reserve identities are text-disjoint from the pinned train export. + +Receipt `HLX-EVAL-SETTLE-2026-09-26-002` sha256 `1b0fb54a5de0fb708f74ad278037458c15e0abea5a7eb6b4b151b7e7fb34567e`. `settlement-apply` still records `reserve_added: 0`; admission is the separate ledger command. Vendor calls: 0. BEST was not moved. Do not train. + + +## Clean-unbind contract — 2026-09-26 + +Phase: clean-unbind capacity recovery. The 241 classify reserve identities stay protected. This section does not retune `semantic_family`, `attest`, or `evaluation.enabled`. + +Executed shape: `hyperlexical.export._unbind_dual_scheme_rows`. Fillers are the source atom's own tokens. Positional text is those tokens joined by spaces, with roles `pos_0..pos_n-1`. Type-slot text is `TOKEN:` / `SLOT:` / `MARKER:` prefixed by index, and the fillers stay the same tokens. Role schemes are only `positional` and `type_slot`. Token count is 2 through `LIVE_UNBIND_MAX_TOKENS` (6). Atom length is at most `LIVE_UNBIND_MAX_LEN` (80). A gloss is not unbind gold. `class` is copied, not invented; this acquisition uses `INFERRED` and `lineage=none`. + +Contamination identity: `hyperlexical.holdout_guard.normalized_text_sha256` (NFKC, casefold, URL and punctuation stripped, SHA-256). + +Clean predicate: `soft_ceiling.clean_surface` on rows whose `split` is `train`, which is the call in `holdout_eligibility.census`. Definition string: `unbind_clean_definition = soft_ceiling.clean_surface`. The clean predicate is whitespace-collapsed lowercase equality of the row text against train text. It is not the contamination hash. `oov_filler_surface` is a different surface and is not this predicate. + +Scorer, not run in this phase: `hyperlexical.eval_forward.score_unbind_exact`. Gold is the filler list. Empty filler lists are skipped. Metrics are `unbind_exact`, `unbind_token_f1`, `unbind_slot_f1`, and the strict variants, including `by_role_scheme`. Baseline name: `unbind_copy_token`. + +Synthetic row, placeholders only: + +```json +{ + "text": "EXAMPLE_TOKEN_A EXAMPLE_TOKEN_B", + "split": "eval", + "lineage": "none", + "typology": [], + "stage": "noise", + "roles": ["pos_0", "pos_1"], + "fillers": ["EXAMPLE_TOKEN_A", "EXAMPLE_TOKEN_B"], + "role_scheme": "positional", + "task": "unbind", + "provenance": "source:EXAMPLE_LOCATOR", + "class": "INFERRED", + "license": "EXAMPLE_RIGHTS_GRANT", + "target_origin": "source_lemma_tokens" +} +``` + +The type-slot twin uses text `TOKEN:EXAMPLE_TOKEN_A SLOT:EXAMPLE_TOKEN_B`, roles `TOKEN` and `SLOT`, and the same fillers. Unbind settlement decisions are `ACCEPT`, `CORRECT_TARGET`, `REJECT`, and `UNRESOLVED` in `hyperlex.eval_unbind_settlement.v1`. That log is not the classify settlement schema. Admission still goes through `IdentityLedger.admit`. `unbind_clean` is set only for hashes kept by `clean_surface`. + + + +## Clean-unbind admission HLX-EVAL-ADMIT-2026-09-26-003 + +Source: Princeton WordNet 3.0 index lemmas (`index.noun`, `index.verb`, `index.adj`, `index.adv`). Glosses in `data.*` were not read and were not used as targets. Raw artifact sha256 `cbda5ea6eef7f36a97a43d4a75f85e07fccbb4f23657d27b4ccbc93e2646ab59`. License file sha256 `7731175a77952e259390b496fab905e57118b8d19ad3a8383c67eee724ff443f`. Rights: WordNet 3.0 Copyright 2006 by Princeton University, with permission to use, copy, modify, and distribute for any purpose without fee or royalty when the notice is preserved. Unresolved-rights rows were not in this source. + +Fillers are the source lemma tokens (`target_origin=source_lemma_tokens`). Operator settlement `HLX-EVAL-UNBIND-SETTLE-2026-09-26-001` appended 250 `ACCEPT` events on that basis. `CORRECT_TARGET` 0. `REJECT` 0. `UNRESOLVED` 0. Settlement receipt sha256 `3ada2dae58bde22ef1ed1ac5a4004be75d7bf3f52cac590f24900de71015194b`. The classify settlement log was not rewritten. + +Screen of shaped dual-scheme rows: 128233 unique texts. Novel and clean: 128044. Rejected `TRAIN_CONSUMED` 184. Rejected existing `EVAL_RESERVE` 5. Those 5 were not reused. Novelty rate among shaped rows: 0.9985. Planning cap admitted 250 of the admissible set. `IdentityLedger.admit` appended 500 events. Events sha256 before `f5e0008f27a80b11bc7e5b98e48e9e99cada04ee8f9455ed5ece6f99c1de3266`, after `8223ae11bb42bd1a98ebcd739d1cfbc470085e241826b662703faefdfe752da6`. Routed to `TRAIN_CANDIDATE`: 0. Acquisition receipt sha256 `be5671d4cf586b7a9ce3f45b4f5b8b5d0574edd4d1eaeef0c9644ca8ad2678a8`. Admission receipt sha256 `53df5397a13974e03bd60310fca2c29589e7a0fa6236dd576cf4ddf43a75bf15`. Census receipt sha256 `a5e9ae8ef6b65b5c187633e09b8700a5ef800eb1d97bd7a78aa2a9db26cf0a16`. + +Reserve after admission: classify 241, classify_observed 123, classify_non_none 188, unbind_clean 250. The first three did not decrease. Admitted rows are `class=INFERRED`, `lineage=none`. Role schemes: positional 125, type_slot 125. Filler counts: 2 tokens 64, 3 tokens 56, 4 tokens 56, 5 tokens 46, 6 tokens 28. Source-index metadata, not a Hyperlex family: noun 72, verb 68, adv 62, adj 48. Unique filler targets: 125. Each target has 2 surfaces (the two role schemes). Maximum surfaces per target: 2. + +Planning progress, not a validity threshold: classify 241/606, OBSERVED 123/287, non-none 188/604, clean unbind 250/250. Statistical minimum remains `NOT_COMPUTABLE`. + +`select_003_gate.eligible` is true. `training_overlap_identities` is 0. `spent_or_abandoned_in_reserve` is 0. Reserve identities 491. SELECT-003 was not drafted. This is representation completeness, not training readiness. Evaluation quality is still one lexicon, `INFERRED`, `lineage=none`. These rights-cleared active families still have zero settled classify support and were not collected here: relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, spiritual-mystic. + +Vendor calls: 0. BEST was not moved. Do not train. + + +## SELECT-003 preregistration HLX-EXP-2026-09-26-SELECT-003 + +Phase: preregistration only. Experiment id `HLX-EXP-2026-09-26-SELECT-003`. Training is not authorized. BEST is not moved. No holdout is scored. Vendor calls: 0. + +Predecessors are not evidence for or against the hypothesis. SELECT-001 closed at the launch gate with the hypothesis `UNTESTED`. SELECT-002 is `EXECUTION_INVALID` with epochs 0 and gradient steps 0, hypothesis `UNTESTED`. + +Hypothesis: with training otherwise equivalent to the reconstructed `seed-morph78` baseline, does selecting checkpoints by `classify_macro_f1_nonnone` improve non-none classification macro-F1 while preserving the established unbind and classification safeguards? + +The single scientific variable is `HLX_SELECT_METRIC`. Baseline: unset, which resolves to `unbind_exact`, with `HYPERLEX_SAVE_BEST_UNBIND=1`. Candidate: `classify_macro_f1_nonnone`. Frozen with the reconstructed recipe: `HYPERLEX_FILLER_FILTER=off`, `HYPERLEX_TASK_ROUTING=legacy_split`, `HYPERLEX_UNBIND_LOSS_WEIGHT=1.0`, init `seed-morph65`, `HLX_SEED` unset, `HLX_E2_DISJOINT` absent, `HLX_VOCAB_TRAIN_ONLY` absent, `HLX_ALLOW_NO_HOLDOUT` unset, `HYPERLEX_RELEASE_SET` absent. Pinned-export execution is infrastructure, not a scientific variable. Checkpoint selection inside the trainer still uses the pinned export's val split. The reserve is the comparison surface for the decision rule. Substituting the reserve for that val split is a second variable and is not part of this experiment. + +Training input: pinned export, 9150 rows, sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`. Rows before the reserve filter 9150, after 9150. Reserve/training row-id overlap 0. Reserve/training text-hash overlap 0. Any difference is an admission failure. + +BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, weights sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. Trunk weights sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. `layout.FAMILIES` was not edited. + +The bound GEN-0 reserve is unchanged: classify 241, classify_observed 123, classify_non_none 188, unbind_clean 250, identities 491. Events sha256 `8223ae11bb42bd1a98ebcd739d1cfbc470085e241826b662703faefdfe752da6`. Ledger projection sha256 `d071b7aec8154203ce7f9ae9531639b8d638f86c2ac0c3af38bead9b3c4a48f9`. Spent 0. Abandoned 0. Text-disjoint from training. Rights-cleared. No identities are added or removed under this experiment id. + +Receipts bound with the reserve: classification settlement `HLX-EVAL-SETTLE-2026-09-26-002` canonical sha256 `1b0fb54a5de0fb708f74ad278037458c15e0abea5a7eb6b4b151b7e7fb34567e`; classification admission `HLX-EVAL-ADMIT-2026-09-26-002` sha256 `9898d13140b1adf9e496ce4a7e71f9573d3a0f87e97355ea01de3ecafb66e836`; clean-unbind acquisition `be5671d4cf586b7a9ce3f45b4f5b8b5d0574edd4d1eaeef0c9644ca8ad2678a8`; clean-unbind settlement `3ada2dae58bde22ef1ed1ac5a4004be75d7bf3f52cac590f24900de71015194b`; clean-unbind admission `53df5397a13974e03bd60310fca2c29589e7a0fa6236dd576cf4ddf43a75bf15`; census `a5e9ae8ef6b65b5c187633e09b8700a5ef800eb1d97bd7a78aa2a9db26cf0a16`. + +Primary metric: + +```text +name: classify_macro_f1_nonnone +label_universe_sha256: 227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175 +absent_class_policy: omit_when_gold_support_is_zero +scorer: hyperlexical.classify_metrics.macro_f1_nonnone +``` + +The sealed class set is the non-none lineages present on the bound reserve, with identity support: approval-disapproval 1, betting-sharp 53, crypto-degen 3, gaming-meta 60, identity-affiliation 16, internet-slang 19, memetic 2, music-entertainment 1, politics-civic 13, social-status 6, technology-ai 4, workplace-career 10. Sum 188. The scorer's macro is the unweighted mean of per-class F1 over gold labels other than `none` that have support n>0. Classes with gold support 0 are omitted. They are not entered as F1=0. A later taxonomy expansion does not enter this experiment's metric. The six families with zero rights-cleared settled support stay outside the universe: relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, spiritual-mystic. + +Slice mapping, one contamination function for every slice (`normalized_text_sha256`): + +- `classify_macro_f1_nonnone` uses the 188 `classify_non_none` identities. Gold is `lineage`. The 53 `none` identities are not in this row set. +- Classification accuracy uses the 241 `classify` identities. Gold is `lineage`, including `none`. Scorer: `accuracy`. +- OBSERVED-label accuracy uses the 123 `classify_observed` identities. Gold is `lineage`. Scorer: `accuracy`. These three slices share classification settlement `HLX-EVAL-SETTLE-2026-09-26-002` and admission `HLX-EVAL-ADMIT-2026-09-26-002`. +- Unbind clean exact uses the 250 `unbind_clean` identities. Gold is the filler list. Scorer: `score_unbind_exact` (`unbind_exact`). Settlement is `HLX-EVAL-UNBIND-SETTLE-2026-09-26-001`, not the classify log. Clean predicate remains `soft_ceiling.clean_surface`. + +Decision thresholds are `BLOCKED_PENDING_OPERATOR_AUTHORIZATION`. Inherited authorization from SELECT-002 is `NOT_COMPUTABLE`. No canonical rule carries numeric margins across an execution-invalid predecessor, and SELECT-002 sealed its margins for that experiment id only. Preservation names are prepared and inactive: unbind clean exact, classification accuracy, OBSERVED-label accuracy. Promote and reject thresholds are not set. + +Representation completeness passes. Training provenance passes. Holdout disjointness passes. Rights and provenance pass. Clean-unbind capacity passes. Evaluation quality is limited and disclosed: non-none support is 188 and imbalanced, and the 250 clean-unbind rows are Princeton WordNet 3.0 only, all `INFERRED`, `lineage=none`, 125 filler targets, 2 surfaces per target. The experiment is not authorized to run. + +## SELECT-003 threshold authorization — 2026-09-26 + +Operator decision for `HLX-EXP-2026-09-26-SELECT-003` only. This is a new authorization. It does not transfer SELECT-001 or SELECT-002 margins. The sealed preregistration file is not edited. Training is not authorized. BEST is not moved. The reserve is not scored. + +Primary metric remains `classify_macro_f1_nonnone`. Label universe sha256 `227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175`. Absent-class policy `omit_when_gold_support_is_zero`. The universe stays the twelve supported non-none classes on the sealed reserve. + +Comparison baseline is `hyperlex-encoder-modernbert-base-seed-morph78`, weights sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`, scored later on the same GEN-0 reserve: classify 241, classify_observed 123, classify_non_none 188, unbind_clean 250. + +`PROMOTE` requires every condition. Primary: candidate `classify_macro_f1_nonnone` strictly greater than seed-morph78. Equality does not promote. Unbind clean exact on the 250 sealed identities stays within 0.01 below seed-morph78. Classification accuracy on the 241 sealed identities stays within 0.02. OBSERVED-label accuracy on the 123 sealed identities stays within 0.05. Integrity must also pass: one scientific variable, pinned export exact, 9150 rows before and after the reserve filter, row-id overlap 0, canonical text-hash overlap 0, no reserve identity spent or abandoned before scoring, sealed ledger unchanged, sealed class universe unchanged, checkpoint selection follows `HLX_SELECT_METRIC`, complete provenance, no contamination-guard failure, no schema or name-gate failure, and no `CHAR_WINS`. + +`REJECT` if the candidate primary metric is lower, or the unbind delta is below -0.01, or classification accuracy delta is below -0.02, or OBSERVED-label accuracy delta is below -0.05, or any hard failure occurs: `CHAR_WINS`, more than one scientific variable, training input other than the sealed pin, effective training rows other than 9150, reserve binding change, class-universe change, unverifiable reserve or execution provenance, or a hard integrity or contamination guard failure. + +`INCONCLUSIVE` if the primary metrics are equal and the preservation and integrity guards pass. Also `INCONCLUSIVE` when execution and scoring are valid but the sealed rule cannot be applied deterministically, and that reason is not itself a hard integrity failure. An inconclusive result is not resolved by changing thresholds. + +These floors are new SELECT-003 rules. A later promote would mean checkpoint selection by non-none macro-F1 improved the sealed twelve supported families without exceeding the three allowed regressions. It would not mean improvement across the eighteen-family ontology. Families outside the gold universe remain relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, and spiritual-mystic. Representation completeness passes. Evaluation quality stays limited. Vendor calls: 0. Do not train. + + +## SELECT-003 execution HLX-EXP-2026-09-26-SELECT-003 + +One launch was authorized from Spark commit `53f128a68a6603a98d9d5a3e56cc357f4a17aa0c` with a clean tree. The sealed preregistration, threshold authorization, candidate environment, and non-launching preflight were not edited. `HYPERLEX_ALLOW_TRAIN=1` was a process overlay only. `HLX_ALLOW_NO_HOLDOUT` stayed unset. + +Container `hlx-train-select003-1790441469` on `lmsysorg/sglang:dev-qwen38-27b-dflash2` (`sha256:616a3e97f45191af975896cfa644279096cb31bd408a071c2e99ca7209c3cafe`) started 2026-09-26T16:51:09Z and exited 1 at 2026-09-26T16:51:12Z. The trainer refused before `load_training_bundle`: `HYPERLEX_ALLOW_TRAIN=1` with no holdout manifest. Epochs 0. Gradient steps 0. No candidate checkpoint. The pinned export was not consumed. The reserve was not scored and was not spent. Decision `EXECUTION_INVALID`. The hypothesis is `UNTESTED`. This is not `PROMOTE`, `REJECT`, or `INCONCLUSIVE`. + +BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. Vendor calls: 0. Do not retry this experiment id. Do not train. +## SELECT-003 closed — PREFLIGHT_LAUNCH_HOLDOUT_GATE_MISMATCH + +`HLX-EXP-2026-09-26-SELECT-003` is permanently `EXECUTION_INVALID`. Cause: `PREFLIGHT_LAUNCH_HOLDOUT_GATE_MISMATCH`. The non-launching preflight reported `TRAINING_READY` without running `require_holdout_for_training` under `HYPERLEX_ALLOW_TRAIN=1`. The trainer then refused before `load_training_bundle` because the sealed candidate had no `HLX_HOLDOUT_MANIFESTS` and `HLX_ALLOW_NO_HOLDOUT` stayed unset. Epochs 0. Gradient steps 0. Candidate checkpoint: none. Reserve scored: false. `BEST_moved`: false. Hypothesis: `UNTESTED`. The sealed preregistration and threshold authorization are not rewritten. This id is not retried. + +## Controlled-experiment admission — CONTROLLED_RESERVE + +The canonical controlled-experiment holdout contract is `CONTROLLED_RESERVE`. A launch with `HLX_EXPERIMENT_ID` set is admitted by `admit_training_run`, which both preflight and `run_loop` call, in this order: experiment binding, launch gate, sealed reserve, pinned training input, train/reserve disjointness, single scientific variable, BEST and trunk digests, output directory, then ready. + +A sealed reserve binding satisfies the holdout requirement when the ledger is `EVAL_RESERVE`, all four slices are present, the binding digest matches `events.jsonl`, and training overlap by row id and by normalized text hash is 0. Overlap rejects the run. It does not drop training rows. `HLX_HOLDOUT_MANIFESTS` is not a second requirement. `HLX_ALLOW_NO_HOLDOUT` does not admit a controlled experiment. Legacy launches that are not controlled experiments still use the manifest gate. + +`HYPERLEX_ALLOW_TRAIN=1` is part of the effective environment hash. `HLX_ADMISSION_ONLY=1` is not. With that flag, `python -m hyperlexical.train --run` returns after admission and does not construct an optimizer, enter epoch 0, or take a gradient step. `TRAINING_READY` means that same admission returned `ADMISSION_PASS`. + +```text +RUNE.PREFLIGHT_LAUNCH_PARITY(x) = + effective_environment_hash(preflight) == effective_environment_hash(launch) + AND admission_gate_sequence(preflight) == admission_gate_sequence(launch) + AND admission_result(preflight) == admission_result(launch) +``` + +## SELECT-004 readiness + +`HLX-EXP-2026-09-26-SELECT-004` is not preregistered and is not authorized to run. It may be drafted. SELECT-001, SELECT-002, and SELECT-003 produced no scientific evidence about `HLX_SELECT_METRIC`. A draft may test the same hypothesis: baseline unset, which resolves to `unbind_exact` with `HYPERLEX_SAVE_BEST_UNBIND=1`; candidate `classify_macro_f1_nonnone`. + +No canonical rule carries numeric thresholds across experiments. SELECT-003's floors do not transfer. SELECT-004 decision thresholds are `BLOCKED_PENDING_OPERATOR_AUTHORIZATION` until a fresh authorization is sealed for that id. + +The draft must keep the pinned export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`, 9150 rows, BEST sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`, and trunk sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. The GEN-0 reserve stays classify 241, classify_observed 123, classify_non_none 188, unbind_clean 250, identities 491. It is not scored and its lifecycle is not changed. Admission uses that reserve. It does not draw another one and it does not attach a legacy manifest. + +Vendor calls: 0. BEST was not moved. Do not train. +## SELECT-004 preregistered — HLX-EXP-2026-09-26-SELECT-004 + +`HLX-EXP-2026-09-26-SELECT-004` is `PREREGISTERED`. Preregistration sha256 `365f7ce7141e8ee899fc52d85584a32c21ecfcb596e316b1438463bd8db85c4a`. That seal records repository commit `493c75981bf443baf2899b71b504fe7c7a529cf9` and a clean tree. Later documentation does not edit the sealed file. + +Predecessors are not evidence. SELECT-001 remains `CLOSED_AT_LAUNCH_GATE`. SELECT-002 remains `EXECUTION_INVALID` with epochs 0 and gradient steps 0. SELECT-003 remains `EXECUTION_INVALID`, cause `PREFLIGHT_LAUNCH_HOLDOUT_GATE_MISMATCH`, epochs 0, gradient steps 0, candidate checkpoint none, reserve not scored. The hypothesis is `UNTESTED`. + +The single scientific variable is `HLX_SELECT_METRIC`. Baseline is unset, which resolves to `unbind_exact` with `HYPERLEX_SAVE_BEST_UNBIND=1`. Candidate is `classify_macro_f1_nonnone`. Experiment diff sha256 `aa6e42e50a9547908edfe4860371a2ec05f7ec20ea87dfa829555d9e7b05bc4c`. `experimental_variable_count` is 1. Metadata differences are the experiment id, the output directory, and the residual-dump path. + +`CONTROLLED_RESERVE` binding sha256 `c16e69559dd6582f687532ce6e2a2e9b52a70db1aa066d11e7e68e723936b371`. Ledger events sha256 `8223ae11bb42bd1a98ebcd739d1cfbc470085e241826b662703faefdfe752da6`. Projection sha256 `d071b7aec8154203ce7f9ae9531639b8d638f86c2ac0c3af38bead9b3c4a48f9`. Counts remain classify 241, classify_observed 123, classify_non_none 188, unbind_clean 250, identities 491, lifecycle `EVAL_RESERVE`. Training row-id overlap is 0. Training canonical-text overlap is 0. The reserve was not scored, spent, abandoned, or relabeled. Historical spent and abandoned identities in the same ledger are not this reserve. + +Training input is the pinned export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`, mode `PINNED_EXPORT`. Declared rows, consumed rows, and effective optimization rows are 9150. No runtime exclusion reduced the set. `HLX_ALLOW_NO_HOLDOUT` stayed unset. No legacy holdout manifest was attached. + +BEST remains `hyperlex-encoder-modernbert-base-seed-morph78`, sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. Trunk sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. BEST was not moved. + +Decision thresholds are `BLOCKED_PENDING_OPERATOR_AUTHORIZATION`. SELECT-003 numeric floors do not transfer. The prepared preservation metric names are `unbind_clean_exact`, `classification_accuracy`, and `observed_label_accuracy`. Those names have no active numeric thresholds. + +Admission-only parity passed. Preflight and the trainer entrypoint, both under `HYPERLEX_ALLOW_TRAIN=1` with `HLX_ADMISSION_ONLY=1`, share environment hash `759c591151c88074973c3ba2a555be551d3460c04c8cedf59ab36624de215885`. `admission_result` is `ADMISSION_PASS`. Status is `PREREGISTERED`. `optimizer_loaded` is false. Epochs 0. Gradient steps 0. The SELECT-004 output directory was absent before and after. `HLX_ADMISSION_ONLY` is excluded from the environment hash. + +`TRAINING_READY` exists only when all three are true: the scientific contract is sealed, the decision rule is sealed, and real entrypoint admission passes. SELECT-004 stays below `TRAINING_READY` until a fresh threshold authorization for this experiment id. `training_launch_authorized` is false. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. Classification support is 188 rights-cleared non-none identities. Clean unbind is 250 identities, 125 targets, 2 role-scheme surfaces per target, from Princeton WordNet 3.0 only. Unsupported families remain relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, and spiritual-mystic. A later result must not claim those families. Label universe sha256 `227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175`. Absent-class policy `omit_when_gold_support_is_zero`. + +Vendor calls: 0. Do not train. The next decision is a fresh SELECT-004 threshold authorization. +## SELECT-004 thresholds authorized — HLX-EXP-2026-09-26-SELECT-004 + +Fresh operator authorization for `HLX-EXP-2026-09-26-SELECT-004` only. It does not inherit authority from SELECT-001, SELECT-002, or SELECT-003. The sealed preregistration bytes stay `365f7ce7141e8ee899fc52d85584a32c21ecfcb596e316b1438463bd8db85c4a`. + +`PROMOTE` requires `classify_macro_f1_nonnone` strictly greater than `seed-morph78` on the sealed twelve-class universe, sha256 `227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175`, absent-class policy `omit_when_gold_support_is_zero`. Equality does not promote. The new SELECT-004 preservation floors are unbind clean exact within 0.01 on 250 identities, classification accuracy within 0.02 on 241 identities, and OBSERVED-label accuracy within 0.05 on 123 identities. A primary decrease, a preservation breach, or a hard integrity failure is `REJECT`. Equal primary metrics with every guard passing are `INCONCLUSIVE`. An inconclusive result is not resolved by changing thresholds. + +Authorization artifact sha256 `7389783b52a5d5cf14ceca874cc12351d3f6571b6be8a8666c2040be38972625`. Post-authorization admission used `HYPERLEX_ALLOW_TRAIN=1` and `HLX_ADMISSION_ONLY=1`. Preflight and the trainer entrypoint share environment hash `a2b18512be8329ee63ad06e98f894c6138cf7eac05f6e82d53d4132e99a27f5d`. `admission_result` is `ADMISSION_PASS`. Status is `TRAINING_READY` because the scientific contract is sealed, the decision rule is sealed, and the real entrypoint passed. `optimizer_loaded` is false. Epochs 0. Gradient steps 0. Training was not started. Output directory stayed absent. BEST sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1` was not moved. Trunk sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. `training_launch_authorized` is false. `HLX_ALLOW_NO_HOLDOUT` stayed unset. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. The decision covers classify 241, classify_observed 123, classify_non_none 188, and unbind_clean 250. It does not establish performance for relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, or spiritual-mystic. Clean unbind stays limited to the sealed WordNet slice. Vendor calls: 0. Do not train. The next decision is launch authorization only. diff --git a/specs/007-hyperlexical-model/label-taxonomy-proposal.md b/specs/007-hyperlexical-model/label-taxonomy-proposal.md new file mode 100644 index 00000000..1f656616 --- /dev/null +++ b/specs/007-hyperlexical-model/label-taxonomy-proposal.md @@ -0,0 +1,361 @@ +# Label taxonomy proposal — evaluation reserve + +**Status:** structure accepted 2026-09-26 with amendments (see Operator acceptance). Row settlement not applied. `evaluation.enabled` is false. +**Date:** 2026-09-26 +**Shelf:** held-out stream `hs-20260925T211358Z` (339 rows). No row text in this file. +**Does not:** change `hyperlexical.layout.FAMILIES`, resize the classify head, call Jev, run `attest-apply`, or admit `EVAL_RESERVE`. + +The production head stays the nine-way list in `layout.py`: `betting-sharp`, `crypto-degen`, `ai-native`, `brainrot-aura`, `kinship-address`, `political-status`, `gaming-meta`, `workplace-corp`, `none`. Dataset schema `dataset_row.v0.1` keeps that lineage enum, plus `ytd_leaf`. This proposal is the evaluation-label ontology for the reserve. It becomes live only after an operator accepts the definitions. + +## Why the current reserve cannot use the old list + +On the 339-row shelf the only rights-cleared non-none labels are `gaming-meta` (99), `betting-sharp` (61), and `crypto-degen` (5). Five production families have no settled label. Macro-F1 over a three-class subset is a narrow score. Adding more rows under those three names does not widen it. + +Hint-only rows already name `kinship-address`, `political-status`, `workplace-corp`, `brainrot-aura`, and `ai-native`. Those hints are not labels. They show the queued sheet is broader than the Kaikki topic map, which is allowed to emit only betting, crypto, and gaming. + +## Four surfaces + +Distinctions that are not a stable semantic region stay off the primary label. + +| Surface | Role | Values on this proposal | +|---|---|---| +| `semantic_family` | Primary class. One region per row. | The sixteen names below, `none`, or `LABEL_UNRESOLVED`. | +| `attest` | How the label was settled. Orthogonal to family. | `OBSERVED`, `INFERRED`, `UNLABELLED`. Empty attest stays empty. | +| `register` | How the wording sits in the language. | `slang`, `domain-specific`, `high-register`, `general`. Unset until an operator sets it. | +| `function` | What the phrase is doing, when that is not the family. | `address`, `evaluation`, `intensification`, `reference`, `affiliation`, or unset. | + +`OBSERVED` / `INFERRED` stay on `attest`. They do not become families. Tone, certainty, and insider/outsider stay off the primary list. A family name is not combined with an attribute (`gaming-meta-observed-negative-insider` is not a class). + +`attest` on this shelf is unchanged: 205 `INFERRED`, 134 `UNLABELLED`, 0 `OBSERVED`. This document does not write the attest column. + +## Promotion rule + +```text +LABEL_PROMOTE(x) = + definition_clear + AND operator_agreement_possible + AND not_redundant_with_existing_label + AND sufficient_support_exists +``` + +`sufficient_support_exists` is `NOT_COMPUTABLE`. No numeric minimum is declared. Until that rule passes, a name is `CANDIDATE_LABEL` and `evaluation.enabled` is false. A single-example class is not an evaluation class. + +Nothing in this draft is promoted. The sixteen names are the proposed active set for the operator to accept or cut. Acceptance of a definition is not activation for macro-F1. + +## Proposed active set (16 non-none) + +`none` remains the abstain class. It is not one of the sixteen. + +Each block is a definition the operator can settle without a model. Positive and near-miss lines are illustrations of the region. They are not reserve rows and they are not taken from the stream. + +### gaming-meta + +- **definition:** Jargon of video games, esports, and gamer communities: mechanics, ranked play, balance, party and lobby talk. +- **include:** Terms whose ordinary sense is a game mechanic, a competitive-play judgment, or a gamer-community formula. +- **exclude:** Athletic sports. Betting lines about sports. A meme that merely mentions a game. +- **near-miss:** A sports-competition term. A general insult with no game sense. +- **parent:** production family of the same name. Kaikki topics `video-games`, `computer-games`, `role-playing-games` already map here. +- **evaluation.enabled:** false. + +### betting-sharp + +- **definition:** Jargon of sports betting, gambling, and poker: odds, lines, bankroll, handicapping. +- **include:** Terms whose ordinary sense is a wager, a line, a stake, or poker-table talk. +- **exclude:** Ordinary sports commentary with no stake. Crypto trading slang. A game mechanic. +- **near-miss:** `sports-competition`. A single word that is also a normal verb. +- **parent:** production family of the same name. Kaikki topics `gambling` and `poker` already map here. +- **evaluation.enabled:** false. + +### crypto-degen + +- **definition:** Jargon of cryptocurrency, DeFi, NFTs, and speculative token trading. +- **include:** Terms whose ordinary sense is a chain, a token, a trade, or that community's stake in a position. +- **exclude:** Ordinary finance with no token or chain sense. A meme about money in general. +- **near-miss:** `finance-retail` and `market-structure` (both candidates, not active). +- **parent:** production family of the same name. +- **evaluation.enabled:** false. + +### internet-slang + +- **definition:** Short-lived or platform-native wording that is slang of general internet speech and is not tied to one trade or hobby. +- **include:** Terms a reader places in internet speech without needing a game, a book, a chart, or a workplace. +- **exclude:** A domain term that happens to be posted online. A meme name whose job is the meme itself (`memetic`). A status claim (`social-status`). +- **near-miss:** `brainrot-aura` rows, which mix this family with `memetic` and `social-status`. They stay unresolved rather than all landing here. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### memetic + +- **definition:** A phrase whose primary job is to circulate as a named meme, copypasta, or recognizable bit. +- **include:** The wording is the meme, or a stable mutation of one. +- **exclude:** Ordinary slang that is not a named bit. A domain term that people joke about. +- **near-miss:** `internet-slang`. Humor as a tone is not this family. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### social-status + +- **definition:** Wording whose primary job is to rank people, scenes, or the self: aura, clout, mid, cooked-as-status, and their kin. +- **include:** The term assigns or withholds standing. +- **exclude:** A game rank that is a mechanic (`gaming-meta`). A political allegiance (`politics-civic`, candidate). +- **near-miss:** `approval-disapproval`, when the phrase judges an act rather than a person's standing. +- **evaluation.enabled:** false. Support on this shelf: 0 clean rows. + +### relationship-dating + +- **definition:** Jargon of dating, romance, and couple-craft as a social practice. +- **include:** Terms whose ordinary sense is a dating move, a romantic role, or couple-status. +- **exclude:** Kinship and address (`bro`, `sis`, family terms). Those are not this family. They do not map here. +- **near-miss:** Production `kinship-address`. Candidate `identity-affiliation`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### approval-disapproval + +- **definition:** A stable region of evaluative slang whose job is to praise, dismiss, or rate a thing. +- **include:** The term is a verdict word, not a domain object. +- **exclude:** The same verdict spoken inside a domain, when the domain is the family and evaluation is only the `function`. A gaming phrase that evaluates a play stays `gaming-meta` with `function: evaluation`. +- **near-miss:** `social-status`. Positive versus negative is an attribute, not two families. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### conflict-aggression + +- **definition:** Jargon of fights, feuds, call-outs, and competitive hostility that is not a sport, a game, or a bet. +- **include:** The term names a conflict move or a hostile stance. +- **exclude:** Athletic competition. In-game combat vocabulary. Criminal procedure (`crime-illicit`, candidate). +- **near-miss:** `sports-competition`. An insult that is only status (`social-status`). +- **evaluation.enabled:** false. Support on this shelf: 0. + +### technology-ai + +- **definition:** Jargon of AI, machine learning, software builders, and AI-era coinages. +- **include:** Terms whose ordinary sense is a model, a training run, an agent, an eval, or builder talk. +- **exclude:** Ordinary computing words with no community sense. The weak mapper already refuses bare `en:Computing`. +- **near-miss:** Production `ai-native`, which this name would replace only after operator acceptance. The head is not renamed here. +- **evaluation.enabled:** false. This shelf has 5 hint-only rows, 0 labels. + +### work-hustle + +- **definition:** Jargon of jobs, offices, careers, and hustle culture. +- **include:** Terms whose ordinary sense is workplace status, management talk, or gig-work craft. +- **exclude:** A hobby called a grind. Crypto or betting talk about "the job." +- **near-miss:** Production `workplace-corp`, the same region under the old name. +- **evaluation.enabled:** false. This shelf has 20 hint-only rows, 0 labels. + +### sports-competition + +- **definition:** Jargon of athletic sports and sporting competition, with no wager and no video game. +- **include:** Terms whose ordinary sense is a play, a position, or a sporting result. +- **exclude:** Betting lines (`betting-sharp`). Video-game play (`gaming-meta`). Bare `en:Sports` stays unlabeled, matching `weak_tag_family.py`. +- **near-miss:** `betting-sharp`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### music-entertainment + +- **definition:** Jargon of music scenes, fandom, and stage entertainment. +- **include:** Terms whose ordinary sense is a scene, a track-craft word, or fandom talk. +- **exclude:** A meme that uses a song title. A fashion term. +- **near-miss:** `memetic`. `fashion-aesthetic`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### fashion-aesthetic + +- **definition:** Jargon of dress, look, and aesthetic scenes. +- **include:** Terms whose ordinary sense is a look, a garment-community word, or an aesthetic label. +- **exclude:** Status slang with no look. A costume inside a game. +- **near-miss:** `social-status`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### regional-cultural + +- **definition:** Wording whose primary identity is a place, a dialect community, or a local scene, rather than a trade. +- **include:** The term is opaque outside that region or dialect, and the region is the region of meaning. +- **exclude:** A domain term that happens to be used in one city. AAVE or dialect material is not dumped here by default; it stays unresolved until an operator can say the region is the family. +- **near-miss:** `internet-slang`. +- **evaluation.enabled:** false. Support on this shelf: 0. + +### spiritual-mystic + +- **definition:** Jargon of spiritual, occult, and mystic scenes. +- **include:** Terms whose ordinary sense is a practice, a belief-community word, or a ritual formula in that scene. +- **exclude:** Ordinary metaphor ("manifest" as office talk). A meme about fate. +- **near-miss:** `work-hustle` when the word is motivational rather than mystic. +- **evaluation.enabled:** false. Support on this shelf: 0. + +## Candidate labels (not active) + +These stay `CANDIDATE_LABEL`. They are not evaluation classes. The broader list in the operator note is the pool; this shelf does not justify promoting them. + +| Candidate | Holds | Why it is not active | +|---|---|---| +| `politics-civic` | Production `political-status` (20 hint-only rows). | Not in the sixteen. No settled examples. | +| `identity-affiliation` | Production `kinship-address` (20 hint-only rows). | Address and kinship are not `relationship-dating`. The candidate is the holding pen, not a gold family. | +| `finance-retail` | Nothing on this shelf. | Easy to collapse into `crypto-degen`. | +| `market-structure` | Nothing on this shelf. | Easy to collapse into `crypto-degen` or `betting-sharp`. | +| `sexual-romantic` | Nothing on this shelf. | Overlaps `relationship-dating` once that family has a definition. | +| `substance-party` | Nothing on this shelf. | No cluster on this shelf. | +| `crime-illicit` | Nothing on this shelf. | No cluster on this shelf. | +| `health-fitness` | Nothing on this shelf. | No cluster on this shelf. | + +`ytd_leaf` stays a schema enum value for the production dataset. It is not a semantic family in this proposal. + +## What was not done + +- Jev was not called. The lineage rule was not run. No model scored a row. +- `attest-apply` was not run. The attest column stays empty. +- The ledger was not mutated. `EVAL_RESERVE` stays 0. +- `layout.FAMILIES` was not edited. +- `brainrot-aura` was not split by reading phrases. + +## Remap of `hs-20260925T211358Z` + +Rules, in order. A hint is never upgraded to a label by this file. + +1. Rights-cleared row, `label_source=INFERRED`, label equals hint, label is `gaming-meta`, `betting-sharp`, `crypto-degen`, or `none`: `PROPOSED_REMAP` onto the same `semantic_family`. `attest` stays `INFERRED`. Not reserved. +2. Label and hint disagree: `LABEL_UNRESOLVED`. +3. `UNLABELLED`, including every hint-only family: `LABEL_UNRESOLVED`. A proposed target may be recorded for the operator. It is not a label. +4. Reddit and Know Your Meme rows: `RIGHTS_UNRESOLVED` as well as `LABEL_UNRESOLVED`. + +| Old label | Hint | Rows | Remap status | Proposed family | Notes | +|---|---|---|---|---|---| +| gaming-meta | gaming-meta | 98 | `PROPOSED_REMAP` | `gaming-meta` | Rights-clear. One disagreeing row is excluded. | +| betting-sharp | betting-sharp | 61 | `PROPOSED_REMAP` | `betting-sharp` | Rights-clear. | +| crypto-degen | crypto-degen | 5 | `PROPOSED_REMAP` | `crypto-degen` | Rights-clear. Support is thin. Evaluation stays off. | +| none | none | 40 | `PROPOSED_REMAP` | `none` | Encyclopedic prose. `register` may be recorded as `high-register` from `source_type`, not from a model. | +| gaming-meta | betting-sharp | 1 | `LABEL_UNRESOLVED` | — | Label/hint collision. | +| UNLABELLED | gaming-meta | 20 | `LABEL_UNRESOLVED` | — | Hint is not a label. | +| UNLABELLED | betting-sharp | 21 | `LABEL_UNRESOLVED` | — | Hint is not a label. Includes rights-unresolved Reddit rows. | +| UNLABELLED | crypto-degen | 6 | `LABEL_UNRESOLVED` | — | All six are Reddit. Rights unresolved. | +| UNLABELLED | workplace-corp | 20 | `LABEL_UNRESOLVED` | `work-hustle` if the operator accepts the rename | Target is a proposal. | +| UNLABELLED | ai-native | 5 | `LABEL_UNRESOLVED` | `technology-ai` if the operator accepts the rename | Target is a proposal. | +| UNLABELLED | kinship-address | 20 | `LABEL_UNRESOLVED` | candidate `identity-affiliation` | Not `relationship-dating`. | +| UNLABELLED | political-status | 20 | `LABEL_UNRESOLVED` | candidate `politics-civic` | Candidate, not active. | +| UNLABELLED | brainrot-aura | 20 | `LABEL_UNRESOLVED` | — | Spans `internet-slang`, `memetic`, and `social-status`. Not split. | +| UNLABELLED | no hint | 2 | `LABEL_UNRESOLVED` | — | Know Your Meme. Rights unresolved. | + +Totals: `PROPOSED_REMAP` 204 (164 non-none, 40 `none`). `LABEL_UNRESOLVED` 135. Admitted to the reserve: 0. + +### Coverage of the sixteen + +| Family | Proposed remap (not gold, not reserved) | Unresolved hints pointing near it | +|---|---|---| +| gaming-meta | 98 | 20 hint-only, plus 1 collision | +| betting-sharp | 61 | 21 hint-only | +| crypto-degen | 5 | 6 hint-only, rights unresolved | +| work-hustle | 0 | 20 `workplace-corp` hints | +| technology-ai | 0 | 5 `ai-native` hints | +| internet-slang | 0 | inside the 20 `brainrot-aura` bundle | +| memetic | 0 | inside the same bundle | +| social-status | 0 | inside the same bundle | +| relationship-dating | 0 | 0 (kinship was not mapped here) | +| approval-disapproval | 0 | 0 | +| conflict-aggression | 0 | 0 | +| sports-competition | 0 | 0 | +| music-entertainment | 0 | 0 | +| fashion-aesthetic | 0 | 0 | +| regional-cultural | 0 | 0 | +| spiritual-mystic | 0 | 0 | + +Eleven of the sixteen have no row on this shelf. The taxonomy is wider than the shelf. That is intentional. Empty families are not filled by invention, and they are not evaluation classes. + +## Operator settlement + +Taxonomy text is ready for review. Row settlement is not ready. + +- Accept, cut, or rewrite the sixteen definitions before any attest. +- Accept or reject the two renames (`workplace-corp` → `work-hustle`, `ai-native` → `technology-ai`) before those hints are eligible. +- Decide whether `politics-civic` and `identity-affiliation` stay candidates. +- Leave `brainrot-aura` unresolved until an operator splits it by hand. +- Then, and only then, `attest-apply`. This file does not authorize that command. + +`SELECT-003` stays undrafted. GEN-1 was not created. BEST stays `seed-morph78`. + + +## Operator acceptance — 2026-09-26 + +The operator accepted the ontology structure and amended the draft. This section supersedes the sixteen-name list, the name `work-hustle`, and the candidate status of `identity-affiliation` and `politics-civic`. The earlier sections stay as the draft record. The production head in `layout.FAMILIES` is unchanged. + +### Authorized active set (18 non-none) + +`gaming-meta`, `betting-sharp`, `crypto-degen`, `internet-slang`, `memetic`, `social-status`, `relationship-dating`, `approval-disapproval`, `conflict-aggression`, `technology-ai`, `workplace-career`, `sports-competition`, `music-entertainment`, `fashion-aesthetic`, `regional-cultural`, `spiritual-mystic`, `identity-affiliation`, `politics-civic`. + +`none` remains abstain. + +`taxonomy.active` is true for these eighteen. `evaluation.enabled` is false for every one of them until operator-settled support exists and a later governance decision turns evaluation on. Those two flags are independent. + +### Renames and promotions + +- `work-hustle` is not a family. The region is `workplace-career`: corporate jargon, employment and status language, career language, workplace hierarchy, and hustle or grind language. Finer shade sits on `function` or `register`. +- `identity-affiliation` is active. It covers group membership, social belonging, in-group and out-group identity, affiliative address, and role affiliation. It is not `relationship-dating`. Familial address used socially is `semantic_family: identity-affiliation` with `function: address`. +- `politics-civic` is active. It covers political roles, civic identity, governmental status, political-group terminology, and public institutional positioning. It is a descriptive region, not an ideological judgment, and it is not `social-status` or `identity-affiliation`. + +### Still candidates + +`finance-retail`, `market-structure`, `sexual-romantic`, `substance-party`, `crime-illicit`, `health-fitness`. + +`brainrot-aura` is not a family. The source cluster is disambiguated row by row into `internet-slang`, `memetic`, `social-status`, another active family, `none`, or `LABEL_UNRESOLVED`. + +### Source hint is not a label + +```yaml +source_hint: + value: + provenance: +semantic_family: + value: + settled_by: + settled_at: +attest: + value: OBSERVED | INFERRED | UNLABELLED + settled_by: + settled_at: +``` + +A source hint is evidence shown to the operator. It does not become `semantic_family`. An existing `INFERRED` label does not become `OBSERVED`. Jev and any other model do not settle either field. + +### Taxonomy-level mappings (not row settlement) + +| Current condition | Mapping | +|---|---| +| gaming-meta label and hint | `gaming-meta` | +| betting-sharp label and hint | `betting-sharp` | +| crypto-degen label and hint | `crypto-degen` | +| none label and hint | `none` | +| workplace-corp hint | `workplace-career` | +| ai-native hint | `technology-ai` | +| kinship-address hint | `identity-affiliation` | +| political-status hint | `politics-civic` | +| brainrot-aura hint | no batch mapping | +| gaming-meta label with betting-sharp hint | no batch mapping | +| Reddit or Know Your Meme | excluded from `EVAL_RESERVE` until rights are resolved | + +Hint-family mapping accepted is not a row label settled. + +### Settlement lanes for `hs-20260925T211358Z` + +Private sheets, mode 0600, under the stream `attest/` directory. Decision cells are empty. `semantic_family`, `attest`, `register`, and `function` are empty. A proposed family is evidence in its own column, not a preselected answer. + +| Lane | Rows | Operator choice | +|---|---|---| +| A CONFIRM | 204 same-label proposed remaps | `ACCEPT`, `RECLASSIFY`, `UNRESOLVED` | +| B PROPOSED FAMILY | 20 workplace-corp, 5 ai-native, 20 kinship-address, 20 political-status | `ACCEPT FAMILY`, `CHOOSE DIFFERENT FAMILY`, `NONE`, `UNRESOLVED` | +| C DISAMBIGUATE | 20 brainrot-aura, 1 label/hint collision | a listed destination, another active family, `none`, or `UNRESOLVED` | +| D RIGHTS BLOCKED | 14 Reddit, 2 Know Your Meme | semantic notes allowed; reserve admission impossible | + +33 Wiktionary rows are hint-only for `gaming-meta` (20) or `betting-sharp` (13). They are not in the four named lanes. They stay `LABEL_UNRESOLVED` on a private holding sheet. They were not given a batch settlement. + +`attest-apply` was not run. The current command only accepts the production eight plus `none` or `reject`, and it writes `label_source=OBSERVED` for every accepted value. That command must not be used on these lanes: it would reject the new names and would collapse `attest` into `OBSERVED`. + +No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. + +## Settlement path — 2026-09-26 + +The operator sheets stay the interface: lane A CONFIRM, lane B PROPOSED FAMILY, lane C DISAMBIGUATE, lane D RIGHTS BLOCKED, and the hint-only holding sheet. Decision cells stay empty until a person fills them. The tool validates completed rows and does not choose answers. + +The apply command is `python -m hyperlexical.identity_ledger settlement-apply`. It does not call production `attest-apply` and does not change that command. `ACCEPT` means the operator-entered settlement, including an explicit `attest` of `INFERRED` or `OBSERVED`. `NONE` writes `semantic_family=none`. `UNRESOLVED` stays null and cannot enter `EVAL_RESERVE`. A row with `RIGHTS_UNRESOLVED` cannot enter `EVAL_RESERVE` even when the family and attest are filled. `source_hint` is not copied into `semantic_family`. + +`taxonomy.active`, `evaluation.enabled`, and `production.enabled` are independent. The eighteen families are `taxonomy.active=true`, `evaluation.enabled=false`, `production.enabled=false`. Candidate names stay inactive unless a later activation names them. `layout.FAMILIES` is unchanged. No row was settled. `EVAL_RESERVE` stays 0. Vendor calls: 0. SELECT-003 was not drafted. + + +## Row settlement — 2026-09-26 + +Stream `hs-20260925T211358Z` was settled through `identity_ledger settlement-apply`, not `attest-apply`. Every row has an explicit decision. Blank is 0. `UNRESOLVED` is 84. Settled is 255 (`ACCEPT` 204, `RECLASSIFY` 32, `NONE` 19). `OBSERVED` 128 and `INFERRED` 127 are separate from the decision. `source_hint` was not copied into `semantic_family`. + +The eighteen families were not expanded. One economics row was left `UNRESOLVED` because the fitting region is the inactive candidate `finance-retail`. `brainrot-aura` was not added. `evaluation.enabled` stays false. Rights-blocked rows can carry a semantic decision and still cannot enter `EVAL_RESERVE`. Vendor calls: 0. SELECT-003 was not drafted. Do not train. diff --git a/tests/shadow/admission_fixtures.py b/tests/shadow/admission_fixtures.py new file mode 100644 index 00000000..a14576cf --- /dev/null +++ b/tests/shadow/admission_fixtures.py @@ -0,0 +1,183 @@ +"""Fixtures for controlled-experiment admission tests. Not collected by pytest.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.identity_ledger import IdentityLedger, derived_state +from hyperlexical.selection_surface import row_id + + +def sha256_bytes(raw: bytes) -> str: + return hashlib.sha256(raw).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + payload = "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows) + path.write_text(payload, encoding="utf-8") + return sha256_bytes(path.read_bytes()) + + +def classify_row(text: str, *, split: str = "train") -> dict: + return { + "text": text, + "task": "classify", + "split": split, + "class": "OBSERVED", + "lineage": "none", + "role_scheme": "positional", + } + + +def _label_row(text: str, label: dict) -> dict: + return { + "text": text, + "task": label["task"], + "split": label.get("split") or "val", + "class": label.get("class") or "", + "lineage": label.get("lineage") or "", + "role_scheme": "positional", + } + + +def seal_reserve( + root: Path, + *, + experiment_id: str, + extra: list[tuple[str, dict]] | None = None, +) -> dict: + """Four-slice EVAL_RESERVE ledger plus a binding file. No training rows.""" + labels = [ + ( + "reserve classify surface", + { + "task": "classify", + "class": "OBSERVED", + "lineage": "gaming-meta", + "split": "val", + }, + ), + ( + "reserve unbind surface", + { + "task": "unbind", + "class": "INFERRED", + "lineage": "none", + "split": "val", + "unbind_clean": True, + }, + ), + ] + labels.extend(extra or []) + ledger = IdentityLedger() + for text, label in labels: + row = _label_row(text, label) + digest = normalized_text_sha256(text) + ledger.observe( + digest, + source_artifact="fixture", + row_ids=[row_id(row)], + labels=[label], + provenance="fixture", + experiment_id=experiment_id, + ) + ledger.transition( + digest, + "EVAL_RESERVE", + source_artifact="fixture", + provenance="fixture", + ) + ledger_dir = root / "ledger" + ledger.save(ledger_dir) + events_sha = sha256_bytes((ledger_dir / "events.jsonl").read_bytes()) + reserved = [ + record + for record in ledger.identities.values() + if derived_state(record) == "EVAL_RESERVE" + ] + binding = { + "schema": "hyperlex.reserve_binding.v1", + "experiment_id": experiment_id, + "ledger_events_sha256": events_sha, + "counts": ledger.reserve_counts(), + "identities": len(reserved), + "lifecycle": "EVAL_RESERVE", + } + binding_path = root / "reserve-binding.json" + binding_path.write_text(json.dumps(binding, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return {"ledger": ledger_dir, "binding": binding_path, "counts": binding["counts"]} + + +def write_weights(path: Path, payload: bytes) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(payload) + return sha256_bytes(payload) + + +def arm_controlled( + monkeypatch, + tmp_path: Path, + rows: list[dict], + *, + experiment_id: str = "HLX-EXP-TEST", + extra_reserve: list[tuple[str, dict]] | None = None, + candidate_extra: dict | None = None, + best_sha: str | None = None, + trunk_sha: str | None = None, + create_output: bool = False, +) -> dict: + """Launch environment for one controlled experiment. Output stays absent unless asked.""" + pinned = tmp_path / "pinned.jsonl" + digest = write_jsonl(pinned, rows) + reserve = seal_reserve(tmp_path, experiment_id=experiment_id, extra=extra_reserve) + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + actual_trunk = write_weights(trunk / "model.safetensors", b"trunk-weights") + actual_best = write_weights(tmp_path / "best.safetensors", b"best-weights") + out = tmp_path / "train-out" + if create_output: + out.mkdir() + baseline = {"HYPERLEX_FILLER_FILTER": "off"} + candidate = { + "HYPERLEX_FILLER_FILTER": "off", + "HLX_SELECT_METRIC": "classify_macro_f1_nonnone", + "HLX_EXPERIMENT_ID": experiment_id, + "HYPERLEX_TRAIN_OUT": str(out), + } + if candidate_extra: + candidate.update(candidate_extra) + baseline_path = tmp_path / "baseline-env.json" + candidate_path = tmp_path / "candidate-env.json" + baseline_path.write_text(json.dumps(baseline, sort_keys=True), encoding="utf-8") + candidate_path.write_text(json.dumps(candidate, sort_keys=True), encoding="utf-8") + monkeypatch.setenv("HLX_EXPERIMENT_ID", experiment_id) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(pinned)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", digest) + monkeypatch.setenv("HLX_TRAIN_EXPORT_ROWS", str(len(rows))) + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(reserve["ledger"])) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(reserve["binding"])) + monkeypatch.setenv("HLX_BASELINE_ENV", str(baseline_path)) + monkeypatch.setenv("HLX_CANDIDATE_ENV", str(candidate_path)) + monkeypatch.setenv("HLX_SELECT_METRIC", "classify_macro_f1_nonnone") + monkeypatch.setenv("HYPERLEX_FILLER_FILTER", "off") + monkeypatch.setenv("HLX_BEST_SHA256", best_sha if best_sha is not None else actual_best) + monkeypatch.setenv("HLX_BEST_WEIGHTS", str(tmp_path / "best.safetensors")) + monkeypatch.setenv("HLX_TRUNK_SHA256", trunk_sha if trunk_sha is not None else actual_trunk) + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setenv("HYPERLEX_TRAIN_OUT", str(out)) + monkeypatch.setenv("HYPERLEX_EXPORT_DIR", str(tmp_path / "export-out")) + monkeypatch.delenv("HLX_HOLDOUT_MANIFESTS", raising=False) + monkeypatch.delenv("HLX_ALLOW_NO_HOLDOUT", raising=False) + return { + "pinned": pinned, + "digest": digest, + "out": out, + "trunk": trunk, + "reserve": reserve, + "rows": len(rows), + } diff --git a/tests/shadow/test_hyperlexical_admission_parity.py b/tests/shadow/test_hyperlexical_admission_parity.py new file mode 100644 index 00000000..bd19fc0b --- /dev/null +++ b/tests/shadow/test_hyperlexical_admission_parity.py @@ -0,0 +1,347 @@ +"""Preflight and the trainer share one admission result.""" + +from __future__ import annotations + +import hashlib +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +from admission_fixtures import arm_controlled, classify_row # noqa: E402 +from hyperlexical.admission import ( # noqa: E402 + effective_environment_hash, + effective_environment_material, +) +from hyperlexical.preflight import main as preflight_main # noqa: E402 +from hyperlexical import train as train_mod # noqa: E402 + + +@pytest.fixture(autouse=True) +def _clear(monkeypatch): + for key in ( + "HLX_TRAIN_EXPORT_PATH", + "HLX_TRAIN_EXPORT_SHA256", + "HLX_TRAIN_EXPORT_ROWS", + "HLX_EXPERIMENT_ID", + "HLX_HOLDOUT_MANIFESTS", + "HLX_ALLOW_NO_HOLDOUT", + "HLX_ADMISSION_ONLY", + "HLX_EVAL_RESERVE_LEDGER", + "HLX_RESERVE_BINDING", + "HLX_BASELINE_ENV", + "HLX_CANDIDATE_ENV", + "HLX_SELECT_METRIC", + "HLX_BEST_SHA256", + "HLX_BEST_WEIGHTS", + "HLX_TRUNK_SHA256", + "HYPERLEX_ALLOW_TRAIN", + "HYPERLEX_INCLUDE_LIVE", + "HYPERLEX_EXPORT_DIR", + "HYPERLEX_TRAIN_OUT", + "HYPERLEX_TRUNK_DIR", + "HYPERLEX_FILLER_FILTER", + "HYPERLEX_RELEASE_SET", + "HLX_THRESHOLD_AUTHORIZATION", + ): + monkeypatch.delenv(key, raising=False) + + +def _refuse_export(*_args, **_kwargs): + raise AssertionError("export_dataset must not run") + + +def _refuse_train(): + raise AssertionError("training execution was reached") + + +def _pair(monkeypatch, capsys): + monkeypatch.setenv("HLX_ADMISSION_ONLY", "1") + monkeypatch.setattr("hyperlexical.preflight.export_dataset", _refuse_export) + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + monkeypatch.setattr("hyperlexical.loop._enter_training_execution", _refuse_train) + pre_code = preflight_main() + pre = json.loads(capsys.readouterr().out) + return pre_code, pre + + +def _launch(capsys): + try: + code = train_mod.main(["--offline", "--run", "--include-live"]) + except SystemExit as exc: + err = capsys.readouterr() + return exc, err + out = json.loads(capsys.readouterr().out) + return code, out + + +def test_environment_hash_includes_launch_overlay_and_skips_admission_only(monkeypatch): + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_ADMISSION_ONLY", "1") + material = effective_environment_material() + assert material["HYPERLEX_ALLOW_TRAIN"] == "1" + assert "HLX_ADMISSION_ONLY" not in material + assert "HLX_THRESHOLD_AUTHORIZATION" in material + assert material["HLX_THRESHOLD_AUTHORIZATION"] is None + first = effective_environment_hash() + monkeypatch.setenv("HLX_THRESHOLD_AUTHORIZATION", "/tmp/not-a-file") + assert effective_environment_hash() != first + monkeypatch.delenv("HLX_THRESHOLD_AUTHORIZATION") + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "0") + assert effective_environment_hash() != first + + +def test_controlled_reserve_passes_preflight_and_admission_only(monkeypatch, tmp_path, capsys): + armed = arm_controlled(monkeypatch, tmp_path, [classify_row(f"train {i}") for i in range(8)]) + code, pre = _pair(monkeypatch, capsys) + assert code == 0 + assert pre["admission_result"] == "ADMISSION_PASS" + assert pre["status"] == "PREREGISTERED" + assert pre["ready_to_train"] is False + assert pre["scientific_contract_sealed"] is True + assert pre["decision_rule_sealed"] is False + assert pre["decision_threshold_state"] == "BLOCKED_PENDING_OPERATOR_AUTHORIZATION" + assert pre["training_launch_authorized"] is False + assert pre["controlled_holdout_contract"] == "CONTROLLED_RESERVE" + assert pre["training_input_mode"] == "PINNED_EXPORT" + assert pre["training_export_sha256_actual"] == armed["digest"] + assert pre["training_export_rows"] == 8 + assert pre["live_export_generation_enabled"] is False + assert pre["reserve_train_row_id_overlap"] == 0 + assert pre["reserve_train_text_hash_overlap"] == 0 + assert pre["training_rows_removed"] == 0 + assert pre["epochs"] == 0 + assert pre["gradient_steps"] == 0 + assert pre["optimizer_loaded"] is False + assert pre["training_started"] is False + assert pre["hlx_allow_no_holdout"] is False + launch_code, launch = _launch(capsys) + assert launch_code == 0 + assert launch["admission_result"] == pre["admission_result"] + assert launch["status"] == "PREREGISTERED" + assert launch["ready_to_train"] is False + assert launch["environment_hash"] == pre["environment_hash"] + assert launch["admission_gate_sequence"] == pre["admission_gate_sequence"] + assert launch["epochs"] == 0 + assert launch["optimizer_loaded"] is False + assert not armed["out"].exists() + + +def _assert_same_failure(monkeypatch, capsys): + code, pre = _pair(monkeypatch, capsys) + assert code == 2 + assert pre["admission_result"] == "ADMISSION_FAIL" + assert pre["status"] == "ADMISSION_FAIL" + exc, _err = _launch(capsys) + assert isinstance(exc, SystemExit) + assert str(exc) == pre["error"] + assert exc.receipt["environment_hash"] == pre["environment_hash"] + assert exc.receipt["admission_gate_sequence"] == pre["admission_gate_sequence"] + assert exc.receipt["failed_gate"] == pre["failed_gate"] + return pre + + +def test_missing_pin_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.delenv("HLX_TRAIN_EXPORT_PATH") + monkeypatch.delenv("HLX_TRAIN_EXPORT_SHA256") + monkeypatch.delenv("HLX_TRAIN_EXPORT_ROWS") + pre = _assert_same_failure(monkeypatch, capsys) + assert "no pinned export" in pre["error"] + assert pre["failed_gate"] == "pinned_training_input" + + +def test_missing_reserve_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.delenv("HLX_EVAL_RESERVE_LEDGER") + monkeypatch.delenv("HLX_RESERVE_BINDING") + pre = _assert_same_failure(monkeypatch, capsys) + assert "no sealed evaluation reserve" in pre["error"] + assert pre["failed_gate"] == "holdout_reserve" + + +def test_pinned_digest_mismatch_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", "b" * 64) + pre = _assert_same_failure(monkeypatch, capsys) + assert "digest mismatch" in pre["error"] + assert pre["failed_gate"] == "pinned_training_input" + + +def test_reserve_overlap_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled( + monkeypatch, + tmp_path, + [classify_row("reserve classify surface")], + ) + pre = _assert_same_failure(monkeypatch, capsys) + assert "overlaps the sealed reserve" in pre["error"] + assert pre["failed_gate"] == "train_reserve_disjointness" + assert pre["training_rows_removed"] == 0 + + +def test_second_scientific_variable_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled( + monkeypatch, + tmp_path, + [classify_row("train row")], + candidate_extra={"HYPERLEX_TASK_ROUTING": "route_rows"}, + ) + pre = _assert_same_failure(monkeypatch, capsys) + assert "scientific variable count" in pre["error"] + assert pre["failed_gate"] == "single_variable" + + +def test_best_mismatch_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled( + monkeypatch, + tmp_path, + [classify_row("train row")], + best_sha="c" * 64, + ) + pre = _assert_same_failure(monkeypatch, capsys) + assert "BEST weights sha256 mismatch" in pre["error"] + assert pre["failed_gate"] == "best_trunk" + + +def test_trunk_mismatch_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled( + monkeypatch, + tmp_path, + [classify_row("train row")], + trunk_sha="d" * 64, + ) + pre = _assert_same_failure(monkeypatch, capsys) + assert "trunk weights sha256 mismatch" in pre["error"] + assert pre["failed_gate"] == "best_trunk" + + +def test_output_directory_collision_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled( + monkeypatch, + tmp_path, + [classify_row("train row")], + create_output=True, + ) + pre = _assert_same_failure(monkeypatch, capsys) + assert "output directory collision" in pre["error"] + assert pre["failed_gate"] == "output_directory" + + +def _write_thresholds(tmp_path: Path, **overrides) -> Path: + payload = { + "schema": "hyperlex.threshold_authorization.v1", + "experiment_id": "HLX-EXP-TEST", + "sealed": True, + "decision_thresholds": {"unbind_clean_exact": 0.5}, + } + payload.update(overrides) + path = tmp_path / "threshold-authorization.json" + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + +def test_sealed_decision_rule_reaches_training_ready(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row(f"train {i}") for i in range(4)]) + path = _write_thresholds(tmp_path) + monkeypatch.setenv("HLX_THRESHOLD_AUTHORIZATION", str(path)) + code, pre = _pair(monkeypatch, capsys) + assert code == 0 + assert pre["admission_result"] == "ADMISSION_PASS" + assert pre["status"] == "TRAINING_READY" + assert pre["ready_to_train"] is True + assert pre["decision_rule_sealed"] is True + assert pre["decision_threshold_state"] == "SEALED" + assert pre["training_launch_authorized"] is False + assert pre["optimizer_loaded"] is False + launch_code, launch = _launch(capsys) + assert launch_code == 0 + assert launch["status"] == pre["status"] + assert launch["environment_hash"] == pre["environment_hash"] + assert launch["optimizer_loaded"] is False + + +def test_threshold_authorization_for_another_experiment_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + path = _write_thresholds(tmp_path, experiment_id="HLX-EXP-OTHER") + monkeypatch.setenv("HLX_THRESHOLD_AUTHORIZATION", str(path)) + pre = _assert_same_failure(monkeypatch, capsys) + assert "experiment_id does not match" in pre["error"] + assert pre["failed_gate"] == "ready" + assert pre["status"] == "ADMISSION_FAIL" + + +def test_unsealed_threshold_authorization_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + path = _write_thresholds(tmp_path, sealed=False) + monkeypatch.setenv("HLX_THRESHOLD_AUTHORIZATION", str(path)) + pre = _assert_same_failure(monkeypatch, capsys) + assert "not sealed" in pre["error"] + assert pre["failed_gate"] == "ready" + + +def _rebind_ledger(monkeypatch, tmp_path: Path, armed: dict, ledger) -> None: + directory = tmp_path / "ledger-rebound" + ledger.save(directory) + binding_path = Path(armed["reserve"]["binding"]) + binding = json.loads(binding_path.read_text(encoding="utf-8")) + digest = hashlib.sha256((directory / "events.jsonl").read_bytes()).hexdigest() + binding["ledger_events_sha256"] = digest + rebound = tmp_path / "reserve-binding-rebound.json" + rebound.write_text(json.dumps(binding), encoding="utf-8") + monkeypatch.setenv("HLX_EVAL_RESERVE_LEDGER", str(directory)) + monkeypatch.setenv("HLX_RESERVE_BINDING", str(rebound)) + + +def test_historical_spent_identity_is_not_the_controlled_reserve(monkeypatch, tmp_path, capsys): + from hyperlexical.identity_ledger import IdentityLedger + + armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + ledger = IdentityLedger.load(armed["reserve"]["ledger"]) + ledger.mark_historical( + "ab" * 32, + "evaluation_spent", + source_artifact="historical-fixture", + provenance="fixture", + ) + _rebind_ledger(monkeypatch, tmp_path, armed, ledger) + code, pre = _pair(monkeypatch, capsys) + assert code == 0 + assert pre["admission_result"] == "ADMISSION_PASS" + assert pre["status"] == "PREREGISTERED" + assert pre["reserve_lifecycle"] == "EVAL_RESERVE" + + +def test_reserve_identity_that_left_eval_reserve_rejects_both(monkeypatch, tmp_path, capsys): + from hyperlexical.identity_ledger import IdentityLedger, derived_state + + armed = arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + ledger = IdentityLedger.load(armed["reserve"]["ledger"]) + reserved = [ + digest + for digest, record in ledger.identities.items() + if derived_state(record) == "EVAL_RESERVE" + ] + ledger.transition( + reserved[0], + "EVAL_BOUND", + source_artifact="fixture", + provenance="fixture", + ) + _rebind_ledger(monkeypatch, tmp_path, armed, ledger) + pre = _assert_same_failure(monkeypatch, capsys) + assert "EVAL_BOUND" in pre["error"] + assert pre["failed_gate"] == "holdout_reserve" + + +def test_allow_no_holdout_rejects_both(monkeypatch, tmp_path, capsys): + arm_controlled(monkeypatch, tmp_path, [classify_row("train row")]) + monkeypatch.setenv("HLX_ALLOW_NO_HOLDOUT", "1") + pre = _assert_same_failure(monkeypatch, capsys) + assert "HLX_ALLOW_NO_HOLDOUT" in pre["error"] + assert pre["failed_gate"] == "experiment_binding" diff --git a/tests/shadow/test_hyperlexical_clean_unbind.py b/tests/shadow/test_hyperlexical_clean_unbind.py new file mode 100644 index 00000000..401a2834 --- /dev/null +++ b/tests/shadow/test_hyperlexical_clean_unbind.py @@ -0,0 +1,305 @@ +"""Clean-unbind admission. Contamination hash and the clean predicate stay distinct.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.clean_unbind import ( # noqa: E402 + admit_clean_unbind, + dual_scheme_rows, + gate_rows, + read_wordnet_index, + schema_example, + select_diverse, + structural_reason, +) +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import IdentityLedger # noqa: E402 +from hyperlexical.unbind_settlement import ( # noqa: E402 + append_events, + event_from_decision, + load_events, +) + +LICENSE = "Example rights grant for unit fixtures" +OPERATOR = "unit-operator" +WHEN = "2026-09-26T00:00:00Z" + + +def _rows(tokens, **kwargs): + return dual_scheme_rows( + tokens, + license=kwargs.pop("license", LICENSE), + provenance=kwargs.pop("provenance", "source:unit"), + target_origin=kwargs.pop("target_origin", "source_lemma_tokens"), + **kwargs, + ) + + +def _accept(row, decision="ACCEPT", previous=""): + from hyperlexical.clean_unbind import target_sha256 + + return event_from_decision( + normalized_text_sha256=normalized_text_sha256(row["text"]), + decision=decision, + target_sha256=target_sha256(row["fillers"]), + target_provenance=row["target_origin"], + operator=OPERATOR, + settled_at=WHEN, + decision_basis="source_lemma_token_identity", + previous_target_sha256=previous, + ) + + +def _settled(rows, **overrides): + events = [] + for row in rows: + cloned = dict(row) + if "decision" in overrides: + decision = overrides["decision"] + else: + decision = "ACCEPT" + events.append(_accept(cloned, decision=decision)) + return {event["normalized_text_sha256"]: event for event in events} + + +def _train(text): + return {"text": text, "split": "train", "task": "unbind", "role_scheme": "positional"} + + +def _classify(text): + return { + "text": text, + "task": "classify", + "class": "OBSERVED", + "lineage": "gaming-meta", + "split": "eval", + "fillers": [], + "roles": [], + "role_scheme": None, + } + + +def test_schema_example_uses_placeholders_only(): + example = schema_example() + blob = json.dumps(example) + assert "EXAMPLE_TOKEN_A" in blob + assert example["positional"]["task"] == "unbind" + assert example["scorer"]["this_module_scores"] is False + assert example["scorer"]["clean_slice"] == "soft_ceiling.clean_surface" + + +def test_novel_clean_unbind_is_admitted_and_dirty_is_not(): + novel = _rows(["qxalpha", "qxbeta"]) + dirty = _rows(["qxdirty", "qxphrase"]) + ledger = IdentityLedger() + report = admit_clean_unbind( + ledger, + novel + dirty, + batch_id="unit-clean", + source_artifact="unit", + train_rows=[_train(row["text"]) for row in dirty], + settlements=_settled(novel + dirty), + ) + assert report["unique_admitted_to_eval_reserve"] == 2 + assert report["reserve_counts_after"]["unbind_clean"] == 2 + assert report["rejection_counts"]["not_clean"] == 2 + assert report["unique_routed_to_train_candidate"] == 0 + for row in dirty: + assert ledger.identity(normalized_text_sha256(row["text"])) is None + + +def test_missing_target_model_target_and_unresolved_rights_are_rejected(): + missing = _rows(["qxmiss", "qxtarget"])[0] + missing["fillers"] = [] + modeled = _rows(["qxmodel", "qxtarget"], target_origin="model") + unresolved = _rows(["qxrights", "qxtarget"], license="RIGHTS_UNRESOLVED") + ledger = IdentityLedger() + _admissible, rejections, _account = gate_rows( + [missing, modeled[0], unresolved[0]], + train_rows=[], + ledger=ledger, + settlements={}, + ) + reasons = {item["reason"] for item in rejections} + assert "missing_target" in reasons + assert "model_derived_target" in reasons + assert "rights_unresolved" in reasons + assert ledger.reserve_counts()["unbind_clean"] == 0 + + +def test_blocked_ledger_states_reject_the_same_surface(): + surface = ["qxblock", "qxsurface"] + rows = _rows(surface) + positional = rows[0] + digest = normalized_text_sha256(positional["text"]) + reasons = [] + for flag, reason in ( + ("training_consumed", "TRAIN_CONSUMED"), + ("evaluation_spent", "EVAL_SPENT"), + ("evaluation_abandoned", "EVAL_ABANDONED"), + ): + ledger = IdentityLedger() + ledger.observe_row(positional, source_artifact="pin", provenance="pin", catalogued=True) + ledger.mark_historical(digest, flag, source_artifact="pin", provenance=flag) + _kept, rejections, _account = gate_rows( + [positional], + train_rows=[], + ledger=ledger, + settlements=_settled([positional]), + ) + assert rejections[0]["reason"] == reason + reasons.append(reason) + assert ledger.reserve_counts()["unbind_clean"] == 0 + reserved = IdentityLedger() + reserved.admit( + [_classify(positional["text"])], + batch_id="unit-reserve", + source_artifact="unit", + targets={"classify": 5, "classify_observed": 5, "classify_non_none": 5, "unbind_clean": 5}, + ) + assert reserved.reserve_counts()["classify"] == 1 + _kept, rejections, _account = gate_rows( + [positional], + train_rows=[], + ledger=reserved, + settlements=_settled([positional]), + ) + assert rejections[0]["reason"] == "EVAL_RESERVE" + assert reserved.reserve_counts()["classify"] == 1 + assert reserved.reserve_counts()["unbind_clean"] == 0 + assert reasons == ["TRAIN_CONSUMED", "EVAL_SPENT", "EVAL_ABANDONED"] + + +def test_two_surfaces_one_target_stay_distinct_and_classify_reserve_holds(): + pair = _rows(["qxshared", "qxtarget"], source_pos="noun") + assert normalized_text_sha256(pair[0]["text"]) != normalized_text_sha256(pair[1]["text"]) + from hyperlexical.clean_unbind import target_sha256 + + assert target_sha256(pair[0]["fillers"]) == target_sha256(pair[1]["fillers"]) + ledger = IdentityLedger() + ledger.admit( + [_classify("qxclassify qxonly")], + batch_id="unit-classify", + source_artifact="unit", + targets={"classify": 5, "classify_observed": 5, "classify_non_none": 5, "unbind_clean": 5}, + ) + before = ledger.reserve_counts() + report = admit_clean_unbind( + ledger, + pair, + batch_id="unit-pair", + source_artifact="unit", + train_rows=[], + settlements=_settled(pair), + ) + after = ledger.reserve_counts() + assert after["classify"] == before["classify"] == 1 + assert after["classify_observed"] == before["classify_observed"] + assert after["classify_non_none"] == before["classify_non_none"] + assert after["unbind_clean"] == 2 + assert report["diversity"]["unique_targets"] == 1 + assert report["diversity"]["max_surfaces_per_target"] == 2 + assert report["diversity"]["role_scheme"] == {"positional": 1, "type_slot": 1} + + +def test_settlement_correction_appends_and_reject_does_not_admit(tmp_path): + row = _rows(["qxsettle", "qxatom"])[0] + first = _accept(row, decision="ACCEPT") + log = tmp_path / "events.jsonl" + append_events(log, [first]) + corrected = event_from_decision( + normalized_text_sha256=first["normalized_text_sha256"], + decision="CORRECT_TARGET", + target_sha256="b" * 64, + previous_target_sha256=first["target_sha256"], + target_provenance="operator_authored", + operator=OPERATOR, + settled_at=WHEN, + decision_basis="operator_correction", + ) + append_events(log, [corrected]) + body = log.read_text(encoding="utf-8").splitlines() + assert len(body) == 2 + assert json.loads(body[0])["decision"] == "ACCEPT" + assert json.loads(body[1])["decision"] == "CORRECT_TARGET" + assert load_events(log)[0] == json.loads(body[0]) + rejected = _rows(["qxno", "qxadmit"]) + ledger = IdentityLedger() + report = admit_clean_unbind( + ledger, + rejected, + batch_id="unit-reject", + source_artifact="unit", + train_rows=[], + settlements=_settled(rejected, decision="REJECT"), + ) + assert report["unique_admitted_to_eval_reserve"] == 0 + assert report["rejection_counts"]["settlement_reject"] == 2 + assert structural_reason(rejected[0]) is None + + +def test_diverse_selection_is_not_one_token_length(): + short = [] + for index in range(4): + short.extend(_rows([f"qxshort{index}", "qxbeta"], source_pos="noun")) + long = _rows(["qxone", "qxtwo", "qxthree", "qxfour", "qxfive", "qxsix"], source_pos="noun") + picked = select_diverse(short + long, limit=4) + lengths = {len(row["fillers"]) for row in picked} + assert 6 in lengths + assert 2 in lengths + + +def test_wordnet_index_reader_keeps_lemma_text_out_of_provenance(tmp_path): + (tmp_path / "index.noun").write_text( + " copyright line\nqxunit_qxlemma n 1 1 @ 1 0 00000001\nqxonly n 1 1 @ 1 0 00000002\n", + encoding="utf-8", + ) + for name in ("index.verb", "index.adj", "index.adv"): + (tmp_path / name).write_text(" copyright\n", encoding="utf-8") + atoms = read_wordnet_index(tmp_path) + assert len(atoms) == 1 + assert atoms[0]["tokens"] == ["qxunit", "qxlemma"] + assert "qxunit_qxlemma" not in atoms[0]["lemma_sha256"] + + +def test_cli_train_export_marks_clean_surface_only(tmp_path): + from hyperlexical.identity_ledger import main + + ledger_dir = tmp_path / "ledger" + ledger = IdentityLedger() + ledger.save(ledger_dir) + rows_path = tmp_path / "rows.jsonl" + train_path = tmp_path / "train.jsonl" + novel = _rows(["qxcli", "qxnovel"])[0] + dirty = _rows(["qxcli", "qxdirty"])[0] + rows_path.write_text(json.dumps(novel) + "\n" + json.dumps(dirty) + "\n", encoding="utf-8") + train_path.write_text(json.dumps(_train(dirty["text"])) + "\n", encoding="utf-8") + assert ( + main( + [ + "admit", + "--ledger", + str(ledger_dir), + "--rows", + str(rows_path), + "--batch-id", + "unit-cli", + "--source", + "unit", + "--train-export", + str(train_path), + ] + ) + == 0 + ) + loaded = IdentityLedger.load(ledger_dir) + assert loaded.reserve_counts()["unbind_clean"] == 1 + dirty_record = loaded.identity(normalized_text_sha256(dirty["text"])) + assert dirty_record is not None + assert all(label.get("unbind_clean") is not True for label in dirty_record["labels"]) diff --git a/tests/shadow/test_hyperlexical_eval_settlement.py b/tests/shadow/test_hyperlexical_eval_settlement.py new file mode 100644 index 00000000..d4f70eb4 --- /dev/null +++ b/tests/shadow/test_hyperlexical_eval_settlement.py @@ -0,0 +1,475 @@ +"""Evaluation settlement keeps family, attest, and the production head apart.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.eval_settlement import ( # noqa: E402 + ACTIVE_FAMILIES, + CANDIDATE_FAMILIES, + SHEET_COLUMNS, + SURFACES, + family_flags, + parse_sheet, + receipt_sha_matches, + settlement_allows_reserve, + settle, +) +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import main # noqa: E402 +from hyperlexical.layout import FAMILIES as PRODUCTION_FAMILIES # noqa: E402 +from hyperlexical.weak_tag_family import FAMILIES as WEAK_FAMILIES # noqa: E402 + +STAMP = { + "operator": "operator", + "settled_at": "2026-09-26T16:00:00Z", + "provenance": "unit", + "batch_id": "unit-batch", +} + + +def _stream( + text="phrase alpha", + *, + key="hs-1", + hint="gaming-meta", + source="wiktionary_category", + label="gaming-meta", + label_source="INFERRED", +): + return { + "row_key": key, + "text": text, + "family_hint_not_a_label": hint, + "source_type": source, + "label": label, + "label_source": label_source, + "normtext_sha256": normalized_text_sha256(text), + } + + +def _decision(stream, **overrides): + payload = { + "row_id": stream["row_key"], + "text_hash": normalized_text_sha256(stream["text"]), + "decision": "ACCEPT", + "semantic_family": "gaming-meta", + "attest": "INFERRED", + "register": None, + "function": None, + "source_hint": stream["family_hint_not_a_label"], + "proposed_family": "gaming-meta", + "lane": "A", + } + payload.update(overrides) + return payload + + +def _apply(stream, decision): + streams = stream if isinstance(stream, list) else [stream] + decisions = decision if isinstance(decision, list) else [decision] + return settle(streams, decisions, **STAMP) + + +def _sheet(path: Path, stream, *, lane="A", decision="", family="", attest="", register="", function="", proposed=None): + hint = stream["family_hint_not_a_label"] + hint_cell = "NO_HINT" if hint in (None, "") else hint + label = "UNLABELLED" if stream["label"] is None else str(stream["label"]) + rights = "CC-BY-SA" if stream["source_type"] in ("wiktionary_category", "wikipedia_prose") else "RIGHTS_UNRESOLVED" + if proposed is None: + proposed = family or ("" if hint in (None, "") else str(hint)) + row = [ + lane, + stream["row_key"], + stream["text"], + hint_cell, + "stream.family_hint_not_a_label", + label, + stream["label_source"], + rights, + proposed, + "", + family, + attest, + register, + function, + decision, + ] + path.write_text("\t".join(SHEET_COLUMNS) + "\n" + "\t".join(row) + "\n", encoding="utf-8") + + +def test_evaluation_only_families_are_accepted(): + for family, hint, proposed in ( + ("workplace-career", "workplace-corp", "workplace-career"), + ("technology-ai", "ai-native", "technology-ai"), + ("identity-affiliation", "kinship-address", "identity-affiliation"), + ("politics-civic", "political-status", "politics-civic"), + ): + stream = _stream(f"phrase {family}", hint=hint, label=None, label_source="UNLABELLED") + record = _apply( + stream, + _decision( + stream, + semantic_family=family, + proposed_family=proposed, + lane="B", + decision="ACCEPT FAMILY", + ), + )["records"][0] + assert record["semantic_family"] == family + assert record["source_hint"] == hint + assert record["attest"] == "INFERRED" + assert "label_source" not in record + + +def test_production_family_is_accepted_on_the_evaluation_surface(): + stream = _stream("phrase gaming") + record = _apply(stream, _decision(stream, semantic_family="gaming-meta"))["records"][0] + assert record["semantic_family"] == "gaming-meta" + assert record["source_hint"] == "gaming-meta" + assert family_flags("gaming-meta")["production.enabled"] is False + + +def test_source_hint_is_preserved_and_not_copied(): + stream = _stream("phrase career", hint="workplace-corp", label=None, label_source="UNLABELLED") + record = _apply( + stream, + _decision( + stream, + semantic_family="workplace-career", + proposed_family="workplace-career", + lane="B", + ), + )["records"][0] + assert record["source_hint"] == "workplace-corp" + assert record["semantic_family"] == "workplace-career" + stream = _stream("phrase blank hint", hint="workplace-corp", label=None, label_source="UNLABELLED") + with pytest.raises(SystemExit, match="ACCEPT requires an explicit semantic_family"): + _apply(stream, _decision(stream, semantic_family="", proposed_family="workplace-career", lane="B")) + + +def test_accept_inferred_remains_inferred(): + stream = _stream("phrase inferred") + record = _apply(stream, _decision(stream, decision="ACCEPT", attest="INFERRED"))["records"][0] + assert record["decision"] == "ACCEPT" + assert record["attest"] == "INFERRED" + assert record["semantic_family"] == "gaming-meta" + + +def test_accept_observed_records_operator_observed(): + stream = _stream("phrase observed", label_source="INFERRED") + record = _apply(stream, _decision(stream, decision="ACCEPT", attest="OBSERVED"))["records"][0] + assert record["attest"] == "OBSERVED" + assert "label_source" not in record + + +def test_reclassify_requires_an_explicit_different_family(): + stream = _stream("phrase move", hint="brainrot-aura", label=None, label_source="UNLABELLED") + with pytest.raises(SystemExit, match="RECLASSIFY requires an explicit family"): + _apply( + stream, + _decision(stream, decision="RECLASSIFY", semantic_family="", proposed_family="", lane="C"), + ) + with pytest.raises(SystemExit, match="must differ"): + _apply( + stream, + _decision( + stream, + decision="RECLASSIFY", + semantic_family="gaming-meta", + proposed_family="gaming-meta", + attest="INFERRED", + ), + ) + record = _apply( + stream, + _decision( + stream, + decision="CHOOSE DIFFERENT FAMILY", + semantic_family="internet-slang", + proposed_family="", + attest="OBSERVED", + lane="C", + ), + )["records"][0] + assert record["decision"] == "RECLASSIFY" + assert record["semantic_family"] == "internet-slang" + assert record["source_hint"] == "brainrot-aura" + assert record["attest"] == "OBSERVED" + + +def test_none_produces_semantic_family_none(): + stream = _stream("phrase none", hint="none", label="none") + record = _apply( + stream, + _decision(stream, decision="NONE", semantic_family="", proposed_family="none", attest="INFERRED"), + )["records"][0] + assert record["semantic_family"] == "none" + assert record["decision"] == "NONE" + assert record["attest"] == "INFERRED" + with pytest.raises(SystemExit, match="reject is a production attest token"): + _apply(stream, _decision(stream, decision="reject")) + + +def test_unresolved_remains_unsettled_and_outside_reserve(): + stream = _stream("phrase open") + result = _apply( + stream, + _decision(stream, decision="UNRESOLVED", semantic_family="", attest="", proposed_family="gaming-meta"), + ) + record = result["records"][0] + assert record["semantic_family"] is None + assert record["attest"] is None + assert result["receipt"]["settled_row_count"] == 0 + assert result["receipt"]["unresolved_row_count"] == 1 + assert settlement_allows_reserve(record) is False + with pytest.raises(SystemExit, match="UNRESOLVED requires null"): + _apply(stream, _decision(stream, decision="UNRESOLVED", semantic_family="gaming-meta", attest="")) + + +def test_rights_unresolved_settlement_cannot_enter_reserve(): + stream = _stream( + "phrase reddit", + hint="betting-sharp", + source="reddit_title", + label=None, + label_source="UNLABELLED", + ) + record = _apply( + stream, + _decision( + stream, + lane="D", + semantic_family="betting-sharp", + proposed_family="betting-sharp", + attest="INFERRED", + ), + )["records"][0] + assert record["rights"] == "RIGHTS_UNRESOLVED" + assert record["semantic_family"] == "betting-sharp" + assert settlement_allows_reserve(record) is False + cleared = _apply(_stream("phrase clear"), _decision(_stream("phrase clear"), attest="OBSERVED"))["records"][0] + assert settlement_allows_reserve(cleared) is True + + +def test_inactive_candidate_is_rejected_until_separately_activated(): + stream = _stream("phrase candidate", hint="none", label="none") + with pytest.raises(SystemExit, match="candidate family is inactive"): + _apply(stream, _decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")) + record = settle( + [stream], + [_decision(stream, semantic_family="sexual-romantic", proposed_family="sexual-romantic")], + activated={"sexual-romantic"}, + **STAMP, + )["records"][0] + assert record["semantic_family"] == "sexual-romantic" + assert family_flags("sexual-romantic")["taxonomy.active"] is False + assert family_flags("sexual-romantic", activated={"sexual-romantic"}) == { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + with pytest.raises(SystemExit, match="activation refused"): + settle( + [stream], + [_decision(stream, semantic_family="brainrot-aura", proposed_family="")], + activated={"brainrot-aura"}, + **STAMP, + ) + + +def test_invalid_family_and_attest_are_rejected(): + stream = _stream("phrase bad") + with pytest.raises(SystemExit, match="not in the evaluation taxonomy"): + _apply( + stream, + _decision(stream, semantic_family="workplace-corp", proposed_family="workplace-corp"), + ) + with pytest.raises(SystemExit, match="invalid attest"): + _apply(stream, _decision(stream, attest="UNLABELLED")) + + +def test_row_id_and_hash_mismatch_are_rejected(): + stream = _stream("phrase identity") + with pytest.raises(SystemExit, match="not in the held-out stream"): + _apply(stream, _decision(stream, row_id="hs-missing")) + with pytest.raises(SystemExit, match="text_hash does not match"): + _apply(stream, _decision(stream, text_hash="0" * 64)) + + +def test_duplicate_settlement_is_rejected_and_log_is_append_only(tmp_path): + stream = _stream("phrase once") + decision = _decision(stream) + with pytest.raises(SystemExit, match="duplicate settlement"): + _apply(stream, [decision, decision]) + from hyperlexical.eval_settlement import commit_settlement + + first = _apply(stream, decision) + log = tmp_path / "events.jsonl" + receipt = tmp_path / "receipt.json" + commit_settlement(first["records"], first["receipt"], log_path=log, receipt_path=receipt) + before = log.read_bytes() + with pytest.raises(SystemExit, match="duplicate settlement"): + commit_settlement( + first["records"], + first["receipt"], + log_path=log, + receipt_path=tmp_path / "other.json", + ) + assert log.read_bytes() == before + assert receipt_sha_matches(json.loads(receipt.read_text(encoding="utf-8"))) + assert "phrase once" not in receipt.read_text(encoding="utf-8") + + +def test_legacy_production_attest_apply_behavior_is_unchanged(): + assert PRODUCTION_FAMILIES == ( + "betting-sharp", + "crypto-degen", + "ai-native", + "brainrot-aura", + "kinship-address", + "political-status", + "gaming-meta", + "workplace-corp", + "none", + ) + assert tuple(WEAK_FAMILIES) == PRODUCTION_FAMILIES[:-1] + legacy_valid = set(PRODUCTION_FAMILIES) | {"reject"} + assert "workplace-career" not in legacy_valid + assert "technology-ai" not in legacy_valid + + def legacy_outcome(value: str) -> str: + if value not in legacy_valid: + return "invalid" + if value == "reject": + return "attest_reject" + return "label_source=OBSERVED" + + assert legacy_outcome("gaming-meta") == "label_source=OBSERVED" + assert legacy_outcome("reject") == "attest_reject" + assert legacy_outcome("workplace-career") == "invalid" + module = (ROOT / "scripts" / "shadow" / "hyperlexical" / "eval_settlement.py").read_text(encoding="utf-8") + assert "import layout" not in module + assert ".admit(" not in module + private = Path.home() / "hlx-private" / "heldout-stream" / "bin" / "hs_run.py" + if private.is_file(): + source = private.read_text(encoding="utf-8") + start = source.index("def cmd_attest_apply") + end = source.index("\ndef main()") + body = source[start:end] + assert 'valid = set(FAMILIES) | {"none", "reject"}' in body + assert 'r["label_source"] = "OBSERVED"' in source + assert "workplace-career" not in body + + +def test_blank_sheet_settles_nothing_and_does_not_fill_decisions(tmp_path): + stream = _stream("phrase blank") + sheet = tmp_path / "lane-A.tsv" + _sheet(sheet, stream, decision="", family="", attest="") + before = sheet.read_bytes() + parsed = parse_sheet(sheet, {stream["row_key"]: stream}, **{k: STAMP[k] for k in ("operator", "provenance", "settled_at")}) + assert parsed["records"] == [] + assert parsed["unset_row_count"] == 1 + assert sheet.read_bytes() == before + + +def test_three_taxonomy_flags_stay_distinct(): + assert len(ACTIVE_FAMILIES) == 18 + for name in ACTIVE_FAMILIES: + assert family_flags(name) == { + "taxonomy.active": True, + "evaluation.enabled": False, + "production.enabled": False, + } + for name in CANDIDATE_FAMILIES: + assert family_flags(name)["taxonomy.active"] is False + assert family_flags(name)["evaluation.enabled"] is False + assert set(SURFACES) == {"A", "B", "C", "D", "HELD_OUTSIDE_LANES"} + + +def test_cli_blank_sheet_writes_zero_receipt_without_a_ledger(tmp_path): + stream = _stream("phrase cli") + rows = tmp_path / "rows.jsonl" + rows.write_text(json.dumps(stream) + "\n", encoding="utf-8") + sheet = tmp_path / "holding.tsv" + _sheet(sheet, stream, lane="HELD_OUTSIDE_LANES", proposed="gaming-meta") + receipt = tmp_path / "receipt.json" + log = tmp_path / "events.jsonl" + code = main( + [ + "settlement-apply", + "--stream-rows", + str(rows), + "--sheet", + str(sheet), + "--operator", + "tool-ready", + "--provenance", + "blank-sheet validation", + "--batch-id", + "unit-cli", + "--receipt", + str(receipt), + "--settlement-log", + str(log), + "--settled-at", + "2026-09-26T16:00:00Z", + "--stream-run-id", + "hs-unit", + ] + ) + assert code == 0 + body = json.loads(receipt.read_text(encoding="utf-8")) + assert body["settled_row_count"] == 0 + assert body["reserve_added"] == 0 + assert body["vendor_calls"] == 0 + assert body["ledger_mutated"] is False + assert body["evaluation_enabled"] is False + assert body["stream_run_id"] == "hs-unit" + assert body["settled_at"] == "2026-09-26T16:00:00Z" + assert body["input_sheets"][0]["identity"] == "holding.tsv" + assert len(body["input_sheets"][0]["sha256"]) == 64 + assert body["unset_row_count"] == 1 + assert body["unresolved_row_count"] == 0 + assert log.read_text(encoding="utf-8") == "" + assert not (tmp_path / "ledger.json").exists() + assert "phrase cli" not in receipt.read_text(encoding="utf-8") + + +def test_private_lane_sheets_parse_without_rewriting(): + root = Path.home() / "hlx-private" / "heldout-stream" + stream_path = root / "store" / "rows.jsonl" + attest = root / "attest" + names = { + "A": "lane-A-confirm-20260926.tsv", + "B": "lane-B-proposed-family-20260926.tsv", + "C": "lane-C-disambiguate-20260926.tsv", + "D": "lane-D-rights-blocked-20260926.tsv", + "HELD_OUTSIDE_LANES": "holding-hint-only-outside-lanes-20260926.tsv", + } + if not stream_path.is_file() or not all((attest / name).is_file() for name in names.values()): + pytest.skip("private lane sheets are host-local") + from hyperlexical.eval_settlement import load_stream + + stream = load_stream(stream_path) + expected = {"A": 204, "B": 65, "C": 21, "D": 16, "HELD_OUTSIDE_LANES": 33} + for lane, name in names.items(): + path = attest / name + before = path.read_bytes() + parsed = parse_sheet( + path, + stream, + operator="tool-ready", + provenance="blank-sheet validation", + settled_at="2026-09-26T16:00:00Z", + ) + assert parsed["unset_row_count"] + len(parsed["records"]) == expected[lane] + assert parsed["lane_rows"] == {lane: expected[lane]} + assert path.read_bytes() == before diff --git a/tests/shadow/test_hyperlexical_holdout_disjoint.py b/tests/shadow/test_hyperlexical_holdout_disjoint.py new file mode 100644 index 00000000..03ef720b --- /dev/null +++ b/tests/shadow/test_hyperlexical_holdout_disjoint.py @@ -0,0 +1,273 @@ +"""Pinned training rows and a controlled holdout must be disjoint.""" + +from __future__ import annotations + +import hashlib +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.heldout_census import normalize_group_text # noqa: E402 +from hyperlexical.holdout_eligibility import census, exclusion_reason # noqa: E402 +from hyperlexical.holdout_guard import ( # noqa: E402 + assert_pinned_holdout_disjoint, + filter_holdout_rows, + load_holdout_spec, + normalized_text_sha256, + require_holdout_for_training, +) +from hyperlexical.preflight import main as preflight_main # noqa: E402 +from hyperlexical.selection_surface import row_id # noqa: E402 +from hyperlexical.train_input import load_training_bundle # noqa: E402 + + +def _row(text, *, split="train", task="classify", cls="OBSERVED", lineage="brainrot"): + return { + "text": text, + "fillers": ["atom"], + "roles": ["pos_0"], + "role_scheme": "positional", + "class": cls, + "split": split, + "task": task, + "lineage": lineage, + } + + +def _sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _write_pin(path: Path, rows: list[dict]) -> str: + payload = "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows) + path.write_text(payload, encoding="utf-8") + return _sha(path) + + +def _manifest(path: Path, rows: list[dict], *, experiment_id: str = "HLX-EXP-TEST") -> Path: + path.write_text( + json.dumps( + { + "schema": "hyperlex.holdout_manifest.v1", + "status": "UNSCORED_SEALED", + "experiment_id": experiment_id, + "row_ids": [row_id(row) for row in rows], + "normalized_text_sha256": [normalized_text_sha256(row["text"]) for row in rows], + } + ), + encoding="utf-8", + ) + return path + + +def _arm(monkeypatch, tmp_path: Path, rows: list[dict], manifest: Path) -> None: + pinned = tmp_path / "pinned.jsonl" + digest = _write_pin(pinned, rows) + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(pinned)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", digest) + monkeypatch.setenv("HLX_TRAIN_EXPORT_ROWS", str(len(rows))) + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", str(manifest)) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + + +def test_same_row_id_is_not_eligible(): + row = _row("cobalt lantern") + reason = exclusion_reason( + row, + train_ids={row_id(row)}, + train_hashes=set(), + other_ids=set(), + other_hashes=set(), + ) + assert reason == "train_row_id" + + +def test_same_normalized_text_different_row_id_is_not_eligible(): + train = _row("Blue Quartz Lantern!", split="train") + hold = _row("blue quartz lantern", split="val", cls="INFERRED", lineage="none") + assert row_id(train) != row_id(hold) + assert normalized_text_sha256(train["text"]) == normalized_text_sha256(hold["text"]) + reason = exclusion_reason( + hold, + train_ids=set(), + train_hashes={normalized_text_sha256(train["text"])}, + other_ids=set(), + other_hashes=set(), + ) + assert reason == "train_text_hash" + + +def test_different_text_is_eligible(): + row = _row("north cobble path", split="val") + reason = exclusion_reason( + row, + train_ids=set(), + train_hashes={normalized_text_sha256("other phrase")}, + other_ids=set(), + other_hashes=set(), + ) + assert reason == "eligible" + + +def test_draw_and_filter_share_the_canonical_normalizer(): + surface = "Hello, World! https://example.com/a" + assert normalize_group_text(surface) == "hello world" + digest = hashlib.sha256(normalize_group_text(surface).encode("utf-8")).hexdigest() + assert digest == normalized_text_sha256(surface) + assert normalized_text_sha256("Hello World") == digest + + +def test_text_collision_fails_controlled_preflight(monkeypatch, tmp_path, capsys): + train = [_row(f"train row {i}") for i in range(3)] + collided = _row("train row 1", split="val", task="unbind", lineage="none") + manifest = _manifest(tmp_path / "holdout.json", [collided]) + _arm(monkeypatch, tmp_path, train, manifest) + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr("hyperlexical.preflight.export_dataset", lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("export"))) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert "CONTROLLED_RESERVE" in report["error"] + + +def test_row_id_collision_fails_controlled_preflight(monkeypatch, tmp_path, capsys): + train = [_row("unique cobalt")] + manifest = _manifest(tmp_path / "holdout.json", train) + _arm(monkeypatch, tmp_path, train, manifest) + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr( + "hyperlexical.preflight.export_dataset", + lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("export")), + ) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert "CONTROLLED_RESERVE" in report["error"] + + +def test_disjoint_holdout_is_admissible(monkeypatch, tmp_path, capsys): + train = [_row(f"train row {i}") for i in range(4)] + hold = _row("fresh holdout phrase", split="val") + manifest = _manifest(tmp_path / "holdout.json", [hold]) + _arm(monkeypatch, tmp_path, train, manifest) + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr( + "hyperlexical.preflight.export_dataset", + lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("export")), + ) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert report["status"] != "TRAINING_READY" + assert "CONTROLLED_RESERVE" in report["error"] + + +def test_pinned_disjoint_holdout_keeps_all_9150_rows(monkeypatch, tmp_path): + rows = [_row(f"pinned sentence {i}") for i in range(9150)] + hold = _row("outside the pin", split="val", task="unbind", lineage="none") + manifest = _manifest(tmp_path / "holdout.json", [hold]) + _arm(monkeypatch, tmp_path, rows, manifest) + called = {"n": 0} + + def _export(*_args, **_kwargs): + called["n"] += 1 + raise AssertionError("export_dataset") + + bundle = load_training_bundle( + tmp_path, + include_live=True, + live_store=None, + export_dataset=_export, + ) + assert called["n"] == 0 + assert len(bundle["rows"]) == 9150 + spec = load_holdout_spec() + report = assert_pinned_holdout_disjoint(bundle["rows"], spec) + kept, removed = filter_holdout_rows(list(bundle["rows"]), spec) + assert report["holdout_filter_training_rows_removed"] == 0 + assert removed == 0 + assert len(kept) == 9150 + + +def test_colliding_holdout_never_reaches_the_trainer(monkeypatch, tmp_path): + train = [_row("shared surface")] + hold = _row("Shared Surface", split="val") + manifest = _manifest(tmp_path / "holdout.json", [hold]) + _arm(monkeypatch, tmp_path, train, manifest) + bundle = load_training_bundle( + tmp_path, + include_live=True, + live_store=None, + export_dataset=lambda *_a, **_k: (_ for _ in ()).throw(AssertionError("export")), + ) + spec = load_holdout_spec() + reached = {"trainer": False} + with pytest.raises(SystemExit, match="ADMISSION FAIL"): + assert_pinned_holdout_disjoint(bundle["rows"], spec) + reached["trainer"] = True + assert reached["trainer"] is False + kept, removed = filter_holdout_rows(list(bundle["rows"]), spec) + assert removed == 1 + assert len(kept) == 0 + + +def test_abandoned_lifecycle_is_not_admissible(monkeypatch, tmp_path): + manifest = _manifest(tmp_path / "holdout.json", [_row("historical phrase", split="val")]) + digest = _sha(manifest) + before = manifest.read_bytes() + (tmp_path / "holdout-lifecycle.json").write_text( + json.dumps( + { + "schema": "hyperlex.holdout_lifecycle.v1", + "status": "UNSCORED_ABANDONED", + "reason": "EXPERIMENT_CLOSED_BEFORE_VALID_EXECUTION", + "manifest_sha256": digest, + "reusable": False, + "scored": False, + } + ), + encoding="utf-8", + ) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", str(manifest)) + with pytest.raises(SystemExit, match="UNSCORED_ABANDONED"): + require_holdout_for_training() + assert manifest.read_bytes() == before + + +def test_census_excludes_text_collisions_and_abandoned_ids(): + train = [_row("shared surface"), _row("train only")] + abandoned = _row("abandoned phrase", split="val") + fresh = _row("fresh phrase", split="val", task="unbind", lineage="none") + source = [ + _row("test row", split="test"), + train[0], + _row("Shared Surface", split="val", cls="INFERRED"), + abandoned, + fresh, + ] + report = census(source, train, extra_rows=[abandoned]) + assert report["source_rows"] == 5 + assert report["excluded_by_split_test"] == 1 + assert report["excluded_by_training_row_id"] == 1 + assert report["excluded_by_training_normalized_text_hash"] == 1 + assert report["excluded_by_spent_or_abandoned"] == 1 + assert report["eligible_rows"] == 1 + assert report["unbind_eligible"] == 1 + assert report["waterfall_sums_to_source"] is True diff --git a/tests/shadow/test_hyperlexical_holdout_guard.py b/tests/shadow/test_hyperlexical_holdout_guard.py index 1ebbf596..af1e24a5 100644 --- a/tests/shadow/test_hyperlexical_holdout_guard.py +++ b/tests/shadow/test_hyperlexical_holdout_guard.py @@ -19,6 +19,7 @@ holdout_receipt, load_holdout_spec, normalized_text_sha256, + require_holdout_for_training, ) from hyperlexical.layout import label_maps # noqa: E402 from hyperlexical.loop import prepare_unbind_splits, run_loop # noqa: E402 @@ -35,6 +36,10 @@ def _clear_holdout_env(monkeypatch): for key in ( "HLX_HOLDOUT_MANIFESTS", "HLX_ALLOW_NO_HOLDOUT", + "HLX_EXPERIMENT_ID", + "HLX_TRAIN_EXPORT_PATH", + "HLX_TRAIN_EXPORT_SHA256", + "HLX_TRAIN_EXPORT_ROWS", "HYPERLEX_ALLOW_TRAIN", "HYPERLEX_RELEASE_SET", "HYPERLEX_TASK_ROUTING", @@ -62,17 +67,24 @@ def _row(text, fillers, *, split="val", cls="OBSERVED", task="unbind"): } -def _write_manifest(path: Path, *, row_ids=(), text_hashes=()) -> Path: - path.write_text( - json.dumps( - { - "schema": "hyperlex.holdout_manifest.v2", - "row_ids": list(row_ids), - "normalized_text_sha256": list(text_hashes), - } - ), - encoding="utf-8", - ) +def _write_manifest( + path: Path, + *, + row_ids=(), + text_hashes=(), + status=None, + experiment_id=None, +) -> Path: + body = { + "schema": "hyperlex.holdout_manifest.v2", + "row_ids": list(row_ids), + "normalized_text_sha256": list(text_hashes), + } + if status is not None: + body["status"] = status + if experiment_id is not None: + body["experiment_id"] = experiment_id + path.write_text(json.dumps(body), encoding="utf-8") return path @@ -277,24 +289,143 @@ def test_cli_manifest_and_run_loop_log(monkeypatch, tmp_path, capsys): tmp_path / "cli.json", row_ids=[row_id(unbind)], text_hashes=[normalized_text_sha256(surface)], + status="UNSCORED_SEALED", + experiment_id="HLX-EXP-TEST", ) + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(tmp_path)) monkeypatch.setenv("HYPERLEX_EXPORT_DIR", str(tmp_path / "export")) (tmp_path / "config.json").write_text("{}\n", encoding="utf-8") + pinned = tmp_path / "pinned.jsonl" + payload = "".join(json.dumps(row, sort_keys=True) + "\n" for row in (classify, unbind)) + pinned.write_text(payload, encoding="utf-8") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(pinned)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", hashlib.sha256(pinned.read_bytes()).hexdigest()) def _export(*_args, **_kwargs): - return {"rows": [classify, unbind], "sha256": "abc", "counts": {}} + raise AssertionError("export_dataset must not run for a pinned experiment") + + written = {"n": 0} + + def _write(*_args, **_kwargs): + written["n"] += 1 monkeypatch.setattr("hyperlexical.loop.export_dataset", _export) - monkeypatch.setattr("hyperlexical.loop.write_export", lambda *_a, **_k: None) - assert train_mod.main(["--offline", "--run", "--holdout-manifest", str(manifest)]) == 4 + monkeypatch.setattr("hyperlexical.loop.write_export", _write) + with pytest.raises(SystemExit, match="ADMISSION FAIL"): + train_mod.main(["--offline", "--run", "--holdout-manifest", str(manifest)]) + assert written["n"] == 0 captured = capsys.readouterr() - logged = captured.out - assert "not enough classify" in captured.err - file_sha = hashlib.sha256(manifest.read_bytes()).hexdigest() - assert f"manifest_sha256={file_sha}" in logged - assert "classify_val=1" in logged - assert "unbind_val=1" in logged - assert "blue quartz" not in logged - assert surface not in logged + assert "blue quartz" not in captured.out + assert surface not in captured.out + assert surface not in captured.err + + +def _admit(monkeypatch, path: Path, experiment_id: str = "HLX-EXP-TEST") -> None: + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_EXPERIMENT_ID", experiment_id) + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", str(path)) + + +def test_unscored_sealed_with_matching_binding_is_admissible(monkeypatch, tmp_path): + manifest = _write_manifest( + tmp_path / "fresh.json", + row_ids=["abc123"], + status="UNSCORED_SEALED", + experiment_id="HLX-EXP-TEST", + ) + _admit(monkeypatch, manifest) + spec = require_holdout_for_training() + assert spec.row_ids == frozenset({"abc123"}) + assert spec.manifests[0]["status"] == "UNSCORED_SEALED" + assert spec.manifests[0]["experiment_id"] == "HLX-EXP-TEST" + + +@pytest.mark.parametrize("status", ["SCORED_SPENT", "SCORED", "FROZEN_NOT_SCORED"]) +def test_spent_scored_and_unknown_status_are_rejected(monkeypatch, tmp_path, status): + manifest = _write_manifest( + tmp_path / "closed.json", + row_ids=["abc123"], + status=status, + experiment_id="HLX-EXP-TEST", + ) + _admit(monkeypatch, manifest) + with pytest.raises(SystemExit, match="not admissible"): + require_holdout_for_training() + + +def test_missing_status_is_rejected(monkeypatch, tmp_path): + manifest = _write_manifest(tmp_path / "bare.json", row_ids=["abc123"]) + _admit(monkeypatch, manifest) + with pytest.raises(SystemExit, match="status is missing"): + require_holdout_for_training() + + +def test_wrong_experiment_binding_is_rejected(monkeypatch, tmp_path): + manifest = _write_manifest( + tmp_path / "wrong.json", + row_ids=["abc123"], + status="UNSCORED_SEALED", + experiment_id="HLX-EXP-OTHER", + ) + _admit(monkeypatch, manifest, "HLX-EXP-TEST") + with pytest.raises(SystemExit, match="experiment binding"): + require_holdout_for_training() + + +def test_unbound_experiment_is_rejected(monkeypatch, tmp_path): + manifest = _write_manifest( + tmp_path / "unbound.json", + row_ids=["abc123"], + status="UNSCORED_SEALED", + ) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", str(manifest)) + with pytest.raises(SystemExit, match="experiment binding"): + require_holdout_for_training() + + +def test_no_holdout_is_rejected_when_override_is_unset(monkeypatch): + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + with pytest.raises(SystemExit, match="no holdout manifest"): + require_holdout_for_training() + + +def test_spent_companion_does_not_admit(monkeypatch, tmp_path): + fresh = _write_manifest( + tmp_path / "fresh.json", + row_ids=["abc123"], + status="UNSCORED_SEALED", + experiment_id="HLX-EXP-TEST", + ) + spent = _write_manifest( + tmp_path / "spent.json", + row_ids=["def456"], + status="SCORED_SPENT", + experiment_id="HLX-EXP-TEST", + ) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", f"{fresh},{spent}") + with pytest.raises(SystemExit, match="SCORED_SPENT"): + require_holdout_for_training() + + +def test_spent_manifest_still_excludes_rows(monkeypatch, tmp_path): + secret = _row("blue quartz lantern", ["quartz"], split="val") + keep = _row("north cobble path", ["cobble"], split="val") + manifest = _write_manifest( + tmp_path / "spent.json", + row_ids=[row_id(secret)], + status="SCORED_SPENT", + experiment_id="HLX-EXP-TEST", + ) + _arm(monkeypatch, manifest) + _train, val, stats = prepare_unbind_splits([secret, keep]) + assert [r["text"] for r in val] == ["north cobble path"] + assert stats["n_holdout_removed_val"] == 1 + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + with pytest.raises(SystemExit, match="SCORED_SPENT"): + require_holdout_for_training() diff --git a/tests/shadow/test_hyperlexical_identity_ledger.py b/tests/shadow/test_hyperlexical_identity_ledger.py new file mode 100644 index 00000000..c4a060cb --- /dev/null +++ b/tests/shadow/test_hyperlexical_identity_ledger.py @@ -0,0 +1,202 @@ +"""Text identity owns contamination. Admission never consults a model score.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.holdout_guard import normalized_text_sha256 # noqa: E402 +from hyperlexical.identity_ledger import ( # noqa: E402 + PLANNING_TARGETS, + IdentityLedger, + acquisition_gap, + assert_training_disjoint_from_reserve, + derived_state, +) +from hyperlexical.selection_surface import row_id # noqa: E402 + +TINY = { + "classify": 1, + "classify_observed": 1, + "classify_non_none": 1, + "unbind_clean": 1, +} + + +def _row(text, *, split="train", task="classify", cls="OBSERVED", lineage="brainrot", declared=None): + row = { + "text": text, + "fillers": ["atom"], + "roles": ["pos_0"], + "role_scheme": "positional", + "class": cls, + "split": split, + "task": task, + "lineage": lineage, + } + if declared is not None: + row["row_id"] = declared + return row + + +def test_identity_function_is_the_holdout_guard(): + assert normalized_text_sha256("Hello, WORLD") == normalized_text_sha256("hello world") + digest = normalized_text_sha256("Hello, WORLD") + ledger = IdentityLedger() + ledger.observe_row(_row("Hello, WORLD"), source_artifact="unit", provenance="unit", catalogued=True) + assert digest in ledger.identities + blob = json.dumps(ledger.project()) + assert "Hello" not in blob + assert "WORLD" not in blob + + +def test_historical_blocks_are_monotonic(): + ledger = IdentityLedger() + row = _row("consumed phrase alpha") + digest = ledger.observe_row(row, source_artifact="pin", provenance="pin", catalogued=True) + ledger.mark_historical(digest, "training_consumed", source_artifact="pin", provenance="trained") + assert derived_state(ledger.identity(digest)) == "TRAIN_CONSUMED" + with pytest.raises(SystemExit, match="monotonic"): + ledger.transition(digest, "EVAL_RESERVE", source_artifact="nope", provenance="nope") + spent = ledger.observe_row(_row("spent phrase beta"), source_artifact="v2", provenance="v2", catalogued=True) + ledger.mark_historical(spent, "evaluation_spent", source_artifact="v2", provenance="spent") + with pytest.raises(SystemExit, match="monotonic"): + ledger.transition(spent, "EVAL_RESERVE", source_artifact="nope", provenance="nope") + abandoned = ledger.observe_row( + _row("abandoned phrase gamma"), source_artifact="select", provenance="select", catalogued=True + ) + ledger.mark_historical( + abandoned, "evaluation_abandoned", source_artifact="select", provenance="abandoned" + ) + with pytest.raises(SystemExit, match="monotonic"): + ledger.transition(abandoned, "EVAL_RESERVE", source_artifact="nope", provenance="nope") + + +def test_generation_reset_does_not_clear_flags(): + ledger = IdentityLedger() + digest = ledger.observe_row(_row("keep this trained"), source_artifact="pin", provenance="pin", catalogued=True) + ledger.mark_historical(digest, "training_consumed", source_artifact="pin", provenance="trained") + with pytest.raises(SystemExit, match="generation reset"): + ledger.request_generation_reset(governed_receipt="", confirm=False) + result = ledger.request_generation_reset(governed_receipt="receipts/governed.json", confirm=True) + assert result["applied"] is False + assert result["flags_cleared"] == 0 + assert ledger.identity(digest)["training_consumed"] is True + assert derived_state(ledger.identity(digest)) == "TRAIN_CONSUMED" + + +def test_admit_reserves_by_slice_not_by_score(): + ledger = IdentityLedger() + rows = [ + _row("observed classify one", cls="OBSERVED", lineage="brainrot"), + _row("inferred classify two", cls="INFERRED", lineage="brainrot"), + _row("clean unbind", task="unbind", cls="INFERRED", lineage="brainrot"), + _row("second clean unbind", task="unbind", cls="INFERRED", lineage="brainrot"), + ] + clean = {normalized_text_sha256(rows[2]["text"]), normalized_text_sha256(rows[3]["text"])} + report = ledger.admit( + rows, + batch_id="batch-a", + source_artifact="acquire", + targets=TINY, + unbind_clean_hashes=clean, + ) + assert report["raw_rows"] == 4 + assert report["unique_canonical_text_identities"] == 4 + assert report["unique_admitted_to_eval_reserve"] + report["unique_routed_to_train_candidate"] == 4 + assert report["unique_routed_to_train_candidate"] >= 1 + assert report["model_outcomes_consulted"] is False + counts = ledger.reserve_counts() + assert counts["classify_observed"] == 1 + assert counts["classify_non_none"] >= 1 + assert counts["unbind_clean"] == 1 + + +def test_duplicate_text_does_not_increase_reserve_and_new_text_with_old_id_is_separate(): + ledger = IdentityLedger() + first = _row("same surface", cls="OBSERVED", lineage="brainrot") + report = ledger.admit( + [first], + batch_id="batch-b", + source_artifact="acquire", + targets=TINY, + ) + assert report["unique_admitted_to_eval_reserve"] == 1 + twin = _row("same surface", split="val", cls="OBSERVED", lineage="brainrot") + assert row_id(twin) != row_id(first) + again = ledger.admit([twin], batch_id="batch-c", source_artifact="acquire", targets=TINY) + assert again["unique_admitted_to_eval_reserve"] == 0 + assert again["rejected_existing_identities"]["EVAL_RESERVE"] == 1 + assert ledger.reserve_counts()["classify"] == 1 + fresh = _row("different surface", cls="INFERRED", lineage="brainrot", declared=row_id(first)) + collided = ledger.admit([fresh], batch_id="batch-d", source_artifact="acquire", targets=TINY) + assert collided["row_id_collisions_with_new_text"] == 1 + assert collided["unique_admitted_to_eval_reserve"] == 0 + assert collided["unique_routed_to_train_candidate"] == 1 + assert normalized_text_sha256(fresh["text"]) != normalized_text_sha256(first["text"]) + + +def test_blocked_hashes_are_rejected_and_scores_are_refused(): + ledger = IdentityLedger() + consumed = _row("already trained") + digest = ledger.observe_row(consumed, source_artifact="pin", provenance="pin", catalogued=True) + ledger.mark_historical(digest, "training_consumed", source_artifact="pin", provenance="trained") + report = ledger.admit([consumed], batch_id="batch-e", source_artifact="acquire", targets=TINY) + assert report["unique_admitted_to_eval_reserve"] == 0 + assert report["rejected_existing_identities"]["TRAIN_CONSUMED"] == 1 + scored = _row("brand new phrase") + scored["model_score"] = 0.9 + with pytest.raises(SystemExit, match="model outcome"): + ledger.admit([scored], batch_id="batch-f", source_artifact="acquire", targets=TINY) + + +def test_training_guard_blocks_reserve_and_ignores_historical_overlap(): + ledger = IdentityLedger() + reserved = _row("held for eval", cls="OBSERVED", lineage="brainrot") + ledger.admit([reserved], batch_id="batch-g", source_artifact="acquire", targets=TINY) + with pytest.raises(SystemExit, match="ADMISSION FAIL"): + assert_training_disjoint_from_reserve([reserved], ledger) + historical = _row("old train and old holdout") + digest = ledger.observe_row(historical, source_artifact="pin", provenance="pin", catalogued=True) + ledger.mark_historical(digest, "training_consumed", source_artifact="pin", provenance="trained") + ledger.mark_historical(digest, "evaluation_abandoned", source_artifact="select", provenance="abandoned") + report = assert_training_disjoint_from_reserve([historical], ledger) + assert report["eval_reserve_training_disjoint"] is True + + +def test_roundtrip_and_gap_labels(tmp_path): + ledger = IdentityLedger() + ledger.admit( + [_row("only one", cls="OBSERVED", lineage="brainrot")], + batch_id="batch-h", + source_artifact="acquire", + targets=TINY, + ) + ledger.save(tmp_path) + loaded = IdentityLedger.load(tmp_path) + assert loaded.reserve_counts()["classify_observed"] == 1 + blob = (tmp_path / "ledger.json").read_text(encoding="utf-8") + assert '"text":' not in blob + gap = acquisition_gap({}) + assert [row["minimum_required"] for row in gap] == ["NOT_COMPUTABLE"] * 4 + assert {row["slice"]: row["planning_target"] for row in gap} == PLANNING_TARGETS + gate = loaded.select_003_gate(train_hashes=set()) + assert gate["select_003"] == "NOT_DRAFTED" + assert gate["eligible"] is False + + +def test_same_batch_duplicate_text_counts_once(): + ledger = IdentityLedger() + row = _row("batch twin", cls="OBSERVED", lineage="brainrot") + twin = _row("batch twin", split="val", cls="OBSERVED", lineage="brainrot") + report = ledger.admit([row, twin], batch_id="batch-i", source_artifact="acquire", targets=TINY) + assert report["raw_rows"] == 2 + assert report["unique_canonical_text_identities"] == 1 + assert report["unique_admitted_to_eval_reserve"] == 1 + assert report["duplicate_text_rows_not_counted"] == 1 diff --git a/tests/shadow/test_hyperlexical_train_input.py b/tests/shadow/test_hyperlexical_train_input.py new file mode 100644 index 00000000..22515ad8 --- /dev/null +++ b/tests/shadow/test_hyperlexical_train_input.py @@ -0,0 +1,260 @@ +"""Pinned export is the training input. Existence of a file is not use.""" + +from __future__ import annotations + +import hashlib +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.loop import run_loop # noqa: E402 +from hyperlexical.preflight import main as preflight_main # noqa: E402 +from hyperlexical.training_routing import route_rows as real_route_rows # noqa: E402 +from hyperlexical.train_input import train_input_receipt as real_train_input_receipt # noqa: E402 + + +@pytest.fixture(autouse=True) +def _clear_pin_env(monkeypatch): + for key in ( + "HLX_TRAIN_EXPORT_PATH", + "HLX_TRAIN_EXPORT_SHA256", + "HLX_TRAIN_EXPORT_ROWS", + "HLX_EXPERIMENT_ID", + "HLX_HOLDOUT_MANIFESTS", + "HLX_ALLOW_NO_HOLDOUT", + "HYPERLEX_ALLOW_TRAIN", + "HYPERLEX_INCLUDE_LIVE", + "HYPERLEX_EXPORT_DIR", + "HYPERLEX_RELEASE_SET", + "HYPERLEX_TASK_ROUTING", + "HYPERLEX_TRUNK_DIR", + "HYPERLEX_LIVE_STORE", + ): + monkeypatch.delenv(key, raising=False) + + +def _classify(text: str) -> dict: + return { + "text": text, + "task": "classify", + "split": "train", + "class": "OBSERVED", + "lineage": "none", + "role_scheme": "positional", + } + + +def _write_export(path: Path, rows: list[dict]) -> str: + payload = "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows) + path.write_text(payload, encoding="utf-8") + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _refuse_export(*_args, **_kwargs): + raise AssertionError("export_dataset must not run") + + +def _pin(monkeypatch, tmp_path: Path, rows: list[dict], *, rows_env: str | None = None) -> tuple[Path, str]: + path = tmp_path / "pinned.jsonl" + digest = _write_export(path, rows) + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(path)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", digest) + monkeypatch.setenv("HYPERLEX_EXPORT_DIR", str(tmp_path / "export-out")) + if rows_env is not None: + monkeypatch.setenv("HLX_TRAIN_EXPORT_ROWS", rows_env) + return path, digest + + +def test_pinned_export_is_consumed_and_recorded(monkeypatch, tmp_path): + marker = "pinned quartz marker" + path, digest = _pin(monkeypatch, tmp_path, [_classify(marker)], rows_env="1") + seen: dict = {} + + def _route(rows): + seen["texts"] = [row.get("text") for row in rows] + return real_route_rows(rows) + + def _receipt(bundle): + fields = real_train_input_receipt(bundle) + seen["receipt"] = fields + seen["loaded_texts"] = [row.get("text") for row in bundle["rows"]] + return fields + + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + monkeypatch.setattr("hyperlexical.loop.route_rows", _route) + monkeypatch.setattr("hyperlexical.loop.train_input_receipt", _receipt) + with pytest.raises(RuntimeError, match="not enough classify train rows"): + run_loop( + tmp_path, + tmp_path / "out", + include_live=True, + live_store=tmp_path / "live-would-be-used.jsonl", + ) + assert seen["texts"] == [marker] + assert seen["loaded_texts"] == [marker] + receipt = seen["receipt"] + assert receipt["training_input_mode"] == "PINNED_EXPORT" + assert receipt["training_export_path"] == str(path.resolve()) + assert receipt["training_export_sha256_expected"] == digest + assert receipt["training_export_sha256_actual"] == digest + assert receipt["data_sha256"] == digest + assert receipt["training_export_rows"] == 1 + assert receipt["live_export_generation_enabled"] is False + + +def test_pinned_export_missing_is_rejected(monkeypatch, tmp_path): + missing = tmp_path / "absent.jsonl" + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(missing)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", "a" * 64) + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + with pytest.raises(SystemExit, match="pinned export missing"): + run_loop(tmp_path, tmp_path / "out", include_live=True) + + +def test_pinned_export_digest_mismatch_is_rejected(monkeypatch, tmp_path): + path = tmp_path / "pinned.jsonl" + _write_export(path, [_classify("pinned quartz marker")]) + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HLX_TRAIN_EXPORT_PATH", str(path)) + monkeypatch.setenv("HLX_TRAIN_EXPORT_SHA256", "b" * 64) + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + with pytest.raises(SystemExit, match="digest mismatch"): + run_loop(tmp_path, tmp_path / "out", include_live=True) + + +def test_experiment_without_pinned_export_is_rejected(monkeypatch, tmp_path): + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + with pytest.raises(SystemExit, match="no pinned export"): + run_loop(tmp_path, tmp_path / "out", include_live=True, live_store=tmp_path / "store.jsonl") + + +def test_pinned_row_count_mismatch_is_rejected(monkeypatch, tmp_path): + _pin(monkeypatch, tmp_path, [_classify("pinned quartz marker")], rows_env="9150") + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + with pytest.raises(SystemExit, match="row count mismatch"): + run_loop(tmp_path, tmp_path / "out") + + +def test_live_build_outside_controlled_experiment_still_exports(monkeypatch, tmp_path): + captured: dict = {} + + def fake_export(root, include_live=False, live_store=None): + captured["include_live"] = include_live + captured["live_store"] = live_store + raise RuntimeError("stop-before-torch") + + monkeypatch.setattr("hyperlexical.loop.export_dataset", fake_export) + with pytest.raises(RuntimeError, match="stop-before-torch"): + run_loop(tmp_path, tmp_path / "out", include_live=True, live_store=tmp_path / "store.jsonl") + assert captured["include_live"] is True + assert captured["live_store"] == tmp_path / "store.jsonl" + + + +def _seal_holdout(tmp_path: Path, experiment_id: str = "HLX-EXP-TEST") -> Path: + path = tmp_path / "holdout.json" + path.write_text( + json.dumps( + { + "schema": "hyperlex.holdout_manifest.v2", + "status": "UNSCORED_SEALED", + "experiment_id": experiment_id, + "row_ids": ["abc123"], + } + ), + encoding="utf-8", + ) + return path + + +def test_preflight_pinned_proves_loader_without_export(monkeypatch, tmp_path, capsys): + """A legacy manifest does not admit a controlled experiment.""" + _pin(monkeypatch, tmp_path, [_classify("pinned quartz marker")], rows_env="1") + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + manifest = _seal_holdout(tmp_path) + monkeypatch.setenv("HLX_HOLDOUT_MANIFESTS", str(manifest)) + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr("hyperlexical.preflight.export_dataset", _refuse_export) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert report["ready_to_train"] is False + assert report["holdout_admitted"] is False + assert "CONTROLLED_RESERVE" in report["error"] + + +def test_preflight_controlled_without_holdout_is_rejected(monkeypatch, tmp_path, capsys): + _pin(monkeypatch, tmp_path, [_classify("pinned quartz marker")], rows_env="1") + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr("hyperlexical.preflight.export_dataset", _refuse_export) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert report["ready_to_train"] is False + assert report["holdout_admitted"] is False + assert "no sealed evaluation reserve" in report["error"] + + +def test_pinned_experiment_without_holdout_is_rejected(monkeypatch, tmp_path): + _pin(monkeypatch, tmp_path, [_classify("pinned quartz marker")], rows_env="1") + (tmp_path / "config.json").write_text("{}", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(tmp_path)) + monkeypatch.setattr("hyperlexical.loop.export_dataset", _refuse_export) + with pytest.raises(SystemExit, match="no sealed evaluation reserve"): + run_loop(tmp_path, tmp_path / "out", include_live=True) + + +def test_preflight_experiment_without_pin_is_not_training_ready(monkeypatch, tmp_path, capsys): + monkeypatch.setenv("HLX_EXPERIMENT_ID", "HLX-EXP-TEST") + monkeypatch.setenv("HYPERLEX_ALLOW_TRAIN", "1") + trunk = tmp_path / "trunk" + trunk.mkdir() + (trunk / "config.json").write_text("{}\n", encoding="utf-8") + monkeypatch.setenv("HYPERLEX_TRUNK_DIR", str(trunk)) + monkeypatch.setattr("hyperlexical.preflight.export_dataset", _refuse_export) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert report["status"] == "ADMISSION_FAIL" + assert report["ready_to_train"] is False + assert report["status"] != "TRAINING_READY" + assert "no sealed evaluation reserve" in report["error"] + + +def test_preflight_live_build_still_calls_export(monkeypatch, tmp_path, capsys): + captured: dict = {} + + def fake_export(root, include_live=False, live_store=None): + captured["include_live"] = include_live + captured["called"] = True + return { + "rows": [], + "sha256": "abc", + "counts": {"n": 0, "live_included": 0}, + "payload": "", + } + + monkeypatch.setattr("hyperlexical.preflight.export_dataset", fake_export) + assert preflight_main() == 2 + report = json.loads(capsys.readouterr().out) + assert captured["called"] is True + assert captured["include_live"] is False + assert report["training_input_mode"] == "LIVE_BUILD" + assert report["live_export_generation_enabled"] is True + assert report["data_sha256"] == "abc" + assert report["status"] != "TRAINING_READY"