diff --git a/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py new file mode 100644 index 00000000..efd0deed --- /dev/null +++ b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py @@ -0,0 +1,685 @@ +"""Review residual-v1 coverage, then resolve constituent senses on the 225 rows. + +The resolver specification is written before any constituent is resolved. Operator +labels are read only after the resolution artifact is hashed. Residual scores are +not recomputed. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter, defaultdict +from pathlib import Path + +from hyperlexical.constituent_sense_resolution_v1 import ( + RELATION_EXPANSION_SYMBOLS, + RULE_VERSION, + coverage_decision, + lesk_tokens, + overlap_score, + resolve_constituent_sense, + resolver_policy, + row_resolution_status, + structural_synset_ids, +) +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + neighbor_keys, + resolve_constituent, +) +from hyperlexical.semantic_compositionality_residual_replay import EXPECTED as RESIDUAL_EXPECTED +from hyperlexical.unbind_sense_screen_v1 import load_exceptions, load_wordnet, parse_data_line + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +REVIEW_PATH = SOURCE / "RESIDUAL_V1_DEVELOPMENT_RESULT_REVIEW.json" +LIMITATION_PATH = SOURCE / "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION.json" +SPEC_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_SPEC.json" +PROCEDURE_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_PROCEDURE.json" +REPLAY_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_REPLAY.jsonl" +ANALYSIS_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_DEVELOPMENT_ANALYSIS.json" +DECISION_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_DECISION.json" + +RESIDUAL_SPEC = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +RESIDUAL_PROVENANCE = SOURCE / "RESIDUAL_SOURCE_PROVENANCE.json" +RESIDUAL_SCORES = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +RESIDUAL_EVALUATION = SOURCE / "RESIDUAL_DEVELOPMENT_EVALUATION.json" +RESIDUAL_DECISION = SOURCE / "RESIDUAL_CANDIDATE_DECISION.json" +RESIDUAL_LICENSE = SOURCE / "RESIDUAL_LICENSE_RECEIPT.json" + +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +OPERATOR_COUNTS = {"HIGH": 101, "SECONDARY": 46, "REJECT": 77, "QUARANTINE": 1, "UNRESOLVED": 0} +POS_NAME = {"n": "noun", "v": "verb", "a": "adj", "r": "adv", "s": "adj"} +FILES = {"noun": "data.noun", "verb": "data.verb", "adj": "data.adj", "adv": "data.adv"} +BASELINE_COUNTS = {"AMBIGUOUS": 370, "EXACT": 4, "UNIQUE": 60, "UNRESOLVED": 70} +ROW_FIELDS = ("row_id", "surface", "pos", "gloss", "synset_offset", "synset_pos", "sense_class") + +EXPECTED = dict(RESIDUAL_EXPECTED) +EXPECTED.update( + { + TRACKER: "14d4daaab1e4740a596acaef7f4dbd9ed11edaf2182ff22f7aa339325febf591", + RESIDUAL_SPEC: "39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d", + RESIDUAL_PROVENANCE: "2c34a7d0d480cde564bda694dbaa349550814c5fb7c647bfa3bbbc9db5e26886", + RESIDUAL_SCORES: "cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7", + RESIDUAL_EVALUATION: "ba782622d4c68d23c53ae0ffb5f54f1e43c8cf059b44b7adb34e0eb356ef3896", + RESIDUAL_DECISION: "45922eba294b0b7d7238ce71c6157641259ea292f65743ba2ee1324ff9b52617", + RESIDUAL_LICENSE: "323a5bad21ac74d6a1fd94c54dba95a0046b633cd7328dd6680972525abc0909", + HYPERLEX / "scripts/shadow/hyperlexical/semantic_compositionality_residual.py": "795b433413e86f05ea91186cfa85a68915f18bc6515e70a5ce5cfa6e1b8847d8", + HYPERLEX / "scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py": "36091d4cde5d7580d66ca2f9d8c2a568c996d10cef5d294ba1ed736505ff21ee", + HYPERLEX / "tests/shadow/test_semantic_compositionality_residual.py": "13df0dab6867f9adcf94605df7da5e779696f5e140ae1386c3d0b6600de6c07f", + } +) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def load_sealed_rows() -> list[dict]: + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if len(manifest["rows"]) != 225: + refuse("manifest row count drifted") + sealed = [] + for row in manifest["rows"]: + if row.get("sense_class") is not None: + refuse("development row has a sense class") + if row.get("pos") != row.get("synset_pos"): + refuse("row POS and synset POS differ") + sealed.append({field: row[field] for field in ROW_FIELDS}) + return sealed + + +def build_catalog() -> tuple[dict, dict]: + synsets, first_glosses = load_wordnet(WORDNET) + full = {} + for pos, name in FILES.items(): + for line in (WORDNET / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parsed = parse_data_line(line) + if parsed is None: + continue + synset, _first = parsed + full[(pos, synset.offset)] = " ".join(line.partition("|")[2].split()) + by_id = {} + index = defaultdict(list) + for (pos, offset), synset in synsets.items(): + if pos not in FILES: + continue + synset_id = f"{pos}:{offset}" + if synset_id in by_id: + continue + if (pos, offset) not in full: + refuse(f"missing gloss {synset_id}") + by_id[synset_id] = { + "gloss_first": " ".join(first_glosses[(pos, offset)].split()), + "gloss_full": full[(pos, offset)], + "lemmas": list(synset.lemmas), + "pointers": list(synset.pointers), + "pos": pos, + "synset_id": synset_id, + } + for lemma in synset.lemmas: + index[lookup_key(lemma)].append(synset_id) + for key, identifiers in index.items(): + index[key] = sorted(set(identifiers)) + return by_id, dict(index) + + +def pointer_tuples(record: dict | None, by_id: dict) -> list[tuple[str, int, str, str]]: + if record is None: + return [] + rows = [] + for pointer in record["pointers"]: + pos = POS_NAME.get(pointer.pos) + target_id = f"{pos}:{pointer.offset}" if pos else f"unknown:{pointer.offset}" + target = by_id.get(target_id) + if target is None or pointer.target < 1 or pointer.target > len(target["lemmas"]): + lemma = "" + else: + lemma = target["lemmas"][pointer.target - 1] + rows.append((pointer.symbol, pointer.target, target_id, lemma)) + return rows + + +def related_clauses(record: dict, by_id: dict) -> list[str]: + found = [] + seen = set() + ordered = sorted( + record["pointers"], + key=lambda pointer: (pointer.symbol, pointer.pos, pointer.offset, pointer.source, pointer.target), + ) + for pointer in ordered: + if pointer.symbol not in RELATION_EXPANSION_SYMBOLS: + continue + pos = POS_NAME.get(pointer.pos) + if pos is None: + continue + target_id = f"{pos}:{pointer.offset}" + if target_id in seen or target_id == record["synset_id"]: + continue + target = by_id.get(target_id) + if target is None: + continue + seen.add(target_id) + found.append(target["gloss_first"]) + return found + + +def candidate_tokens(record: dict, by_id: dict) -> list[str]: + tokens = lesk_tokens(record["gloss_full"]) + for clause in related_clauses(record, by_id): + tokens.extend(lesk_tokens(clause)) + return tokens + + +def lemma_supported(record: dict, constituent: str, exceptions: dict[str, set[str]]) -> bool: + keys = neighbor_keys(constituent, exceptions) + return any(lookup_key(lemma) in keys for lemma in record["lemmas"]) + + +def resolve_once(rows: list[dict], by_id: dict, index: dict, exceptions: dict[str, set[str]], spec_sha: str, procedure_sha: str) -> list[dict]: + records = [] + for row in rows: + parent_synset = f"{row['synset_pos']}:{row['synset_offset']}" + parent = by_id.get(parent_synset) + pointers = pointer_tuples(parent, by_id) + extraction = extract_constituents(row["surface"]) + content = extraction["content_constituents"] + if not content: + records.append( + { + "baseline_resolution_status": None, + "candidate_glosses": [], + "candidate_synsets": [], + "constituent_index": None, + "constituent_pos": None, + "constituent_surface": None, + "margin": None, + "parent_row_id": row["row_id"], + "parent_surface": row["surface"], + "parent_synset": parent_synset, + "primary_evidence_code": "fewer_than_two_content_constituents", + "procedure_sha256": procedure_sha, + "resolution_method": "NONE", + "resolution_status": None, + "row_resolution_status": "UNKNOWN", + "second_score": None, + "selected_synset": None, + "spec_sha256": spec_sha, + "supporting_evidence": [], + "top_score": None, + } + ) + continue + prepared = [] + for index_in_row, constituent in enumerate(content): + structural = structural_synset_ids(pointers, constituent, exceptions) + lexical = lexical_synset_ids(index, constituent, exceptions) + baseline = resolve_constituent( + exact_synset_ids(pointers, constituent, exceptions), + lexical, + ) + prepared.append( + { + "baseline": baseline, + "constituent": constituent, + "lexical": lexical, + "structural": structural, + } + ) + exact_gloss = {} + for index_in_row, item in enumerate(prepared): + structural = list(dict.fromkeys(item["structural"])) + lexical = list(dict.fromkeys(item["lexical"])) + selected = structural[0] if len(structural) == 1 else lexical[0] if not structural and len(lexical) == 1 else None + if selected is None: + continue + record = by_id.get(selected) + if record is None or not lemma_supported(record, item["constituent"], exceptions): + refuse(f"structural selection is not a constituent lemma: {selected}") + exact_gloss[index_in_row] = record["gloss_first"] + statuses = [] + emitted = [] + for index_in_row, item in enumerate(prepared): + structural = list(dict.fromkeys(item["structural"])) + lexical = list(dict.fromkeys(item["lexical"])) + lesk_scores = None + support_extra = [] + if not structural and len(lexical) > 1: + others = [exact_gloss[other] for other in range(len(prepared)) if other != index_in_row and other in exact_gloss] + context = lesk_tokens(row["gloss"]) + for clause in others: + context.extend(lesk_tokens(clause)) + lesk_scores = [] + for synset_id in lexical: + candidate = by_id.get(synset_id) + if candidate is None: + refuse(f"missing candidate synset {synset_id}") + if not lemma_supported(candidate, item["constituent"], exceptions): + refuse(f"candidate does not match constituent {synset_id}") + lesk_scores.append((synset_id, overlap_score(context, candidate_tokens(candidate, by_id)))) + support_extra.append(f"context_tokens:{len(context)}") + decision = resolve_constituent_sense(structural, lexical, lesk_scores) + if decision["selected_synset"] is not None: + chosen = by_id.get(decision["selected_synset"]) + if chosen is None or not lemma_supported(chosen, item["constituent"], exceptions): + refuse(f"selected synset fails the lemma check: {decision['selected_synset']}") + candidates = sorted(set(structural) | set(lexical)) + glosses = [] + for synset_id in candidates: + candidate = by_id.get(synset_id) + if candidate is None: + refuse(f"missing candidate gloss {synset_id}") + glosses.append(candidate["gloss_full"]) + poses = {synset_id.split(":", 1)[0] for synset_id in candidates} + if decision["selected_synset"]: + pos = decision["selected_synset"].split(":", 1)[0] + elif len(poses) == 1: + pos = next(iter(poses)) + else: + pos = None + statuses.append(decision["resolution_status"]) + support = list(decision["supporting_evidence"]) + support_extra + emitted.append( + { + "baseline_resolution_status": item["baseline"], + "candidate_glosses": glosses, + "candidate_synsets": candidates, + "constituent_index": index_in_row, + "constituent_pos": pos, + "constituent_surface": item["constituent"], + "margin": decision["margin"], + "parent_row_id": row["row_id"], + "parent_surface": row["surface"], + "parent_synset": parent_synset, + "primary_evidence_code": decision["primary_evidence_code"], + "procedure_sha256": procedure_sha, + "resolution_method": decision["resolution_method"], + "resolution_status": decision["resolution_status"], + "second_score": decision["second_score"], + "selected_synset": decision["selected_synset"], + "spec_sha256": spec_sha, + "supporting_evidence": support, + "top_score": decision["top_score"], + } + ) + status = row_resolution_status(len(content), statuses) + for item in emitted: + item["row_resolution_status"] = status + records.append(item) + return records + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def main() -> None: + check_sealed() + evaluation = json.loads(RESIDUAL_EVALUATION.read_text(encoding="utf-8")) + if evaluation["scored_count"] != 3 or evaluation["unknown_count"] != 222: + refuse("frozen residual evaluation counts drifted") + if evaluation["high_comparison"] != "NOT_COMPUTABLE": + refuse("frozen HIGH comparison drifted") + if evaluation["semantic_noncompositionality_threshold"] is not None or evaluation["emits_yes_no"] is not False: + refuse("frozen residual evaluation carries a threshold or a yes/no claim") + observed = { + "ambiguous_content_rows": 145, + "fewer_than_two_content_tokens": 59, + "scored_HIGH": 0, + "scored_REJECT": 3, + "scored_SECONDARY": 0, + "scored_rows": 3, + "unknown_rows": 222, + "unresolved_content_rows": 18, + } + if evaluation["abstention_reason_counts"] != { + "ambiguous_content_constituent": 145, + "fewer_than_two_content_constituents": 59, + "unresolved_content_constituent": 18, + }: + refuse("frozen abstention counts drifted") + limitation = { + "conclusion": { + "primary_limitation": "constituent_sense_resolution", + "source_selection_eligible": False, + "threshold_eligible": False, + }, + "finding": "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED", + "interpretation": "The dominant blocker for semantic-compositionality residual evaluation is deterministic constituent sense resolution, not the residual model itself.", + "json_schema_document": None, + "observed": observed, + "residual_candidate_decision_sha256": EXPECTED[RESIDUAL_DECISION], + "residual_candidate_spec_sha256": EXPECTED[RESIDUAL_SPEC], + "residual_development_evaluation_sha256": EXPECTED[RESIDUAL_EVALUATION], + "residual_development_scores_sha256": EXPECTED[RESIDUAL_SCORES], + "residual_model_name": "sentence-transformers/all-MiniLM-L6-v2", + "residual_model_revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", + "residual_source_provenance_sha256": EXPECTED[RESIDUAL_PROVENANCE], + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "schema": "hyperlex.semantic_residual_v1_coverage_limitation.v1", + "threshold_authorization": "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + "threshold_authorization_granted": False, + } + limitation_sha = write_json(LIMITATION_PATH, limitation) + review = { + "authorized_transition": "RESIDUAL_DEVELOPMENT_RESULT_REVIEW", + "coverage_limitation": "CONFIRMED", + "coverage_limitation_sha256": limitation_sha, + "finding": "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED", + "json_schema_document": None, + "next_research_track": RULE_VERSION, + "next_research_track_state_at_review": "SPEC_FROZEN", + "residual_artifacts_modified": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "schema": "hyperlex.residual_v1_development_result_review.v1", + "selected_source": "none", + "source_selection_eligible": False, + "threshold_authorization_granted": False, + "threshold_eligible": False, + } + review_sha = write_json(REVIEW_PATH, review) + policy = resolver_policy() + spec = { + "coverage_limitation_sha256": limitation_sha, + "development_result_review_sha256": review_sha, + "json_schema_document": None, + "policy": policy, + "residual_model_unchanged": { + "composition": "normalized_mean_v1", + "distance": "one_minus_cosine_v1", + "model_name": "sentence-transformers/all-MiniLM-L6-v2", + "model_revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", + }, + "residual_replay_authorized": False, + "residual_scores_included": False, + "schema": "hyperlex.constituent_sense_resolution_v1_spec.v1", + "selected_source": "none", + } + spec_text = json.dumps(spec, sort_keys=True) + if "operator_bucket" in spec_text or "0.2139784896" in spec_text: + refuse("resolver spec contains a label or a residual score") + spec_sha = write_json(SPEC_PATH, spec) + procedure = { + "json_schema_document": None, + "policy": policy, + "rule": RULE_VERSION, + "schema": "hyperlex.constituent_sense_resolution_v1_procedure.v1", + "spec_sha256": spec_sha, + } + procedure_sha = write_json(PROCEDURE_PATH, procedure) + rows = load_sealed_rows() + exceptions = load_exceptions(WORDNET) + first_catalog = build_catalog() + second_catalog = build_catalog() + first = resolve_once(rows, first_catalog[0], first_catalog[1], exceptions, spec_sha, procedure_sha) + second = resolve_once(rows, second_catalog[0], second_catalog[1], exceptions, spec_sha, procedure_sha) + first_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in first) + second_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in second) + if first_text != second_text: + refuse("NOT_DETERMINISTIC") + baseline_counts = Counter( + row["baseline_resolution_status"] for row in first if row["constituent_index"] is not None + ) + for name, expected_count in BASELINE_COUNTS.items(): + if baseline_counts[name] != expected_count: + refuse(f"baseline {name} is {baseline_counts[name]}") + if any("operator_bucket" in row for row in first): + refuse("resolution row carries an operator bucket") + replay_sha = write_jsonl(REPLAY_PATH, first) + if sha256(SPEC_PATH) != spec_sha or sha256(PROCEDURE_PATH) != procedure_sha: + refuse("resolution mutated the frozen spec") + + rejoined = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during resolution") + buckets = {row["row_id"]: row["operator_bucket"] for row in rejoined["rows"]} + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + attempts = [row for row in first if row["constituent_index"] is not None] + status_counts = Counter(row["resolution_status"] for row in attempts) + method_counts = Counter(row["resolution_method"] for row in attempts) + code_counts = Counter(row["primary_evidence_code"] for row in attempts) + baseline_ambiguous_rows = [row for row in attempts if row["baseline_resolution_status"] == "AMBIGUOUS"] + not_resolved = sum(row["resolution_status"] in {"AMBIGUOUS", "UNRESOLVED"} for row in baseline_ambiguous_rows) + lesk_attempts = [row for row in attempts if row["resolution_method"] == "EXTENDED_LESK_V1"] + ties = sum(row["primary_evidence_code"] == "extended_lesk_tie" for row in lesk_attempts) + row_status = {} + for row in first: + row_status[row["parent_row_id"]] = row["row_resolution_status"] + if len(row_status) != 225: + refuse("row status does not cover 225 rows") + ready_by_operator = {name: {"RESIDUAL_READY": 0, "UNKNOWN": 0} for name in OPERATORS} + for row_id, status in row_status.items(): + ready_by_operator[buckets[row_id]][status] += 1 + ready_total = sum(item["RESIDUAL_READY"] for item in ready_by_operator.values()) + unknown_total = sum(item["UNKNOWN"] for item in ready_by_operator.values()) + outcome = coverage_decision( + baseline_ambiguous=BASELINE_COUNTS["AMBIGUOUS"], + baseline_ambiguous_not_resolved=not_resolved, + residual_ready_high=ready_by_operator["HIGH"]["RESIDUAL_READY"], + residual_ready_secondary=ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + ) + attempt_count = len(attempts) + analysis = { + "abstention_rate_fraction": fraction(status_counts["AMBIGUOUS"] + status_counts["UNRESOLVED"], attempt_count), + "baseline_ambiguous_not_resolved": not_resolved, + "baseline_counts": {name: BASELINE_COUNTS[name] for name in ("EXACT", "UNIQUE", "AMBIGUOUS", "UNRESOLVED")}, + "candidate_status": outcome["candidate_status"], + "change_versus_residual_v1": { + "baseline_AMBIGUOUS": BASELINE_COUNTS["AMBIGUOUS"], + "baseline_EXACT": BASELINE_COUNTS["EXACT"], + "baseline_UNIQUE": BASELINE_COUNTS["UNIQUE"], + "baseline_UNRESOLVED": BASELINE_COUNTS["UNRESOLVED"], + "new_AMBIGUOUS": status_counts["AMBIGUOUS"], + "new_EXACT": status_counts["EXACT"], + "new_RESOLVED": status_counts["RESOLVED"], + "new_UNRESOLVED": status_counts["UNRESOLVED"], + }, + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "determinism": "IDENTICAL", + "development_result_review_sha256": review_sha, + "exact_fraction": fraction(status_counts["EXACT"], attempt_count), + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "high_unknown": ready_by_operator["HIGH"]["UNKNOWN"], + "json_schema_document": None, + "method_counts": dict(sorted(method_counts.items())), + "necessary_condition_met": outcome["necessary_condition_met"], + "operator_labels_joined_after_resolution_artifact_was_hashed": True, + "operator_labels_used_during_resolution": False, + "primary_evidence_code_counts": dict(sorted(code_counts.items())), + "procedure_sha256": procedure_sha, + "projection": { + "REJECT_rows": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "HIGH_rows": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "QUARANTINE_rows": ready_by_operator["QUARANTINE"]["RESIDUAL_READY"], + "SECONDARY_rows": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "residual_replay_candidate_rows": ready_total, + "residual_replay_performed": False, + }, + "quarantine_residual_ready": ready_by_operator["QUARANTINE"]["RESIDUAL_READY"], + "quarantine_unknown": ready_by_operator["QUARANTINE"]["UNKNOWN"], + "reject_residual_ready": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "reject_unknown": ready_by_operator["REJECT"]["UNKNOWN"], + "replay_sha256": replay_sha, + "residual_ready_by_operator": ready_by_operator, + "residual_ready_rows": ready_total, + "residual_scores_recomputed": False, + "resolution_counts": { + "AMBIGUOUS": status_counts["AMBIGUOUS"], + "EXACT": status_counts["EXACT"], + "RESOLVED": status_counts["RESOLVED"], + "UNRESOLVED": status_counts["UNRESOLVED"], + }, + "resolved_fraction": fraction(status_counts["RESOLVED"], attempt_count), + "row_count": 225, + "rule": RULE_VERSION, + "schema": "hyperlex.constituent_sense_resolution_v1_development_analysis.v1", + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "secondary_unknown": ready_by_operator["SECONDARY"]["UNKNOWN"], + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "spec_sha256": spec_sha, + "tie_rate_fraction": fraction(ties, len(lesk_attempts)) if lesk_attempts else "0/0", + "total_constituent_attempts": attempt_count, + "unknown_rows": unknown_total, + "unresolved_rate_fraction": fraction(status_counts["UNRESOLVED"], attempt_count), + "yes_no_emitted": False, + } + analysis_sha = write_json(ANALYSIS_PATH, analysis) + decision = { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "development_result_review_sha256": review_sha, + "encoded": True, + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "necessary_condition_met": outcome["necessary_condition_met"], + "next_legal_transition": outcome["next_legal_transition"], + "next_transition_authorized": False, + "procedure_sha256": procedure_sha, + "replay_sha256": replay_sha, + "residual_replay_performed": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "residual_threshold_created": False, + "rule": RULE_VERSION, + "runtime_integration": False, + "schema": "hyperlex.constituent_sense_resolution_v1_decision.v1", + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "select_005_authorized": False, + "selected_source": "none", + "spec_sha256": spec_sha, + "state": "DEVELOPMENT_ANALYZED", + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, decision) + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_coverage_limitation"] = "CONFIRMED" + tracker["residual_threshold_eligible"] = False + tracker["residual_source_selection_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["residual_development_result_review_sha256"] = review_sha + tracker["residual_coverage_limitation_sha256"] = limitation_sha + tracker["constituent_sense_resolution_rule"] = RULE_VERSION + tracker["constituent_sense_resolution_state"] = "DEVELOPMENT_ANALYZED" + tracker["constituent_sense_resolution_candidate_status"] = outcome["candidate_status"] + tracker["constituent_sense_resolution_coverage_finding"] = outcome["coverage_finding"] + tracker["constituent_sense_resolution_encoded"] = True + tracker["constituent_sense_resolution_runtime_applied"] = False + tracker["constituent_sense_resolution_spec_sha256"] = spec_sha + tracker["constituent_sense_resolution_procedure_sha256"] = procedure_sha + tracker["constituent_sense_resolution_replay_sha256"] = replay_sha + tracker["constituent_sense_resolution_analysis_sha256"] = analysis_sha + tracker["constituent_sense_resolution_decision_sha256"] = decision_sha + tracker["constituent_sense_resolution_residual_ready_rows"] = ready_total + tracker["constituent_sense_resolution_high_ready"] = ready_by_operator["HIGH"]["RESIDUAL_READY"] + tracker["constituent_sense_resolution_secondary_ready"] = ready_by_operator["SECONDARY"]["RESIDUAL_READY"] + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = outcome["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "decision_sha256": decision_sha, + "determinism": "IDENTICAL", + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "high_unknown": ready_by_operator["HIGH"]["UNKNOWN"], + "method_counts": analysis["method_counts"], + "necessary_condition_met": outcome["necessary_condition_met"], + "next_legal_transition": outcome["next_legal_transition"], + "procedure_sha256": procedure_sha, + "projection": analysis["projection"], + "reject_residual_ready": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "replay_sha256": replay_sha, + "residual_ready_rows": ready_total, + "resolution_counts": analysis["resolution_counts"], + "review_sha256": review_sha, + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "secondary_unknown": ready_by_operator["SECONDARY"]["UNKNOWN"], + "spec_sha256": spec_sha, + "tie_rate_fraction": analysis["tie_rate_fraction"], + "total_constituent_attempts": attempt_count, + "tracker_sha256": tracker_sha, + "unknown_rows": unknown_total, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py new file mode 100644 index 00000000..f81eabc1 --- /dev/null +++ b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py @@ -0,0 +1,885 @@ +"""Evaluate one pinned GlossBERT checkpoint on Extended Lesk ties. + +The specification is hashed before any Hyperlex constituent is scored. Operator +labels are read only after the readiness projection is hashed. Residual scores +are not loaded and are not recomputed. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from collections import Counter, defaultdict +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["NUMEXPR_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.constituent_sense_resolution_v1_replay import ( + ANALYSIS_PATH as RESOLVER_ANALYSIS_PATH, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + DECISION_PATH as RESOLVER_DECISION_PATH, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EXPECTED as RESOLVER_EXPECTED, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EVENTS, + EVIDENCE_MANIFEST, + HYPERLEX, + LEDGER_FILE, + LIMITATION_PATH, + OPERATOR_COUNTS, + OPERATORS, + PROCEDURE_PATH as RESOLVER_PROCEDURE_PATH, + REPLAY_PATH as RESOLVER_REPLAY_PATH, + REVIEW_PATH, + SOURCE, + SPEC_PATH as RESOLVER_SPEC_PATH, + TRACKER, + WORDNET, + build_catalog, + load_sealed_rows, + refuse, + sha256, + write_json, + write_jsonl, +) +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.model_based_wsd_candidate_v1 import ( + BASELINE_HIGH_READY, + BASELINE_SECONDARY_READY, + BASELINE_TOTAL_READY, + CANONICAL_SOURCE, + IDENTICAL, + LICENSE_NAME, + MAX_TOKENS, + MODEL_FAMILY, + MODEL_NAME, + MODEL_REVISION, + POSITIVE_CLASS_INDEX, + REJECTED, + RULE_VERSION, + candidate_gloss_text, + candidate_policy, + coverage_gate, + format_probability, + gloss_lemma, + overlay_status, + project_row_status, + quoted_context, + resolve_model_scores, + summarize_confidence, +) +from hyperlexical.semantic_compositionality_residual import neighbor_keys +from hyperlexical.unbind_sense_screen_v1 import load_exceptions + +MODEL_DIR = ( + Path("/home/morpheus/hlx-private/eval-reserve-20260926/acquisition/sources/glossbert") + / MODEL_REVISION +) +CODE_LICENSE = ( + Path("/home/morpheus/hlx-private/eval-reserve-20260926/acquisition/sources/glossbert") + / "ORIGINAL_CODE_MIT_LICENSE.txt" +) +SPEC_PATH = SOURCE / "MODEL_BASED_WSD_CANDIDATE_SPEC.json" +PROVENANCE_PATH = SOURCE / "MODEL_BASED_WSD_PROVENANCE.json" +LICENSE_PATH = SOURCE / "MODEL_BASED_WSD_LICENSE_RECEIPT.json" +RAW_PATH = SOURCE / "MODEL_BASED_WSD_RAW_OUTPUT.jsonl" +RESOLUTION_PATH = SOURCE / "MODEL_BASED_WSD_RESOLUTION.jsonl" +PROJECTION_PATH = SOURCE / "MODEL_BASED_WSD_READINESS_PROJECTION.json" +ANALYSIS_PATH = SOURCE / "MODEL_BASED_WSD_DEVELOPMENT_ANALYSIS.json" +DECISION_PATH = SOURCE / "MODEL_BASED_WSD_CANDIDATE_DECISION.json" + +CURRENT_TRACKER = "c72e55c8096143b8675d4aa995c5d4b19de95d256dd234d32a421ba098bc7d4f" +WEIGHTS = { + "config.json": "70f859c899543d9c8f822e64201e1a77530f0de12aabb0daf04b9570f538b638", + "pytorch_model.bin": "60706c7618f8ccbfa7d0a6d1d1009765a7146ea5f4232924ed9f1c46d521c898", + "README.md": "27577f2e0742d2e91ccad99778cc15fe521386de53d7bfe8acedfa0a78f8c186", + "special_tokens_map.json": "303df45a03609e4ead04bc3dc1536d0ab19b5358db685b6f3da123d05ec200e3", + "tokenizer_config.json": "09e49d0e788d25991da77d37b10eaa6a86a4e94e2127de8bedc94eb45baf2d84", + "vocab.txt": "07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3", +} +CODE_LICENSE_SHA = "333e83b83ae30a1c9af100ea742de8ab34ac9d6d8c8aae52e0bbec47cf9df80f" +RUNTIME = { + "numpy": "2.5.3", + "python": "3.12.3", + "tokenizers": "0.23.2", + "torch": "2.14.0+cpu", + "transformers": "5.17.0", +} +FINANCIAL_CONTEXT = 'She deposited the check at the "bank" yesterday.' +RIVER_CONTEXT = 'They picnicked on the grassy "bank" of the river.' +FINANCIAL_SYNSET = "noun:08420278" +RIVER_SYNSET = "noun:09213565" +POOL_ATTEMPTS = 249 +SS_POS = {"1": "noun", "2": "verb", "3": "adj", "4": "adv", "5": "adj"} + +_THIRD_PARTY = ( + "The Hugging Face card is a third-party upload. It is not the authors' " + "Google Drive checkpoint. The card declares MIT. The original GlossBERT " + "code repository is MIT. No API key is required." +) +_CODE_REPOSITORY = "https://github.com/HSLCY/GlossBERT" + +EXPECTED = dict(RESOLVER_EXPECTED) +EXPECTED[TRACKER] = CURRENT_TRACKER +EXPECTED[REVIEW_PATH] = "1c1b69856dd88567167fd5c958cd8db6d39ab9ec74a4e0ed3e667a521c82e6fa" +EXPECTED[LIMITATION_PATH] = "fc8839c15a7638b2bfca1cf0548bfb4d5f433434bae0fea2944a528dd15d6142" +EXPECTED[RESOLVER_SPEC_PATH] = "176e6219ddc3127814a25d39ad26e3571817f7ea8323d685e081d2e0fd867acb" +EXPECTED[RESOLVER_PROCEDURE_PATH] = "9f76b64aa6aac09bd56ca9cc8a847cda12b31a58eea8c4b51426733917d54248" +EXPECTED[RESOLVER_REPLAY_PATH] = "a0c707ab55e02f627a698c33ddc0ca398e34aa0bafc19b13e72422bc26d97d0a" +EXPECTED[RESOLVER_ANALYSIS_PATH] = "6d47610014f74094394355a50419fe24441103bec09458b6f25ea392fb57151c" +EXPECTED[RESOLVER_DECISION_PATH] = "ff8b90ebe5d48151dc68ddbf676e1f27d4cee5e3be2730f999c9b089f1392caa" +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py"] = ( + "927fb5c6f627d5488ecbc66779595812e8c2154ad47d6a07db1e377639f36fc6" +) +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py"] = ( + "d8cd1c1719d9764cb0a77ad51a2d305268181b42294cd1adddd3ff9d4ea57e0b" +) +EXPECTED[HYPERLEX / "tests/shadow/test_constituent_sense_resolution_v1.py"] = ( + "104ac0edc643b76df4ca1fe905e30ed96c7417dddfc14010b23d3b3c613adeb9" +) + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def load_jsonl(path: Path) -> list[dict]: + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line] + + +def load_sense_index(path: Path) -> dict[str, list[dict]]: + index = defaultdict(list) + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + parts = line.split() + if len(parts) < 2 or "%" not in parts[0]: + continue + sense_key, offset = parts[0], parts[1] + lemma, _, rest = sense_key.partition("%") + pos = SS_POS.get(rest.split(":", 1)[0]) + if pos is None: + continue + index[f"{pos}:{offset.zfill(8)}"].append({"lemma": lemma, "sense_key": sense_key}) + return dict(index) + + +def sense_keys_for(synset_id: str, constituent: str, exceptions: dict, sense_index: dict) -> list[str]: + allowed = neighbor_keys(constituent, exceptions) + found = [ + item["sense_key"] + for item in sense_index.get(synset_id, []) + if lookup_key(item["lemma"]) in allowed + ] + return sorted(set(found)) + + +def assert_weights() -> dict: + if not MODEL_DIR.is_dir(): + refuse("glossbert directory is missing") + found = {} + for name, expected in WEIGHTS.items(): + path = MODEL_DIR / name + digest = sha256(path) + if digest != expected: + refuse(f"weight hash mismatch: {name}") + found[name] = digest + if not CODE_LICENSE.is_file() or sha256(CODE_LICENSE) != CODE_LICENSE_SHA: + refuse("original code license file mismatch") + readme = (MODEL_DIR / "README.md").read_text(encoding="utf-8") + if "license: mit" not in readme.lower(): + refuse("model card does not declare mit") + return found + + +def license_payload(weights: dict) -> dict: + return { + "api_required": False, + "artifact_license_file_present": False, + "canonical_source": CANONICAL_SOURCE, + "card_license": "mit", + "code_license_sha256": CODE_LICENSE_SHA, + "datasets": ["SemCor3.0"], + "evaluation_permitted": True, + "json_schema_document": None, + "license_name": LICENSE_NAME, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "original_code_license": "MIT", + "original_code_repository": _CODE_REPOSITORY, + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "schema": "hyperlex.model_based_wsd_candidate_v1_license_receipt.v1", + "source_license_unresolved": False, + "third_party_note": _THIRD_PARTY, + "third_party_upload": True, + "vocab_txt_sha256": weights["vocab.txt"], + } + + +def spec_payload(weights: dict, license_sha: str) -> dict: + return { + "artifacts": { + "config_json_sha256": weights["config.json"], + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "readme_md_sha256": weights["README.md"], + "special_tokens_map_json_sha256": weights["special_tokens_map.json"], + "tokenizer_config_json_sha256": weights["tokenizer_config.json"], + "vocab_txt_sha256": weights["vocab.txt"], + }, + "expected_tier3_attempts": POOL_ATTEMPTS, + "json_schema_document": None, + "license_receipt_sha256": license_sha, + "polarity_control": { + "financial_context": FINANCIAL_CONTEXT, + "financial_synset": FINANCIAL_SYNSET, + "hyperlex_rows_used": False, + "river_context": RIVER_CONTEXT, + "river_synset": RIVER_SYNSET, + }, + "policy": candidate_policy(), + "prior_resolver_replay_sha256": EXPECTED[RESOLVER_REPLAY_PATH], + "residual_scores_included": False, + "runtime": dict(RUNTIME), + "schema": "hyperlex.model_based_wsd_candidate_v1_spec.v1", + "selected_source": "none", + } + + +def prepare_pool(frozen: list[dict], by_id: dict, exceptions: dict, sense_index: dict) -> list[dict]: + attempts = [row for row in frozen if row["constituent_index"] is not None] + counts = Counter(row["resolution_status"] for row in attempts) + if counts["EXACT"] != 64 or counts["RESOLVED"] != 121 or counts["AMBIGUOUS"] != 249 or counts["UNRESOLVED"] != 70: + refuse("frozen resolver counts drifted") + pool = [] + for row in attempts: + if row["resolution_status"] != "AMBIGUOUS": + continue + if row["resolution_method"] != "EXTENDED_LESK_V1": + refuse("ambiguous constituent is not an extended lesk tie") + pairs = [] + for synset_id in row["candidate_synsets"]: + record = by_id.get(synset_id) + if record is None: + refuse(f"missing candidate synset {synset_id}") + lemma = gloss_lemma(record["lemmas"], row["constituent_surface"], exceptions) + if lemma is None: + refuse(f"candidate lemma missing for {synset_id}") + pairs.append( + { + "gloss_text": candidate_gloss_text(lemma, record["gloss_first"]), + "sense_keys": sense_keys_for(synset_id, row["constituent_surface"], exceptions, sense_index), + "synset": synset_id, + } + ) + pool.append( + { + "candidate_pwn30_synsets": list(row["candidate_synsets"]), + "candidate_sense_keys": {item["synset"]: list(item["sense_keys"]) for item in pairs}, + "constituent_index": row["constituent_index"], + "constituent_pos": row["constituent_pos"], + "constituent_surface": row["constituent_surface"], + "context": quoted_context(row["parent_surface"], row["constituent_index"], row["constituent_surface"]), + "pairs": pairs, + "parent_row_id": row["parent_row_id"], + "parent_surface": row["parent_surface"], + "parent_synset": row["parent_synset"], + "prior_lesk_margin": row["margin"], + "prior_lesk_second_score": row["second_score"], + "prior_lesk_top_score": row["top_score"], + "prior_resolution_status": "AMBIGUOUS", + } + ) + if len(pool) != POOL_ATTEMPTS: + refuse(f"tier 3 pool is {len(pool)}") + return pool + + +def import_runtime(): + import numpy + import torch + import tokenizers + import transformers + from transformers import AutoTokenizer, BertForSequenceClassification + + versions = { + "numpy": numpy.__version__, + "python": ".".join(str(part) for part in sys.version_info[:3]), + "tokenizers": tokenizers.__version__, + "torch": torch.__version__, + "transformers": transformers.__version__, + } + if versions != RUNTIME: + refuse(f"runtime drift: {versions}") + torch.set_num_threads(1) + torch.set_num_interop_threads(1) + torch.backends.mkldnn.enabled = False + torch.use_deterministic_algorithms(False) + torch.manual_seed(0) + tokenizer = AutoTokenizer.from_pretrained(MODEL_DIR, local_files_only=True) + model = BertForSequenceClassification.from_pretrained(MODEL_DIR, local_files_only=True) + model.eval() + model.to("cpu") + if model.num_labels != 2: + refuse("classifier head is not binary") + return torch, tokenizer, model + + +def score_pair(torch, tokenizer, model, context: str, gloss_text: str) -> dict: + encoded = tokenizer(context, gloss_text, truncation=False, padding=False, return_tensors="pt") + length = int(encoded["input_ids"].shape[-1]) + if length > MAX_TOKENS: + return {"overflow": True, "positive_probability": None, "token_count": length} + with torch.inference_mode(): + logits = model(**encoded).logits[0] + probability = torch.softmax(logits, dim=-1)[POSITIVE_CLASS_INDEX] + return { + "overflow": False, + "positive_probability": format_probability(float(probability)), + "token_count": length, + } + + +def score_context(torch, tokenizer, model, context: str, pairs: list[dict]) -> dict: + scored = [] + overflow = False + error = None + try: + for pair in pairs: + result = score_pair(torch, tokenizer, model, context, pair["gloss_text"]) + overflow = overflow or result["overflow"] + scored.append( + { + "gloss_text": pair["gloss_text"], + "overflow": result["overflow"], + "positive_probability": result["positive_probability"], + "sense_keys": list(pair["sense_keys"]), + "synset": pair["synset"], + "token_count": result["token_count"], + } + ) + except Exception as exc: + error = type(exc).__name__ + scored = [] + overflow = False + return {"error": error, "overflow": overflow, "scored_pairs": scored} + + +def polarity_control(torch, tokenizer, model, by_id: dict, sense_index: dict) -> dict: + synsets = sorted( + { + synset_id + for synset_id, items in sense_index.items() + if any(item["sense_key"].startswith("bank%1:") and item["lemma"] == "bank" for item in items) + } + ) + pairs = [] + for synset_id in synsets: + record = by_id.get(synset_id) + if record is None: + refuse(f"missing polarity synset {synset_id}") + pairs.append( + { + "gloss_text": candidate_gloss_text("bank", record["gloss_first"]), + "sense_keys": [item["sense_key"] for item in sense_index[synset_id] if item["lemma"] == "bank"], + "synset": synset_id, + } + ) + report = {"hyperlex_rows_used": False} + for name, context, expected in ( + ("financial", FINANCIAL_CONTEXT, FINANCIAL_SYNSET), + ("river", RIVER_CONTEXT, RIVER_SYNSET), + ): + scored = score_context(torch, tokenizer, model, context, pairs) + probabilities = { + row["synset"]: row["positive_probability"] + for row in scored["scored_pairs"] + if row["positive_probability"] is not None + } + decision = resolve_model_scores( + [pair["synset"] for pair in pairs], + {pair["synset"]: sorted(pair["sense_keys"]) for pair in pairs}, + None if scored["error"] or scored["overflow"] else probabilities, + overflow=scored["overflow"], + error=scored["error"], + ) + report[name] = { + "expected_synset": expected, + "selected_synset": decision["selected_synset"], + "status": decision["model_resolution_status"], + "top_probability": decision["model_confidence"], + } + if decision["model_resolution_status"] != "RESOLVED" or decision["selected_synset"] != expected: + report["passed"] = False + return report + report["passed"] = True + report["positive_class_index"] = POSITIVE_CLASS_INDEX + return report + + +def infer_pass(torch, tokenizer, model, pool: list[dict], spec_sha: str) -> list[dict]: + torch.manual_seed(0) + rows = [] + for index, item in enumerate(pool): + scored = score_context(torch, tokenizer, model, item["context"], item["pairs"]) + rows.append( + { + "candidate_pwn30_synsets": list(item["candidate_pwn30_synsets"]), + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "context": item["context"], + "error": scored["error"], + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "overflow": scored["overflow"], + "pairs": scored["scored_pairs"], + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "record_kind": "hyperlex_tier3_constituent", + "spec_sha256": spec_sha, + } + ) + if index % 25 == 0: + print(f"scored {index}", file=sys.stderr, flush=True) + return rows + + +def resolution_rows(pool: list[dict], raw_rows: list[dict]) -> list[dict]: + resolved = [] + for item, raw in zip(pool, raw_rows, strict=True): + probabilities = None + if not raw["error"] and not raw["overflow"]: + probabilities = {pair["synset"]: pair["positive_probability"] for pair in raw["pairs"]} + decision = resolve_model_scores( + item["candidate_pwn30_synsets"], + item["candidate_sense_keys"], + probabilities, + overflow=raw["overflow"], + error=raw["error"], + ) + if decision["selected_synset"] is not None and decision["selected_synset"] not in item["candidate_pwn30_synsets"]: + refuse("selected synset left the candidate set") + resolved.append( + { + "candidate_pwn30_synsets": list(item["candidate_pwn30_synsets"]), + "candidate_sense_keys": item["candidate_sense_keys"], + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "model_candidate_scores": decision["model_candidate_scores"], + "model_confidence": decision["model_confidence"], + "model_margin": decision["model_margin"], + "model_name": MODEL_NAME, + "model_resolution_status": decision["model_resolution_status"], + "model_revision": MODEL_REVISION, + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "primary_evidence_code": decision["primary_evidence_code"], + "prior_lesk_margin": item["prior_lesk_margin"], + "prior_lesk_second_score": item["prior_lesk_second_score"], + "prior_lesk_top_score": item["prior_lesk_top_score"], + "prior_resolution_status": "AMBIGUOUS", + "selected_sense_key": decision["selected_sense_key"], + "selected_synset": decision["selected_synset"], + "spec_sha256": raw["spec_sha256"], + } + ) + return resolved + + +def project_rows(frozen: list[dict], resolutions: list[dict]) -> tuple[list[dict], dict]: + tier3 = {(row["parent_row_id"], row["constituent_index"]): row for row in resolutions} + if len(tier3) != len(resolutions): + refuse("duplicate tier 3 keys") + grouped = defaultdict(list) + for row in frozen: + grouped[row["parent_row_id"]].append(row) + projected = [] + combined = Counter() + for sealed in load_sealed_rows(): + items = [row for row in grouped[sealed["row_id"]] if row["constituent_index"] is not None] + items.sort(key=lambda row: row["constituent_index"]) + statuses = [] + for item in items: + key = (item["parent_row_id"], item["constituent_index"]) + model_status = None + if key in tier3: + if item["resolution_status"] != "AMBIGUOUS" or item["resolution_method"] != "EXTENDED_LESK_V1": + refuse("tier 3 key is not an extended lesk tie") + model_status = tier3[key]["model_resolution_status"] + status = overlay_status(item["resolution_status"], item["resolution_method"], model_status) + statuses.append(status) + combined[status] += 1 + projected.append( + { + "constituent_statuses": statuses, + "content_count": len(statuses), + "parent_row_id": sealed["row_id"], + "projected_status": project_row_status(statuses), + } + ) + if len(projected) != 225: + refuse("projection does not cover 225 rows") + if sum(combined.values()) != 504: + refuse("combined constituent count is not 504") + counts = Counter(row["projected_status"] for row in projected) + if counts["RESIDUAL_READY"] + counts["UNKNOWN"] != 225: + refuse("projected row count drifted") + return projected, { + "combined_constituent_counts": { + "AMBIGUOUS": combined["AMBIGUOUS"], + "ERROR": combined["ERROR"], + "EXACT": combined["EXACT"], + "INVALID": combined["INVALID"], + "LESK_RESOLVED": combined["LESK_RESOLVED"], + "MODEL_RESOLVED": combined["MODEL_RESOLVED"], + "UNRESOLVED": combined["UNRESOLVED"], + }, + "row_counts": {"RESIDUAL_READY": counts["RESIDUAL_READY"], "UNKNOWN": counts["UNKNOWN"]}, + } + + +def update_tracker(outcome: dict, hashes: dict) -> str: + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if sha256(TRACKER) != CURRENT_TRACKER: + refuse("tracker hash drifted before update") + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = CURRENT_TRACKER + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["model_based_wsd_rule"] = RULE_VERSION + tracker["model_based_wsd_state"] = "CANDIDATE_EVALUATED" + tracker["model_based_wsd_candidate_status"] = outcome["candidate_status"] + tracker["model_based_wsd_model_name"] = MODEL_NAME + tracker["model_based_wsd_model_revision"] = MODEL_REVISION + tracker["model_based_wsd_runtime_integration"] = False + tracker["model_based_wsd_spec_sha256"] = hashes.get("spec_sha256") + tracker["model_based_wsd_provenance_sha256"] = hashes.get("provenance_sha256") + tracker["model_based_wsd_license_receipt_sha256"] = hashes.get("license_sha256") + tracker["model_based_wsd_raw_output_sha256"] = hashes.get("raw_sha256") + tracker["model_based_wsd_resolution_sha256"] = hashes.get("resolution_sha256") + tracker["model_based_wsd_readiness_projection_sha256"] = hashes.get("projection_sha256") + tracker["model_based_wsd_analysis_sha256"] = hashes.get("analysis_sha256") + tracker["model_based_wsd_decision_sha256"] = hashes.get("decision_sha256") + tracker["model_based_wsd_residual_ready_rows"] = outcome.get("total_ready") + tracker["model_based_wsd_high_ready"] = outcome.get("high_ready") + tracker["model_based_wsd_secondary_ready"] = outcome.get("secondary_ready") + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_threshold_eligible"] = False + tracker["residual_source_selection_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["constituent_sense_resolution_state"] = "DEVELOPMENT_ANALYZED" + tracker["constituent_sense_resolution_candidate_status"] = "COVERAGE_INSUFFICIENT" + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = outcome["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + return tracker_sha + + +def write_decision(outcome: dict, hashes: dict, analysis_sha: str | None) -> str: + payload = { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "determinism": outcome.get("determinism"), + "high_ready": outcome.get("high_ready"), + "json_schema_document": None, + "license_receipt_sha256": hashes.get("license_sha256"), + "measurement_eligible": False, + "measurement_sample_drawn": False, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "next_legal_transition": outcome["next_legal_transition"], + "next_transition_authorized": False, + "projection_sha256": hashes.get("projection_sha256"), + "provenance_sha256": hashes.get("provenance_sha256"), + "raw_output_sha256": hashes.get("raw_sha256"), + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "residual_threshold_created": False, + "resolution_sha256": hashes.get("resolution_sha256"), + "rule": RULE_VERSION, + "runtime_integration": False, + "schema": "hyperlex.model_based_wsd_candidate_v1_decision.v1", + "secondary_ready": outcome.get("secondary_ready"), + "select_005_authorized": False, + "selected_source": "none", + "spec_sha256": hashes.get("spec_sha256"), + "state": "CANDIDATE_EVALUATED", + "threshold_eligible": False, + "total_ready": outcome.get("total_ready"), + "yes_no_emitted": False, + } + return write_json(DECISION_PATH, payload) + + +def main() -> None: + check_sealed() + weights = assert_weights() + license_sha = write_json(LICENSE_PATH, license_payload(weights)) + spec = spec_payload(weights, license_sha) + spec_text = json.dumps(spec, sort_keys=True) + if "0.2139784896" in spec_text or "operator_bucket" in spec_text: + refuse("spec contains a residual score or an operator field") + spec_sha = write_json(SPEC_PATH, spec) + if sha256(SPEC_PATH) != spec_sha: + refuse("spec hash drifted at freeze") + print(f"SPEC_FROZEN {spec_sha}", file=sys.stderr, flush=True) + sealed_rows = load_sealed_rows() + surfaces = {row["surface"] for row in sealed_rows} + if FINANCIAL_CONTEXT in surfaces or RIVER_CONTEXT in surfaces: + refuse("polarity control collides with a development surface") + frozen = load_jsonl(RESOLVER_REPLAY_PATH) + if sha256(RESOLVER_REPLAY_PATH) != EXPECTED[RESOLVER_REPLAY_PATH]: + refuse("resolver replay changed while loading") + exceptions = load_exceptions(WORDNET) + by_id, _index = build_catalog() + sense_index = load_sense_index(WORDNET / "index.sense") + pool = prepare_pool(frozen, by_id, exceptions, sense_index) + torch, tokenizer, model = import_runtime() + polarity = polarity_control(torch, tokenizer, model, by_id, sense_index) + hashes = {"license_sha256": license_sha, "spec_sha256": spec_sha} + if not polarity["passed"]: + provenance = { + "hyperlex_scoring_started": False, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "polarity_control": polarity, + "schema": "hyperlex.model_based_wsd_candidate_v1_provenance.v1", + "spec_sha256": spec_sha, + } + hashes["provenance_sha256"] = write_json(PROVENANCE_PATH, provenance) + outcome = { + "candidate_status": REJECTED, + "determinism": None, + "next_legal_transition": "MODEL_BASED_WSD_CANDIDATE_REVISION_AUTHORIZATION", + } + hashes["analysis_sha256"] = None + hashes["decision_sha256"] = write_decision(outcome, hashes, None) + update_tracker(outcome, hashes) + print(json.dumps({"candidate_status": REJECTED, "spec_sha256": spec_sha}, sort_keys=True)) + return + provenance = { + "canonical_source": CANONICAL_SOURCE, + "device": "cpu", + "dtype": "float32", + "hyperlex_scoring_started": False, + "json_schema_document": None, + "license_receipt_sha256": license_sha, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "polarity_control_passed": True, + "polarity_financial_winner": polarity["financial"]["selected_synset"], + "polarity_river_winner": polarity["river"]["selected_synset"], + "positive_class_index": POSITIVE_CLASS_INDEX, + "residual_embeddings_used": False, + "runtime": dict(RUNTIME), + "schema": "hyperlex.model_based_wsd_candidate_v1_provenance.v1", + "spec_sha256": spec_sha, + "third_party_upload": True, + "weights": { + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "vocab_txt_sha256": weights["vocab.txt"], + }, + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + hashes["provenance_sha256"] = provenance_sha + if sha256(SPEC_PATH) != spec_sha: + refuse("spec changed after provenance") + print("HYPERLEX_SCORING_START", file=sys.stderr, flush=True) + first = infer_pass(torch, tokenizer, model, pool, spec_sha) + print("PASS_1_DONE", file=sys.stderr, flush=True) + second = infer_pass(torch, tokenizer, model, pool, spec_sha) + print("PASS_2_DONE", file=sys.stderr, flush=True) + first_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in first) + second_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in second) + if first_text != second_text: + mismatch = { + "determinism": "NOT_DETERMINISTIC", + "first_sha256": hashlib.sha256(first_text.encode("utf-8")).hexdigest(), + "record_kind": "determinism_mismatch", + "second_sha256": hashlib.sha256(second_text.encode("utf-8")).hexdigest(), + "spec_sha256": spec_sha, + } + hashes["raw_sha256"] = write_jsonl(RAW_PATH, [mismatch]) + outcome = { + "candidate_status": "NOT_DETERMINISTIC", + "determinism": "NOT_DETERMINISTIC", + "next_legal_transition": "MODEL_BASED_WSD_DETERMINISM_REVIEW_AUTHORIZATION", + } + hashes["decision_sha256"] = write_decision(outcome, hashes, None) + update_tracker(outcome, hashes) + print(json.dumps(outcome, sort_keys=True)) + return + raw_sha = write_jsonl(RAW_PATH, first) + hashes["raw_sha256"] = raw_sha + resolutions = resolution_rows(pool, first) + if any(row["prior_resolution_status"] != "AMBIGUOUS" for row in resolutions): + refuse("tier 3 row was not previously ambiguous") + resolution_sha = write_jsonl(RESOLUTION_PATH, resolutions) + hashes["resolution_sha256"] = resolution_sha + if sha256(SPEC_PATH) != spec_sha: + refuse("scoring mutated the spec") + projected, summary = project_rows(frozen, resolutions) + tier3_counts = Counter(row["model_resolution_status"] for row in resolutions) + confidence = summarize_confidence(resolutions) + projection = { + "combined_constituent_counts": summary["combined_constituent_counts"], + "confidence": confidence, + "determinism": IDENTICAL, + "json_schema_document": None, + "operator_labels_included": False, + "raw_output_sha256": raw_sha, + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "resolution_sha256": resolution_sha, + "row_counts": summary["row_counts"], + "rows": projected, + "schema": "hyperlex.model_based_wsd_candidate_v1_readiness_projection.v1", + "spec_sha256": spec_sha, + "tier3_attempts": len(resolutions), + "tier3_counts": { + "AMBIGUOUS": tier3_counts["AMBIGUOUS"], + "ERROR": tier3_counts["ERROR"], + "INVALID": tier3_counts["INVALID"], + "RESOLVED": tier3_counts["RESOLVED"], + }, + } + projection_text = json.dumps(projection, sort_keys=True) + if "operator_bucket" in projection_text or "HIGH" in projection["row_counts"]: + refuse("projection carries an operator field") + projection_sha = write_json(PROJECTION_PATH, projection) + hashes["projection_sha256"] = projection_sha + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {row["row_id"]: row["operator_bucket"] for row in manifest["rows"]} + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + ready_by = {name: {"RESIDUAL_READY": 0, "UNKNOWN": 0} for name in OPERATORS} + for row in projected: + ready_by[buckets[row["parent_row_id"]]][row["projected_status"]] += 1 + total_ready = summary["row_counts"]["RESIDUAL_READY"] + outcome = coverage_gate( + high_ready=ready_by["HIGH"]["RESIDUAL_READY"], + secondary_ready=ready_by["SECONDARY"]["RESIDUAL_READY"], + total_ready=total_ready, + invalid_output_count=tier3_counts["INVALID"], + error_count=tier3_counts["ERROR"], + determinism=IDENTICAL, + ) + outcome["determinism"] = IDENTICAL + outcome["high_ready"] = ready_by["HIGH"]["RESIDUAL_READY"] + outcome["secondary_ready"] = ready_by["SECONDARY"]["RESIDUAL_READY"] + outcome["total_ready"] = total_ready + analysis = { + "baseline": { + "high_ready": BASELINE_HIGH_READY, + "secondary_ready": BASELINE_SECONDARY_READY, + "total_ready": BASELINE_TOTAL_READY, + }, + "candidate_status": outcome["candidate_status"], + "change_versus_lexical_baseline": { + "high_ready_delta": outcome["high_ready"] - BASELINE_HIGH_READY, + "secondary_ready_delta": outcome["secondary_ready"] - BASELINE_SECONDARY_READY, + "total_ready_delta": total_ready - BASELINE_TOTAL_READY, + }, + "combined_constituent_counts": summary["combined_constituent_counts"], + "confidence": confidence, + "determinism": IDENTICAL, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "operator_labels_joined_after_projection_was_hashed": True, + "operator_labels_used_during_inference": False, + "projection_sha256": projection_sha, + "ready_by_operator": ready_by, + "residual_embeddings_used": False, + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "row_counts": summary["row_counts"], + "rule": RULE_VERSION, + "schema": "hyperlex.model_based_wsd_candidate_v1_development_analysis.v1", + "selected_source": "none", + "spec_sha256": spec_sha, + "tier3_attempts": len(resolutions), + "tier3_counts": projection["tier3_counts"], + "yes_no_emitted": False, + } + for banned in ("accuracy", "precision", "recall", "f1"): + if banned in analysis: + refuse("analysis reports an accuracy metric") + analysis_sha = write_json(ANALYSIS_PATH, analysis) + hashes["analysis_sha256"] = analysis_sha + hashes["decision_sha256"] = write_decision(outcome, hashes, analysis_sha) + tracker_sha = update_tracker(outcome, hashes) + if sha256(SPEC_PATH) != spec_sha or sha256(PROVENANCE_PATH) != provenance_sha: + refuse("late write mutated the frozen spec") + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "combined_constituent_counts": summary["combined_constituent_counts"], + "decision_sha256": hashes["decision_sha256"], + "determinism": IDENTICAL, + "high_ready": outcome["high_ready"], + "license_sha256": license_sha, + "projection_sha256": projection_sha, + "provenance_sha256": provenance_sha, + "raw_sha256": raw_sha, + "reject_ready": ready_by["REJECT"]["RESIDUAL_READY"], + "resolution_sha256": resolution_sha, + "secondary_ready": outcome["secondary_ready"], + "spec_sha256": spec_sha, + "tier3_counts": projection["tier3_counts"], + "total_ready": total_ready, + "tracker_sha256": tracker_sha, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py b/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py new file mode 100644 index 00000000..95346449 --- /dev/null +++ b/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py @@ -0,0 +1,752 @@ +"""Score the semantic-compositionality residual on the 225 development rows. + +The candidate specification is written before any row is encoded. Operator +labels are read only after the score artifact has been hashed. The pass does +not select the source, integrate it, or draw a measurement sample. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from collections import Counter, defaultdict +from datetime import datetime, timezone +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + COMPOSITION_OPERATOR, + DISTANCE_METRIC, + MODEL_NAME, + MODEL_REVISION, + candidate_policy, + distribution, + evaluation_status, + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + primary_abstention, + representation_text, + resolve_constituent, + resolved_synset, + score_record, + select_lemma, + vector_hash, +) +from hyperlexical.unbind_sense_screen_v1 import load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +MODEL_DIR = ( + LEDGER + / "acquisition/sources/all-MiniLM-L6-v2" + / MODEL_REVISION +) +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "RESIDUAL_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "RESIDUAL_LICENSE_RECEIPT.json" +SPEC_PATH = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +SCORES_PATH = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +EVALUATION_PATH = SOURCE / "RESIDUAL_DEVELOPMENT_EVALUATION.json" +DECISION_PATH = SOURCE / "RESIDUAL_CANDIDATE_DECISION.json" + +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +OPERATOR_COUNTS = {"HIGH": 101, "SECONDARY": 46, "REJECT": 77, "QUARANTINE": 1, "UNRESOLVED": 0} +POS_NAME = {"n": "noun", "v": "verb", "a": "adj", "r": "adv", "s": "adj"} +FILE_POS = frozenset({"noun", "verb", "adj", "adv"}) +MAX_SEQUENCE_LENGTH = 256 +OUTPUT_DIMENSION = 384 + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + SOURCE / "MAGPIE_SOURCE_PROVENANCE.json": "bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd", + SOURCE / "MAGPIE_LICENSE_RECEIPT.json": "8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972", + SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl": "84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83", + SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl": "037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66", + SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl": "89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577", + SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json": "74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba", + SOURCE / "MAGPIE_CANDIDATE_DECISION.json": "6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d", + SOURCE / "KM_SOURCE_PROVENANCE.json": "88fbfa077d2394b8ce631ec700c482ad98a0f62f0ab06964f4f43a77d906f9aa", + SOURCE / "KM_LICENSE_RECEIPT.json": "eb4c9406aab7f9021d346ebd24634cad1dcf0e6076f069e73b1d4b2702dbe2f8", + SOURCE / "KM_RAW_EVALUATION_INVENTORY.jsonl": "3a35ddc0b2c04a5386c6112a2bb3cdf22735edbe2fd791f0c2ec542fe9184d81", + SOURCE / "KM_PWN30_ALIGNMENT.jsonl": "531d13cf1bdbc939fc11d9ef5864c6878a9f58210368c596c0aad08ff75e61c9", + SOURCE / "KM_HYPERLEX_SEMANTIC_EVIDENCE.jsonl": "4e98f8f06713ffcf02549305aef140e92b9b8b2790f471f3ce1f41fe208b2f7a", + SOURCE / "KM_DEVELOPMENT_EVALUATION.json": "00a1d1f667635354e20e5002c4ece846fe3a8125a7ca12ebe09bb7e28dedd1a7", + SOURCE / "KM_MAGPIE_COMPARISON.json": "240b3ea468d22a80ac5e5765521cc691b80f914b7baff6bb1ae4819475f54965", + SOURCE / "KM_CANDIDATE_DECISION.json": "93a07e3c78b53a69965497410c34ddb52c2a5d3add3fb2f3cd3fd9ca84eb3d9f", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "b3546102410058d3596c4563604998685753a698b2bc533b353e1f039440f704", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + HYPERLEX / "scripts/shadow/hyperlexical/km_candidate_evaluation.py": "c6e6cb69a215de395023fa44237fc4a9f1a02199fd4b92627b7d9345b2b5fbeb", + HYPERLEX / "scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py": "274da840325e0782c938a42a7e5f8ca7a3945982391cd144a50e674059bb333f", + HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation.py": "3cf19b6e3468e55d0636d4df0d2882bc886e9c7024548653a08c26f7ab43c7a9", + HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py": "f01c3361956ae81df772c7b78448d7a58dad342a1f4fe7d084ef00a889587710", + WORDNET / "data.noun": "489f145e0f68877c0be5bd0eb4117adaaac52f38f6204eb8d85dbe2158b614cc", + WORDNET / "data.verb": "29cc96ed80c9f47d94fe75e332a9df80f4b1c737205f92d2f433d63c6da2ab51", + WORDNET / "data.adj": "f24b635368be441501c9b8001e9271fd3b30b203f00d91e332979e6f8fe35646", + WORDNET / "data.adv": "e66dbbda0e0359e41b7f225bff71dd0c263dc7c66c1b61abc9ba334973d92979", + WORDNET / "README": "adad8d28ddea1db05b67ba1ac23506b025d29e0bcbf23bb35dde346089d8808d", + WORDNET / "LICENSE": "7731175a77952e259390b496fab905e57118b8d19ad3a8383c67eee724ff443f", + MODEL_DIR / "model.safetensors": "53aa51172d142c89d9012cce15ae4d6cc0ca6895895114379cacb4fab128d9db", + MODEL_DIR / "tokenizer.json": "be50c3628f2bf5bb5e3a7f17b1f74611b2561a3a27eeab05e5aa30f411572037", + MODEL_DIR / "vocab.txt": "07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3", +} + +MODEL_FILES = ( + "1_Pooling/config.json", + "README.md", + "config.json", + "config_sentence_transformers.json", + "model.safetensors", + "modules.json", + "sentence_bert_config.json", + "special_tokens_map.json", + "tokenizer.json", + "tokenizer_config.json", + "vocab.txt", +) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def package_license(name: str) -> str: + root = Path("/home/morpheus/hlx-private/venv-residual/lib") + matches = sorted(root.glob(f"python*/site-packages/{name}-*.dist-info/METADATA")) + if not matches: + refuse(f"missing package metadata for {name}") + header = matches[-1].read_text(encoding="utf-8", errors="replace").split("\n\n", 1)[0] + expression = None + generic = None + classified = None + for line in header.splitlines(): + if line.startswith("License-Expression:"): + expression = line.split(":", 1)[1].strip() + elif line.startswith("License:"): + generic = line.split(":", 1)[1].strip() + elif line.startswith("Classifier: License"): + classified = line.split("::")[-1].strip() + found = expression or generic or classified + if not found: + refuse(f"no license line for {name}") + return found + + +def build_indexes() -> tuple[dict, dict]: + synsets, glosses = load_wordnet(WORDNET) + by_id = {} + index = defaultdict(list) + for (pos, offset), synset in synsets.items(): + if pos not in FILE_POS: + continue + synset_id = f"{pos}:{offset}" + if synset_id in by_id: + continue + by_id[synset_id] = { + "gloss": " ".join(glosses[(pos, offset)].split()), + "lemmas": list(synset.lemmas), + "pos": pos, + "synset": synset, + } + for lemma in synset.lemmas: + index[lookup_key(lemma)].append(synset_id) + for key, identifiers in index.items(): + index[key] = sorted(set(identifiers)) + return by_id, dict(index) + + +def pointer_records(synset_id: str, by_id: dict) -> list[tuple[str, int, str, str]]: + record = by_id.get(synset_id) + if record is None: + return [] + rows = [] + for pointer in record["synset"].pointers: + pos = POS_NAME.get(pointer.pos) + if pos is None: + continue + target_id = f"{pos}:{pointer.offset}" + target = by_id.get(target_id) + if target is None or pointer.target < 1 or pointer.target > len(target["lemmas"]): + lemma = "" + else: + lemma = target["lemmas"][pointer.target - 1] + rows.append((pointer.symbol, pointer.target, target_id, lemma)) + return rows + + +def prepare_rows(manifest_rows: list[dict], by_id: dict, index: dict, exceptions: dict[str, set[str]]) -> list[dict]: + prepared = [] + for row in manifest_rows: + if row.get("sense_class") is not None: + refuse("development row has a sense class") + if row.get("pos") != row.get("synset_pos"): + refuse("row POS and synset POS differ") + if not row.get("synset_offset") or not row.get("gloss") or not row.get("surface"): + refuse("development row is missing surface, gloss, or synset") + synset = f"{row['synset_pos']}:{row['synset_offset']}" + extraction = extract_constituents(row["surface"]) + pointers = pointer_records(synset, by_id) + resolutions = [] + synsets = [] + lemmas = [] + representations = [] + for constituent in extraction["content_constituents"]: + exact = exact_synset_ids(pointers, constituent, exceptions) + lexical = lexical_synset_ids(index, constituent, exceptions) + status = resolve_constituent(exact, lexical) + chosen = resolved_synset(exact, lexical) + resolutions.append(status) + synsets.append(chosen) + if chosen is None: + lemmas.append(None) + continue + record = by_id.get(chosen) + if record is None: + refuse(f"resolved synset is not in PWN 3.0: {chosen}") + if status == "EXACT": + pool = [ + lemma + for symbol, target_word, target_id, lemma in pointers + if target_id == chosen and symbol in {"+", "\\"} and target_word > 0 and lemma + ] + else: + pool = list(record["lemmas"]) + lemma = select_lemma(pool, constituent, exceptions) + lemmas.append(lemma) + representations.append(representation_text(lemma, record["pos"], record["gloss"])) + reason = primary_abstention(extraction["constituent_extraction_status"], resolutions) + whole = None + constituent_texts = None + if reason is None: + whole = representation_text(row["surface"], row["pos"], row["gloss"]) + constituent_texts = representations + prepared.append( + { + "constituent_representations": constituent_texts, + "extraction": extraction, + "gloss": row["gloss"], + "lemmas": lemmas, + "pos": row["pos"], + "resolutions": resolutions, + "row_id": row["row_id"], + "surface": row["surface"], + "synset": synset, + "synsets": synsets, + "whole_representation": whole, + } + ) + return prepared + + +def load_encoder(): + import torch + from sentence_transformers import SentenceTransformer + + torch.manual_seed(0) + torch.set_num_threads(1) + try: + torch.set_num_interop_threads(1) + except RuntimeError: + pass + model = SentenceTransformer( + str(MODEL_DIR), + device="cpu", + local_files_only=True, + backend="torch", + model_kwargs={"torch_dtype": torch.float32}, + ) + model.eval() + if int(model.max_seq_length) != MAX_SEQUENCE_LENGTH: + refuse(f"max sequence length is {model.max_seq_length}") + return model + + +def token_length(model, text: str) -> int: + encoded = model.tokenizer(text, add_special_tokens=True, truncation=False) + return len(encoded["input_ids"]) + + +def encode_texts(model, texts: list[str]) -> dict[str, list[float]]: + import torch + + encoded = {} + with torch.inference_mode(): + for text in texts: + torch.manual_seed(0) + vector = model.encode( + [text], + batch_size=1, + convert_to_numpy=True, + device="cpu", + normalize_embeddings=True, + precision="float32", + show_progress_bar=False, + )[0] + values = [float(item) for item in vector.tolist()] + if len(values) != OUTPUT_DIMENSION: + refuse(f"encoder width is {len(values)}") + encoded[text] = values + return encoded + + +def runtime_versions() -> dict: + import numpy + import tokenizers + import torch + import transformers + import sentence_transformers + + return { + "numpy": numpy.__version__, + "python": ".".join(map(str, __import__("sys").version_info[:3])), + "sentence_transformers": sentence_transformers.__version__, + "tokenizers": tokenizers.__version__, + "torch": torch.__version__, + "transformers": transformers.__version__, + } + + +def file_hashes() -> dict[str, str]: + return {name: sha256(MODEL_DIR / name) for name in MODEL_FILES} + + +def iso_mtime(path: Path) -> str: + stamp = datetime.fromtimestamp(path.stat().st_mtime, timezone.utc) + return stamp.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def main() -> None: + check_sealed() + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225: + refuse(f"manifest row count is {len(rows)}") + for row in rows: + row.pop("operator_bucket", None) + row.pop("operator_reason", None) + row.pop("historical_unbind_screen", None) + exceptions = load_exceptions(WORDNET) + by_id, index = build_indexes() + prepared = prepare_rows(rows, by_id, index, exceptions) + if len(prepared) != 225: + refuse("prepared row count drifted") + + import torch + + torch.manual_seed(0) + torch.set_num_threads(1) + versions = runtime_versions() + hashes = file_hashes() + pooling = json.loads((MODEL_DIR / "1_Pooling/config.json").read_text(encoding="utf-8")) + tokenizer_config = json.loads((MODEL_DIR / "tokenizer_config.json").read_text(encoding="utf-8")) + modules = json.loads((MODEL_DIR / "modules.json").read_text(encoding="utf-8")) + provenance = { + "acquired_at": iso_mtime(MODEL_DIR / "model.safetensors"), + "api_embedding_service_used": False, + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "device": "cpu", + "dtype": "float32", + "file_sha256": hashes, + "json_schema_document": None, + "local_dir": str(MODEL_DIR), + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "model_source": "https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2", + "modules": modules, + "not_acquired": ["onnx", "openvino", "pytorch_model.bin", "tf_model.h5"], + "output_dimension": OUTPUT_DIMENSION, + "pooling": pooling, + "runtime_versions": versions, + "schema": "hyperlex.residual_source_provenance.v1", + "tokenizer_do_lower_case": tokenizer_config["do_lower_case"], + "tokenizer_model_max_length": tokenizer_config["model_max_length"], + "weights_sha256": hashes["model.safetensors"], + "wordnet_license_sha256": EXPECTED[WORDNET / "LICENSE"], + "wordnet_readme_sha256": EXPECTED[WORDNET / "README"], + "wordnet_root": str(WORDNET), + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + license_receipt = { + "api_embedding_service_used": False, + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "json_schema_document": None, + "model_license": "apache-2.0", + "model_license_source": "README.md front matter of the pinned revision", + "runtime_licenses": { + "numpy": package_license("numpy"), + "sentence_transformers": package_license("sentence_transformers"), + "tokenizers": package_license("tokenizers"), + "torch": package_license("torch"), + "transformers": package_license("transformers"), + }, + "schema": "hyperlex.residual_license_receipt.v1", + "wordnet_license": "WordNet Release 3.0, Copyright 2006 Princeton University", + "wordnet_license_sha256": EXPECTED[WORDNET / "LICENSE"], + } + receipt_sha = write_json(LICENSE_PATH, license_receipt) + spec = { + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "cuda_available_at_spec_freeze": bool(torch.cuda.is_available()), + "determinism": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "inference_mode": True, + "interop_threads": 1, + "mkl_num_threads": "1", + "normalize_embeddings": True, + "num_threads": 1, + "omp_num_threads": "1", + "precision": "float32", + "seed": 0, + "tokenizers_parallelism": "false", + "use_deterministic_algorithms": False, + }, + "effective_max_sequence_length": MAX_SEQUENCE_LENGTH, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "output_dimension": OUTPUT_DIMENSION, + "policy": candidate_policy(), + "provenance_sha256": provenance_sha, + "row_scores_included": False, + "runtime_versions": versions, + "schema": "hyperlex.residual_candidate_spec.v1", + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + "tokenizer_json_sha256": hashes["tokenizer.json"], + "tokenizer_revision": MODEL_REVISION, + "vocab_sha256": hashes["vocab.txt"], + "weights_sha256": hashes["model.safetensors"], + } + spec_text = json.dumps(spec, indent=2, sort_keys=True, ensure_ascii=True) + if "residual_score" in spec_text or "operator_bucket" in spec_text: + refuse("candidate spec contains a score or an operator label") + spec_sha = write_json(SPEC_PATH, spec) + if sha256(SPEC_PATH) != spec_sha: + refuse("spec hash mismatch") + + model = load_encoder() + for item in prepared: + item["sequence_overflow"] = False + if item["whole_representation"] is None: + continue + texts = [item["whole_representation"], *item["constituent_representations"]] + if any(token_length(model, text) > MAX_SEQUENCE_LENGTH for text in texts): + item["sequence_overflow"] = True + item["whole_representation"] = None + item["constituent_representations"] = None + needed = [] + seen = set() + for item in prepared: + if item["whole_representation"] is None: + continue + for text in [item["whole_representation"], *item["constituent_representations"]]: + if text not in seen: + seen.add(text) + needed.append(text) + needed.sort() + first = encode_texts(model, needed) + second = encode_texts(model, needed) + for text in needed: + if vector_hash(first[text]) != vector_hash(second[text]): + refuse("encoder replay did not match") + vectors = first + + scores = [] + for item in prepared: + whole_vector = None + constituent_vectors = None + if item["whole_representation"] is not None: + whole_vector = vectors[item["whole_representation"]] + constituent_vectors = [vectors[text] for text in item["constituent_representations"]] + record = score_record( + row_id=item["row_id"], + surface=item["surface"], + pos=item["pos"], + synset=item["synset"], + extraction=item["extraction"], + resolutions=item["resolutions"], + resolved_synsets=item["synsets"], + resolved_lemmas=item["lemmas"], + whole_representation=item["whole_representation"], + constituent_representations=item["constituent_representations"], + whole_vector=whole_vector, + constituent_vectors=constituent_vectors, + candidate_spec_sha256=spec_sha, + sequence_overflow=item["sequence_overflow"], + ) + if "operator_bucket" in record: + refuse("score row carries an operator bucket") + scores.append(record) + if [row["row_id"] for row in scores] != [row["row_id"] for row in rows]: + refuse("score order drifted from the manifest") + score_sha = write_jsonl(SCORES_PATH, scores) + if sha256(SCORES_PATH) != score_sha: + refuse("score hash mismatch") + if sha256(SPEC_PATH) != spec_sha: + refuse("scoring mutated the spec") + + rejoined = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {} + for row in rejoined["rows"]: + buckets[row["row_id"]] = row["operator_bucket"] + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + if any(row.get("sense_class") is not None for row in rejoined["rows"]): + refuse("manifest sense class changed") + + by_operator = {name: [] for name in OPERATORS} + unknown_by_operator = Counter() + for record in scores: + bucket = buckets[record["row_id"]] + if bucket not in by_operator: + refuse(f"unexpected operator bucket {bucket}") + if record["score_status"] == "SCORED": + by_operator[bucket].append(record["residual_score"]) + else: + unknown_by_operator[bucket] += 1 + distributions = {name: distribution(by_operator[name]) for name in OPERATORS} + high = distributions["HIGH"] + high_comparison = "NOT_COMPUTABLE" if high["count"] == 0 else "DESCRIPTIVE_ONLY" + scored_count = sum(item["count"] for item in distributions.values()) + status, next_transition = evaluation_status(scored_count) + extraction_counts = Counter(record["constituent_extraction_status"] for record in scores) + resolution_counts = Counter( + status_name + for record in scores + for status_name in record["constituent_resolution_status"] + ) + abstention_counts = Counter( + record["primary_abstention_reason"] + for record in scores + if record["primary_abstention_reason"] + ) + odd_content_tokens = 0 + content_tokens = 0 + for record in scores: + for token in record["content_constituents"]: + content_tokens += 1 + folded = token.casefold() + if any(not (character.isalpha() or character.isdigit() or character in "-'") for character in folded): + odd_content_tokens += 1 + evaluation = { + "abstention_reason_counts": dict(sorted(abstention_counts.items())), + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "candidate_rule": "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1", + "candidate_spec_sha256": spec_sha, + "composition_operator": COMPOSITION_OPERATOR, + "content_tokens": content_tokens, + "content_tokens_outside_letter_digit_hyphen_apostrophe": odd_content_tokens, + "distance_metric": DISTANCE_METRIC, + "distributions_by_operator": distributions, + "emits_yes_no": False, + "encode_replay_matched": True, + "encoded_text_count": len(needed), + "extraction_counts": dict(sorted(extraction_counts.items())), + "high_comparison": high_comparison, + "high_distribution": high, + "high_proxy": "expected noncompositionality; not an identity", + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "operator_counts": {name: OPERATOR_COUNTS[name] for name in OPERATORS}, + "operator_labels_joined_after_score_artifact_was_hashed": True, + "operator_labels_used_as_scoring_inputs": False, + "operator_unknown_counts": {name: unknown_by_operator[name] for name in OPERATORS}, + "primary_analysis_bucket": "HIGH", + "quarantine_is_not_semantic_evidence": True, + "reject_is_not_semantic_no": True, + "resolution_counts": dict(sorted(resolution_counts.items())), + "row_count": 225, + "schema": "hyperlex.residual_development_evaluation.v1", + "score_artifact_sha256": score_sha, + "scored_count": scored_count, + "secondary_is_a_proxy_for_expected_compositionality": True, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "source_provenance_sha256": provenance_sha, + "unknown_count": 225 - scored_count, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + decision = { + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "candidate_rule": "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1", + "candidate_spec_sha256": spec_sha, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "high_comparison": high_comparison, + "high_scored_count": high["count"], + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": next_transition, + "next_transition_authorized": False, + "runtime_integration": False, + "schema": "hyperlex.residual_candidate_decision.v1", + "score_artifact_sha256": score_sha, + "scored_count": scored_count, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "source_provenance_sha256": provenance_sha, + "state": "CANDIDATE_SOURCE_EVALUATED", + "unknown_count": 225 - scored_count, + "why": ( + "The residual is a continuous score. No semantic-noncompositionality " + "threshold is chosen in this pass, and operator labels were joined " + "only after the score file was hashed." + ), + "yes_no_emitted": False, + } + decision_sha = write_json(DECISION_PATH, decision) + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["evaluated_candidates"] = [ + "MAGPIE", + "KORKONTZELOS_MANANDHAR", + "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + ] + tracker["latest_evaluated_candidate"] = "SEMANTIC_COMPOSITIONALITY_RESIDUAL" + tracker["residual_candidate"] = "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1" + tracker["residual_evaluation_status"] = status + tracker["residual_candidate_spec_sha256"] = spec_sha + tracker["residual_source_provenance_sha256"] = provenance_sha + tracker["residual_license_receipt_sha256"] = receipt_sha + tracker["residual_development_scores_sha256"] = score_sha + tracker["residual_development_evaluation_sha256"] = evaluation_sha + tracker["residual_candidate_decision_sha256"] = decision_sha + tracker["residual_model_name"] = MODEL_NAME + tracker["residual_model_revision"] = MODEL_REVISION + tracker["residual_scored_count"] = scored_count + tracker["residual_unknown_count"] = 225 - scored_count + tracker["residual_high_comparison"] = high_comparison + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = next_transition + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + print( + json.dumps( + { + "abstention_reason_counts": evaluation["abstention_reason_counts"], + "candidate_decision_sha256": decision_sha, + "candidate_spec_sha256": spec_sha, + "development_evaluation_sha256": evaluation_sha, + "development_scores_sha256": score_sha, + "distributions_by_operator": distributions, + "encoded_text_count": len(needed), + "evaluation_status": status, + "extraction_counts": evaluation["extraction_counts"], + "high_comparison": high_comparison, + "license_receipt_sha256": receipt_sha, + "next_legal_transition": next_transition, + "resolution_counts": evaluation["resolution_counts"], + "scored_count": scored_count, + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + "tracker_sha256": tracker_sha, + "unknown_count": 225 - scored_count, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py new file mode 100644 index 00000000..0fab3c0c --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py @@ -0,0 +1,681 @@ +"""Replay the frozen 197-row set, then draw one unseen measurement sample. + +The replay is regression evidence. The draw excludes those identities, freezes +the sample, and only then applies v7 once. It does not label, admit, settle, +or append the ledger. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import WordNetLexicon +from hyperlexical.unbind_screen_v5 import WordNetLexicon as V5Lexicon +from hyperlexical.unbind_screen_v5 import apply_v5 +from hyperlexical.unbind_screen_v6 import apply_v6 +from hyperlexical.unbind_screen_v7 import RULE_VERSION, apply_v7, assess + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +V6 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-001/unbind_screen_v6" +V6H = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-HYPOTHESIS-001" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7" +SPEC = Path("/home/morpheus/Hyperlex/specs/007-hyperlexical-model/evaluation-reserve.md") +SCORER = Path(__file__).with_name("unbind_screen_v7.py") +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V6_SAMPLE = "8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2" +EXPECTED_V6_PREDICTIONS = "476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee" +EXPECTED_V6_LABELS = "feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11" +EXPECTED_V6_ACCEPTANCE = "68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f" +EXPECTED_V6_SOURCE = "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba" +EXPECTED_DRAFT = "c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b" +EXPECTED_ACCEPTANCE = "562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5" +EXPECTED_EVIDENCE = "3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3" +EXPECTED_ANALYSIS = "06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766" +EXPECTED_REPLAY = "24f266e16b80da602b011bf7cca13774ce9f76c0c52e81e600a1d9743d61d197" +EXPECTED_GATE = "78225b68c639694bb4c17342c87320ccb44faac557a9145ea117603da0d545ed" +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "operator_reason", "forecast", "diagnostic", + "v3", "v4", "v5", "v6", "v7", "primary_evidence", "supporting_evidence", + "evidence", "v3_bucket", "v4_bucket", "v5_bucket", "v6_bucket", "v7_bucket", + "ordinary_compositional_derivation", +}) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def _require_frozen_parents() -> None: + checks = { + LEDGER / "events.jsonl": EXPECTED_EVENTS, + V6 / "measurement_sample.jsonl": EXPECTED_V6_SAMPLE, + V6 / "measurement_predictions.jsonl": EXPECTED_V6_PREDICTIONS, + V6 / "measurement_labels.jsonl": EXPECTED_V6_LABELS, + V6 / "measurement_error_analysis.json": EXPECTED_ANALYSIS, + V6H / "ACCEPTANCE.json": EXPECTED_V6_ACCEPTANCE, + V6H / "HYPOTHESIS.json": "fd4f5eeb66061ffdaf2aa4341fe86a0b5e988f131d1711d6ea299a4ebb9e1da4", + HYP / "HYPOTHESIS.draft.json": EXPECTED_DRAFT, + HYP / "ACCEPTANCE.json": EXPECTED_ACCEPTANCE, + HYP / "DEVELOPMENT_EVIDENCE.json": EXPECTED_EVIDENCE, + Path("/home/morpheus/Hyperlex/scripts/shadow/hyperlexical/unbind_screen_v6.py"): EXPECTED_V6_SOURCE, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path}") + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V6 / "v6_replay_169_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "v5_bucket": raw["v5_bucket"], + "v6_bucket": raw["v6_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_169", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v6 measurement identities differ") + seen = {row["row_id"] for row in rows} + if seen & set(sample): + raise SystemExit("v6 measurement overlaps the prior reviewed surfaces") + reviewed_norm = {normalize_lexical(row["surface"]) for row in rows} + measurement_norm = {normalize_lexical(row["surface"]) for row in sample.values()} + if reviewed_norm & measurement_norm: + raise SystemExit("v6 measurement reuses a normalized reviewed identity") + for row_id, blind in sample.items(): + pred = predictions[row_id] + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": pred["v3_bucket"], + "v4_bucket": pred["v4_bucket"], + "v5_bucket": pred["v5_bucket"], + "v6_bucket": pred["v6_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v6_measurement_28", + "row_id": row_id, + }) + if len(rows) != 197 or len({row["row_id"] for row in rows}) != 197: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 197 unique") + if len({row["surface"] for row in rows}) != 197: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "v7_replay_197_predictions.jsonl").exists(): + raise SystemExit("v7 replay is already frozen") + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is not authorized") + if "Unbind screen v7 encoded" in SPEC.read_text(encoding="utf-8"): + raise SystemExit("spec already records the v7 replay") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + if acceptance.get("state") != "SPEC_FROZEN" or acceptance.get("implementation_encoded") is not False: + raise SystemExit("acceptance is not the frozen specification") + probes = acceptance["probe_surfaces"]["forbidden_in_executable_rule_code"] + violations = rule_surface_violations(SCORER.read_text(encoding="utf-8"), probes) + lexicon = WordNetLexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v7(raw["v6_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + predictions.append({ + "schema": "hyperlex.unbind_screen_v7_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + "v4_bucket": canonical_bucket(raw["v4_bucket"]), + "v5_bucket": canonical_bucket(raw["v5_bucket"]), + **decision, + }) + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=197) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["draft_hypothesis_sha256"] = file_sha256(HYP / "HYPOTHESIS.draft.json") + report["development_evidence_sha256"] = file_sha256(HYP / "DEVELOPMENT_EVIDENCE.json") + report["error_analysis_sha256"] = file_sha256(V6 / "measurement_error_analysis.json") + report["events_sha256"] = events_sha256() + report["measurement_sample_drawn"] = False + report["measurement_eligible"] = False + report["regression_is_not_generalization"] = True + if report["replay_rows"] != 197: + report["failures"].append("replay count is not 197") + report["assertions"]["A_replay_197"] = False + report["regression"] = "REGRESSION_FAILED" + report["state"] = "ENCODED" + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v6_bucket"] != row["v7_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v7_replay_197_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v7_replay_197_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v7_replay_197_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v7_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + _append_spec(report, changed) + _require_frozen_parents() + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("a measurement sample was drawn") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + verified = report["regression"] == "REGRESSION_VERIFIED" + return { + "schema": "hyperlex.unbind_screen_v7_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "regression": report["regression"], + "measurement_eligible": False, + "measurement_sample_drawn": False, + "regression_is_not_generalization": True, + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v6": "one_high_challenge_over_a_frozen_v6_bucket", + "demotion_stops_for_this_application": True, + "acceptance_sha256": report["acceptance_sha256"], + "draft_hypothesis_sha256": report["draft_hypothesis_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "assertions": report["assertions"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "correct_high": report["correct_high"], + "correct_high_lost": report["correct_high_lost"], + "direct_swaps": report["direct_swaps"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + if file_sha256(HYP / "HYPOTHESIS.draft.json") != EXPECTED_DRAFT: + raise SystemExit("v7 draft bytes changed") + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("v7 acceptance bytes changed") + if hypothesis.get("acceptance_sha256") != EXPECTED_ACCEPTANCE: + raise SystemExit("v7 hypothesis no longer points at the frozen acceptance") + if hypothesis.get("draft_hypothesis_sha256") != EXPECTED_DRAFT: + raise SystemExit("v7 hypothesis no longer points at the frozen draft") + hypothesis["encoded"] = True + hypothesis["implementation_encoded"] = True + hypothesis["encoding_authorized"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = False + hypothesis["measurement_sample_drawn"] = False + hypothesis["applied"] = False + hypothesis["applied_to_measurement"] = False + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["previously_correct_lost"] = receipt["previously_correct_lost"] + hypothesis["correct_high_lost"] = receipt["correct_high_lost"] + hypothesis["direct_swaps"] = receipt["direct_swaps"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + hypothesis["regression_is_not_generalization"] = True + hypothesis["precision"] = "NOT_COMPUTABLE" + text = json.dumps(hypothesis, indent=2, sort_keys=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _append_spec(report: dict, changed: list[dict]) -> None: + moves = report["moves"] + if len(changed) == 1: + row = changed[0] + change = ( + f"One row moves. `{row['surface']}` goes from high to secondary with " + f"`{row['primary_evidence']}`, and the operator bucket is {row['operator_bucket'].lower()}. " + "That firing is not a required fit." + ) + elif not changed: + change = "No row moves." + else: + names = ", ".join(f"`{row['surface']}`" for row in changed) + change = f"{len(changed)} rows move: {names}." + verified = report["regression"] == "REGRESSION_VERIFIED" + if verified: + state = ( + "State is `REGRESSION_VERIFIED`. `measurement_eligible` stays false. " + "The unseen sample was not drawn." + ) + else: + state = ( + "State stays `ENCODED` because the regression gate failed. " + "`measurement_eligible` stays false. The unseen sample was not drawn." + ) + section = f""" +## Unbind screen v7 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7` is encoded over a frozen v6 bucket. The only new predicate is `ordinary_compositional_derivation`. It inspects a provisional high. The recorded sense must be a comparative, syntactic, or phrasal composition, and the synset must not store an unrelated single-word synonym. A gloss that merely shares constituent stems does not fire, and a metaphorical retelling does not fire. A demotion stops at secondary. Reject rows still use the v6 reject challenge. Secondary rows are not reopened, so a demotion v6 already stopped stays stopped. `compositional_recoverability` is not a v7 transition. The acceptance contract is unchanged, sha256 `{EXPECTED_ACCEPTANCE}`. + +The 197 reviewed surfaces were replayed as development and regression evidence, not as a generalization estimate. Previously correct rows lost: {report['previously_correct_lost']}. Previously correct high rows demoted: {report['correct_high_lost']}. Direct swaps between high and reject: {report['direct_swaps']}. High to secondary: {moves['high_to_secondary']}. Reject to secondary: {moves['reject_to_secondary']}. Secondary to high: {moves['secondary_to_high']}. Secondary to reject: {moves['secondary_to_reject']}. {change} No phrase-specific rule fired. This replay is not a precision estimate. Gate report sha256 `{report['replay_rows'] and ''}`. Prediction sha256 `{report['prediction_sha256']}`. Diff sha256 `{report['diff_sha256']}`. + +{state} The pre-registered bar remains high precision 1, reject precision 1, false high 0, and false reject 0, with no false-secondary quota, no accuracy target, and no recall target. No measurement sample was drawn. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `{EXPECTED_EVENTS}`. +""" + # The gate-report hash is the file hash, filled by the caller after the report is dumped. + # This placeholder is replaced below once the file exists. + spec = SPEC.read_text(encoding="utf-8") + if not spec.endswith("\n"): + spec += "\n" + SPEC.write_text(spec + "\n" + section.lstrip("\n"), encoding="utf-8") + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + """Freeze one unseen sample, then apply v7 once.""" + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + if (out_dir / "measurement_predictions.jsonl").exists(): + raise SystemExit("measurement predictions already exist") + if "Unbind screen v7 measurement frozen" in SPEC.read_text(encoding="utf-8"): + raise SystemExit("spec already records the v7 measurement draw") + _require_frozen_parents() + if file_sha256(out_dir / "v7_replay_197_predictions.jsonl") != EXPECTED_REPLAY: + raise SystemExit("v7 replay predictions changed") + if file_sha256(out_dir / "v7_replay_197_gate_report.json") != EXPECTED_GATE: + raise SystemExit("v7 gate report changed") + report = json.loads((out_dir / "v7_replay_197_gate_report.json").read_text(encoding="utf-8")) + if report.get("regression") != "REGRESSION_VERIFIED" or report.get("replay_rows") != 197: + raise SystemExit("measurement draw refused: regression is not verified") + if report.get("previously_correct_lost") != 0 or report.get("failures"): + raise SystemExit("measurement draw refused: regression failures are present") + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("high_precision") != 1.0 or criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_high_allowed") != 0 or criteria.get("false_reject_allowed") != 0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_secondary_rate_required") is not False or criteria.get("accuracy_gain_required") is not False: + raise SystemExit("acceptance bar gained a coverage or accuracy target") + if criteria.get("recall_required") is not False: + raise SystemExit("acceptance bar gained a recall target") + if criteria.get("next_measurement_excludes_reviewed_surfaces") != 197: + raise SystemExit("acceptance exclusion count changed") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("copying criteria changed the acceptance") + + reviewed = [ + json.loads(line) + for line in (out_dir / "v7_replay_197_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + ] + if len(reviewed) != 197 or len({row["row_id"] for row in reviewed}) != 197: + raise SystemExit("reviewed replay is not 197 unique rows") + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + if len(blocked_norm) != 197: + raise SystemExit("reviewed normalized identities are not unique") + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + exhausted = [cell for cell in strata if cell["available"] < per_cell or cell["taken"] < per_cell] + for cell in strata: + if cell["taken"] > cell["available"] or cell["taken"] > per_cell: + raise SystemExit("stratum draw exceeded the frozen policy") + + blinded = [] + sources = {} + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface or len(tokens) < 1: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + row_id = normalized_text_sha256(surface) + identity = normalize_lexical(surface) + if row_id in blocked_hash or identity in blocked_norm: + raise SystemExit(f"draw reused a reviewed identity: {surface}") + blind = { + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V7-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 197, + }, + } + leaked = LEAK_KEYS.intersection(blind) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + blinded.append(blind) + sources[row_id] = {"tokens": tokens, "pos": pos, "surface": surface, "gloss": gloss} + identities = [normalize_lexical(row["surface"]) for row in blinded] + if len(identities) != len(set(identities)): + raise SystemExit("sample contains a duplicate normalized identity") + if set(identities) & blocked_norm or {row["row_id"] for row in blinded} & blocked_hash: + raise SystemExit("sample intersects the reviewed 197") + if {row["surface"] for row in blinded} & {row["surface"] for row in reviewed}: + raise SystemExit("sample reuses a reviewed surface") + blinded.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blinded) + sample_text = (out_dir / "measurement_sample.jsonl").read_text(encoding="utf-8") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash did not stick") + for forbidden in ( + "ordinary_compositional_derivation", + "primary_evidence", + "operator_bucket", + "operator_reason", + "v7_bucket", + "v6_bucket", + "v5_bucket", + "v4_bucket", + "v3_bucket", + ): + if forbidden in sample_text: + raise SystemExit(f"blind sample contains {forbidden}") + frozen_rows = [json.loads(line) for line in sample_text.splitlines() if line.strip()] + if [row["row_id"] for row in frozen_rows] != [row["row_id"] for row in blinded]: + raise SystemExit("frozen sample does not match the draw") + + v4_lexicon = V4Lexicon(WORDNET) + v5_lexicon = V5Lexicon(WORDNET) + predictions = [] + applications = 0 + for row in frozen_rows: + source = sources[row["row_id"]] + if source["surface"] != row["surface"] or source["gloss"] != row["gloss"]: + raise SystemExit("apply input drifted from the frozen sample") + if source["pos"] != row["pos"] or len(source["tokens"]) != row["token_count"]: + raise SystemExit("apply input drifted from the frozen sample") + surface = row["surface"] + pos = row["pos"] + tokens = source["tokens"] + gloss = row["gloss"] + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + v5_decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, v5_lexicon) + v6_decision = apply_v6(v5_decision["v5_bucket"], surface, gloss, pos, v5_lexicon) + decision = apply_v7(v6_decision["v6_bucket"], surface, gloss, pos, v5_lexicon) + applications += 1 + if decision["v6_bucket"] != v6_decision["v6_bucket"]: + raise SystemExit("v7 did not keep the provisional v6 bucket") + predictions.append({ + "schema": "hyperlex.unbind_screen_v7_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V7-001", + "sample_id": "measurement-001", + "row_id": row["row_id"], + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + "v4_bucket": v4_decision["v4_bucket"], + "v5_bucket": v5_decision["v5_bucket"], + "provisional_v6_bucket": v6_decision["v6_bucket"], + **decision, + }) + if applications != len(frozen_rows): + raise SystemExit("v7 application count does not match the sample") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash changed while v7 was applied") + predictions.sort(key=lambda row: row["row_id"]) + if [row["row_id"] for row in predictions] != [row["row_id"] for row in frozen_rows]: + raise SystemExit("predictions do not cover the frozen sample") + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v7_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "MEASUREMENT_ELIGIBLE", + "sample_state": "MEASUREMENT_SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(frozen_rows), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "exhausted_strata": exhausted, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 197, + "normalized_overlap_with_reviewed_197": 0, + "duplicate_normalized_identities": 0, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "sample_frozen_before_apply": True, + "v7_application_count": applications, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "false_secondary_rate_required": False, + "accuracy_gain_required": False, + "recall_required": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v7_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_sample_drawn"] = True + receipt["measurement_state"] = "MEASUREMENT_ELIGIBLE" + receipt["sample_state"] = "MEASUREMENT_SAMPLE_FROZEN" + receipt["measurement_rows"] = len(frozen_rows) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v7_application_count"] = applications + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["exhausted_strata"] = len(exhausted) + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v7_implementation_receipt.json", receipt) + _mark_measured(freeze) + _write_review_sheet(out_dir, frozen_rows, sample_sha) + _append_measurement_spec(freeze, exhausted) + _require_frozen_parents() + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted after the draw") + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("acceptance bytes changed during the draw") + if file_sha256(HYP / "HYPOTHESIS.draft.json") != EXPECTED_DRAFT: + raise SystemExit("draft bytes changed during the draw") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + return freeze + + +def _mark_measured(freeze: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + if hypothesis.get("acceptance_sha256") != EXPECTED_ACCEPTANCE: + raise SystemExit("hypothesis no longer points at the frozen acceptance") + if hypothesis.get("draft_hypothesis_sha256") != EXPECTED_DRAFT: + raise SystemExit("hypothesis no longer points at the frozen draft") + hypothesis["state"] = "MEASUREMENT_ELIGIBLE" + hypothesis["measurement_state"] = "MEASUREMENT_ELIGIBLE" + hypothesis["sample_state"] = "MEASUREMENT_SAMPLE_FROZEN" + hypothesis["measurement_eligible"] = True + hypothesis["measurement_sample_drawn"] = True + hypothesis["applied"] = True + hypothesis["applied_to_measurement"] = True + hypothesis["precision"] = "NOT_COMPUTABLE" + hypothesis["measurement_rows"] = freeze["rows"] + hypothesis["measurement_sample_sha256"] = freeze["sample_sha256"] + hypothesis["measurement_prediction_sha256"] = freeze["prediction_sha256"] + hypothesis["hand_corrections"] = 0 + hypothesis["v7_application_count"] = freeze["v7_application_count"] + hypothesis["operator_labels"] = None + hypothesis["labeling_performed"] = False + hypothesis["next_legal_transition"] = "OPERATOR_LABELS" + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["admitted"] = 0 + hypothesis["settled"] = 0 + hypothesis["gold"] = 0 + text = json.dumps(hypothesis, indent=2, sort_keys=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _write_review_sheet(out_dir: Path, rows: list[dict], sample_sha: str) -> None: + lines = [ + "# Unbind screen v7 blind review", + "", + "Prediction buckets and evidence codes are withheld.", + f"Sample sha256 `{sample_sha}`.", + "", + "| row_id | surface | pos | token_count | gloss |", + "|---|---|---|---|---|", + ] + for row in rows: + gloss = row["gloss"].replace("|", "\\|") + lines.append( + f"| `{row['row_id']}` | {row['surface']} | {row['pos']} | {row['token_count']} | {gloss} |" + ) + lines.append("") + path = out_dir / "measurement_review.md" + path.write_text("\n".join(lines), encoding="utf-8") + path.chmod(0o600) + text = path.read_text(encoding="utf-8") + for forbidden in ("ordinary_compositional_derivation", "primary_evidence", "v7_bucket", "operator_bucket"): + if forbidden in text: + raise SystemExit(f"review sheet contains {forbidden}") + + +def _append_measurement_spec(freeze: dict, exhausted: list[dict]) -> None: + if exhausted: + detail = "; ".join( + f"{cell['source_pos']} x {cell['token_count']} available {cell['available']}, taken {cell['taken']}" + for cell in exhausted + ) + exhausted_sentence = f"Exhausted cells, where fewer than 2 rows remained: {detail}." + else: + exhausted_sentence = "No occupied cell was exhausted. Each occupied cell contributed 2 rows." + section = f""" +## Unbind screen v7 measurement frozen — 2026-09-28 + +The 197-row regression stays `REGRESSION_VERIFIED`. One unseen measurement sample excludes those 197 normalized identities. Normalized overlap with the reviewed set is 0. Duplicate normalized identities inside the sample are 0. The draw is deterministic and stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced {freeze['rows']} rows. {exhausted_sentence} + +The sample was frozen before v7 was applied. Sample sha256 `{freeze['sample_sha256']}`. v7 was then applied once. Hand corrections are 0. Prediction sha256 `{freeze['prediction_sha256']}`. The blind rows carry surface, part of speech, token count, and gloss. They do not carry a bucket or an evidence code. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. + +The acceptance bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor, no accuracy target, and no recall target were added. Criteria sha256 `{freeze['criteria_sha256']}`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `{EXPECTED_EVENTS}`. +""" + spec = SPEC.read_text(encoding="utf-8") + if not spec.endswith("\n"): + spec += "\n" + SPEC.write_text(spec + "\n" + section.lstrip("\n"), encoding="utf-8") + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Draw the unseen v7 measurement sample") + parser.add_argument("command", choices=("draw",)) + args = parser.parse_args() + if args.command != "draw": + raise SystemExit("only the measurement draw is authorized") + freeze = draw_measurement() + print(json.dumps({ + "state": freeze["state"], + "sample_state": freeze["sample_state"], + "rows": freeze["rows"], + "exhausted_strata": freeze["exhausted_strata"], + "sample_sha256": freeze["sample_sha256"], + "prediction_sha256": freeze["prediction_sha256"], + "v7_application_count": freeze["v7_application_count"], + "hand_corrections": freeze["hand_corrections"], + "normalized_overlap_with_reviewed_197": freeze["normalized_overlap_with_reviewed_197"], + "duplicate_normalized_identities": freeze["duplicate_normalized_identities"], + "precision": freeze["precision"], + "events_sha256": freeze["events_sha256"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/specs/007-hyperlexical-model/evaluation-reserve.md b/specs/007-hyperlexical-model/evaluation-reserve.md index 7f96f856..d86a06ce 100644 --- a/specs/007-hyperlexical-model/evaluation-reserve.md +++ b/specs/007-hyperlexical-model/evaluation-reserve.md @@ -351,3 +351,487 @@ Fresh operator authorization for `HLX-EXP-2026-09-26-SELECT-004` only. It does n Authorization artifact sha256 `7389783b52a5d5cf14ceca874cc12351d3f6571b6be8a8666c2040be38972625`. Post-authorization admission used `HYPERLEX_ALLOW_TRAIN=1` and `HLX_ADMISSION_ONLY=1`. Preflight and the trainer entrypoint share environment hash `a2b18512be8329ee63ad06e98f894c6138cf7eac05f6e82d53d4132e99a27f5d`. `admission_result` is `ADMISSION_PASS`. Status is `TRAINING_READY` because the scientific contract is sealed, the decision rule is sealed, and the real entrypoint passed. `optimizer_loaded` is false. Epochs 0. Gradient steps 0. Training was not started. Output directory stayed absent. BEST sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1` was not moved. Trunk sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. `training_launch_authorized` is false. `HLX_ALLOW_NO_HOLDOUT` stayed unset. Representation completeness is `PASS`. Evaluation quality is `LIMITED`. The decision covers classify 241, classify_observed 123, classify_non_none 188, and unbind_clean 250. It does not establish performance for relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, or spiritual-mystic. Clean unbind stays limited to the sealed WordNet slice. Vendor calls: 0. Do not train. The next decision is launch authorization only. + +## SELECT-004 scored — HLX-EXP-2026-09-26-SELECT-004 + +One run finished under the sealed contract. Container `hlx-train-select004-1790448703` exited 0 after 40 epochs. The training schedule was not shortened. Checkpoint selection used `classify_macro_f1_nonnone` on the training-export validation surface. The selected checkpoint is epoch 3 at 0.64448782942204. Epoch 39 scored 0.5864567286803334. Candidate weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Consumed export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`, 9150 rows, reserve filter removed 0. The reserve ledger was not modified during training. + +Reserve scoring used the sealed twelve-family universe `227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175`. seed-morph78 macro-F1 0.022395727019119547, accuracy 0.14107883817427386, OBSERVED accuracy 0.032520325203252036, unbind clean exact 0.092. The candidate macro-F1 0.07245710784313726, accuracy 0.15767634854771784, OBSERVED accuracy 0.08130081300813008, unbind clean exact 0.100. Deltas are +0.05006138082401771, +0.01659751037344398, +0.048780487804878044, and +0.008. The char 3–5 baseline macro-F1 was 0.0, so CHAR_WINS did not occur. Decision **PROMOTE**. State `PROMOTION_ELIGIBLE`. `--apply-best` was not run. BEST remains `seed-morph78` / `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. The production head still emits the nine training families, so nine of the twelve gold families cannot be named and pull the absolute macro-F1 down for both checkpoints. The claim stays inside the sealed slices. Vendor calls: 0. The 40-epoch schedule was left as sealed. A later duration study can measure best-epoch locations; it is not a change to this result. + +## SELECT-004 promoted — HLX-EXP-2026-09-26-SELECT-004 + +Operator authorization applied the selected epoch-3 checkpoint as BEST. The promoted object is `hyperlex-encoder-modernbert-base-seed-select004/model.safetensors`, sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. `model.final.safetensors` (`d5be46be6611e382d837bab8868bb373cbead4f9caa3a666ca06f2b2cda1925b`) was not promoted. Epoch 39 was not promoted. No training ran. The reserve was not rescored. + +Previous BEST `hyperlex-encoder-modernbert-base-seed-morph78` remains in place, sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. + +Decision receipt sha256 `ee493b7d73b5e00ae27ec681bb52405bbcf2983a16a0297edad225aec52b7225`. Decision `PROMOTE`. Selection metric `classify_macro_f1_nonnone` `0.64448782942204`. Primary macro-F1 baseline `0.022395727019119547`, candidate `0.07245710784313726`, delta `0.05006138082401771`. Preservation deltas: classification accuracy `0.01659751037344398`, OBSERVED accuracy `0.048780487804878044`, unbind clean exact `0.008`. CHAR_WINS did not occur. Integrity `PASS`. + +The 491 sealed reserve identities moved `EVAL_RESERVE` to `EVAL_BOUND` to `EVAL_SPENT` through `IdentityLedger.transition` and `persist_append`. Events sha256 before `8223ae11bb42bd1a98ebcd739d1cfbc470085e241826b662703faefdfe752da6`, after `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. Projection sha256 before `d071b7aec8154203ce7f9ae9531639b8d638f86c2ac0c3af38bead9b3c4a48f9`, after `77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0`. The sealed binding file `c16e69559dd6582f687532ce6e2a2e9b52a70db1aa066d11e7e68e723936b371` was not rewritten. Historical consumed, spent, and abandoned identities outside this reserve were not changed. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. This promotion does not establish performance across the eighteen-family ontology. The production head cannot name nine of the twelve sealed gold families. That limitation is separate from the checkpoint pointer. Vendor calls: 0. No further experiment was started. + +Promotion receipt sha256 `2e1f84b476f7355e31380419afad00d0b16d372235a0da6be85adc17fcf92d01`. + +## SELECT-005 blocked — HLX-EXP-2026-09-27-SELECT-005 + +`HLX-EXP-2026-09-27-SELECT-005` is the next unclaimed experiment id. Claimed ids are `HLX-EXP-2026-09-26-SELECT-001` through `HLX-EXP-2026-09-26-SELECT-004`. This id is recorded and not sealed. No preregistration file, arm directory, reserve, or admission receipt was created. + +`TRAINING_BLOCKED`. The proposed variable is the composite `train_schedule`. Control would be `max_epochs=40`, early stopping disabled, restore best. Candidate would be `max_epochs=12`, `minimum_epochs=4` scored epochs through epoch index 3, `early_stopping_patience=4`, strict increase, ties keep the earlier checkpoint, restore best. Both arms would pin `classify_macro_f1_nonnone`, warm start `hyperlex-encoder-modernbert-base-seed-morph65`, and export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430` at 9150 rows. + +The canonical trainer does not implement that candidate. `scripts/shadow/hyperlexical/loop.py` scores `for ep in range(epochs)` and never stops early. `score_now > best_macro` already keeps the earlier checkpoint on ties, and best weights are written back after the loop when the selection metric is `classify_macro_f1_nonnone`. `epoch-progress.jsonl` records the metric and not wall-clock. Spark and public `main` have the same `loop.py` sha256 `f51e7aaff69e9033cc9ba16eee7225bfeefcf521e32236bdb791cc7900e130a5`. Spark HEAD `098ece4d9e8ebb27b0b0d3410b1280ed072d4847` was clean. Public `main` is `2f73f30010cc16ee014ed8d88de73131eb80d0a9`. + +The smallest separate change is an optional break in that epoch loop, default off: after a scored epoch, stop when at least 4 epochs have been scored and `epoch - best_epoch >= 4`, still capped by `max_epochs`, and append per-epoch wall-clock seconds to `epoch-progress.jsonl`. Setting `HYPERLEX_TRAIN_EPOCHS=12` is not that schedule. This record does not apply the change. + +SELECT-004 artifacts were not modified. Its reserve stays `EVAL_SPENT` and was not reused. No new reserve was allocated. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. No optimizer was constructed. Epochs and gradient steps for this id are zero. Launch is not authorized. + +## SELECT-005 still blocked — trainer synced, admission refused + +Spark now carries public `main` `a7d254e8981695072f5be36ab7ebdcc46cbd672c` for the trainer. `scripts/shadow/hyperlexical/loop.py` sha256 `1aa395081d7709be3844bf2568d12100d73d931ecbc51af43c3d6e01dba77e2a`. Early stopping remains default-off. `HLX-EXP-2026-09-27-SELECT-005` is still not sealed. No preregistration file, arm directory, reserve, or admission receipt was created. The private ledger sections above were not replaced by the public projection. + +`TRAINING_BLOCKED`. The trainer can express the candidate schedule. Canonical admission cannot seal it. + +`admit_training_run` allows exactly one scientific difference, and that difference must be `HLX_SELECT_METRIC`. A non-mutating probe pinned both arms to `classify_macro_f1_nonnone` and changed only `HYPERLEX_TRAIN_EPOCHS` (`40` versus `12`) and `HYPERLEX_EARLY_STOP` (`0` versus `1`). Patience and minimum epochs were the same on both arms. The gate `single_variable` failed: `scientific variable count is 2: HYPERLEX_EARLY_STOP,HYPERLEX_TRAIN_EPOCHS`. + +The spent ledger cannot supply a new `EVAL_RESERVE`. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. Live reserve counts are classify 0, classify_observed 0, classify_non_none 0, unbind_clean 0. The same probe failed at `holdout_reserve`: `reserve slice classify is absent`. The reserve scan also includes every identity with `evaluation_reserved` set. That scan is 491 identities, all derived `EVAL_SPENT`. A new reserve written onto this ledger would still fail that lifecycle check. The flag was not cleared. The SELECT-004 reserve was not reused. + +No optimizer was constructed. Epochs and gradient steps for this id are zero. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Launch is not authorized. + +The smallest separate change is an admission-contract patch: accept one declared schedule variable while both arms share `classify_macro_f1_nonnone`, and treat historical `EVAL_SPENT` identities as outside the current reserve without clearing `evaluation_reserved`. Do not allocate a reserve before that contract exists. + + +## SELECT-005 still unsealed — admission contract synced, fresh reserve unavailable + +Public `main` is `fab0de03d75e3280dc34feed425c4783ccf7e2b5`, the squash merge of the admission contract. Spark carries that `admission.py` sha256 `6ae7b63ca1b6e0459d9b795c731668b3eff524269d81bcb92181c0c9d407ab78` and `identity_ledger.py` sha256 `ac49b7240a895952b99b3ef9046f7a8185c8647badac7a45c8640d592057d1c5`. `scripts/shadow/hyperlexical/loop.py` remains sha256 `1aa395081d7709be3844bf2568d12100d73d931ecbc51af43c3d6e01dba77e2a`. The private ledger file was not replaced. + +`HLX-EXP-2026-09-27-SELECT-005` is still not sealed. No preregistration file, arm directory, reserve binding, or admission receipt was created. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `evaluation_reserved` was not cleared. The SELECT-004 reserve stays `EVAL_SPENT` and was not reused. + +A fresh reserve needs all four slices from text that is absent from the ledger and from the pinned export. The ledger has no `EVAL_RESERVE`, `EVAL_BOUND`, or `AVAILABLE` identities. The remaining settled hashes from the held-out stream that are absent from the ledger are `UNRESOLVED`, `NONE`, `RECLASSIFY`, or `ACCEPT` with `RIGHTS_UNRESOLVED`. Unresolved-rights rows stay out of `EVAL_RESERVE`. No novel rights-cleared classify settlement remains. A classify-absent reserve was not written. The WordNet unbind source was not admitted by itself. + +No optimizer was constructed. Epochs and gradient steps for this id are zero. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Launch is not authorized. + + +## SELECT-005 census — fresh reserve still unavailable + +A second join of the 339 settlement events to the 7964 ledger identities found no novel rights-cleared classify row. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The ledger was not appended. `evaluation_reserved` was not cleared. The SELECT-004 reserve was not reused. + +All 6761 unique hashes in the pinned export are already in the ledger. Of the 98 settlement events absent from the ledger, none are in that export. Those 98 are 82 `UNRESOLVED` with cleared rights, 2 `UNRESOLVED` with unresolved rights, 7 `ACCEPT` with unresolved rights, 6 `NONE` with unresolved rights, and 1 `RECLASSIFY` with unresolved rights. `UNRESOLVED` means the operator reviewed the row and declined to settle it. Unresolved-rights rows stay out of `EVAL_RESERVE`. There are 0 novel `CC-BY-SA` rows with decision `ACCEPT`, `RECLASSIFY`, or `NONE`. + +The 28 promoted-accept files are training gold under an older family set and were not remapped. WordNet can still supply `unbind_clean` and was not admitted alone, because a classify-absent reserve fails the slice gate. No preregistration, arm directory, reserve binding, or admission receipt was created. The optimizer was not constructed. Epochs and gradient steps for `HLX-EXP-2026-09-27-SELECT-005` remain 0. `training_launch_authorized` is false. BEST was not moved. + + +## SELECT-005 classify candidates blocked — HLX-EVAL-REVIEW-2026-09-27-001 + +Admission requires each active slice count to be at least 1. The classify floors are classify 1, classify_observed 1, and classify_non_none 1. Planning targets 606, 287, and 604 are not that floor. This pass did not lower the floor. + +Packet `HLX-EVAL-REVIEW-2026-09-27-001` is an operator-review packet, not a reserve. Ready rows: 0. Other screened rows: label unresolved 10, rights blocked 69, provenance blocked 4, cohort duplicate 1. Previously declined rights-cleared rows and unresolved-rights events were not reopened. WordNet and the older promoted-accept files were not used. No operator decision was written. + +Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The ledger was not appended. `HLX-EXP-2026-09-27-SELECT-005` was not sealed. BEST was not moved. + + +## ai-native evaluation family — 2026-09-27 + +Public `main` is `ca9403efe8470d46566abbcd098640f07b33b759`. `ai-native` is taxonomy-active. `evaluation.enabled` stays false. The production head stays nine names. No row was settled. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The review packet `HLX-EVAL-REVIEW-2026-09-27-001` now has 10 ready rows proposing `ai-native`. Their stored class stays `INFERRED`. `classify_observed` is still short by 1 until an operator attests `OBSERVED`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. + + +## Operator attested the ready rows observed + +The operator attested `OBSERVED` on all 10 ready rows in packet `HLX-EVAL-REVIEW-2026-09-27-001`. The proposed family is `ai-native`. Rights-blocked, provenance-blocked, and duplicate rows were not attested. No `ACCEPT`, `RECLASSIFY`, or `NONE` was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. + +## Unbind preview refused as slang — 2026-09-27 + +The operator reviewed the first 20 positional surfaces from the novel WordNet unbind pool. None are admitted as slang. Eighteen are refused. `give a damn` and `in one's birthday suit` are quarantined for provenance review. Slang, idiom, colloquialism, and profanity stay distinct. No `ACCEPT`, `REJECT`, `CORRECT_TARGET`, or `UNRESOLVED` was written to the unbind settlement log. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. These rows are not classify gold and must not be trained as slang. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Disposition packet `HLX-EVAL-UNBIND-PREVIEW-2026-09-27-001`. + +## Unbind preview roles — 2026-09-27 + +Slang classification and structure unbinding stay orthogonal. On the same 20-surface preview, the operator marked 7 high-value unbind candidates, 8 secondary candidates, and 5 rejects. The rejects are `beta vulgaris`, `gulf of oman`, `u. s. air force`, `department of the federal government`, and `court of assize and nisi prius`. The slang disposition is unchanged: none admitted, 18 refused, and `give a damn` plus `in one's birthday suit` quarantined as slang. No unbind settlement decision was written. The 15 candidates are not admitted. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-PREVIEW-2026-09-27-001`. + +## Unbind screen specified — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v1` is a candidate-selection rule, not an admission filter. Specification agreement with the 20-item preview is 20/20. That figure is specification fit, not held-out precision. A held-out validation sample of 32 admissible positional surfaces, excluding those 20, is recorded with proposed buckets and `gold` null. Provisional screen counts on 63882 admissible positional surfaces are high-value 1965, secondary 60559, reject 1358. Those counts are not a draw. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-001`. + +## Unbind screen v2 — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v2` replaces the v1 proposal as the screening hypothesis. It is not authorized as the SELECT-005 screen. Surface patterns remain candidate-generation heuristics. Hard exclusions are proper name, titled entity, taxonomy, productive number, and unconstrained free composition. Agreement after the revision is 20/20 on the first preview and 32/32 on the reviewed sample. That agreement is fit, not held-out precision. A new validation sample excludes all 52 reviewed surfaces and carries `gold` null. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-002`. + +## Unbind screen v3 — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v3` is a hypothesis. It is not authorized as the SELECT-005 screen. Candidate-generation patterns do not assign high-value. The automatic screen emits reject or unresolved only. v2 pool counts stay frozen at high-value 1977, secondary 55427, reject 6478, over 63882 positional surfaces, and were not recomputed. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-003`. + +## Unbind screen v3 held out — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v3` stays a proposed refinement. Fifty-two surfaces are frozen as development data and thirty-two as validation-development data. Agreement on those eighty-four is fit, not held-out precision. A fresh sample of 29 admissible positional surfaces excludes all 84. Predictions were not hand-corrected. Held-out precision is not computed. v2 pool counts stay frozen at high-value 1977, secondary 55427, reject 6478, over 63882 positional surfaces. The v3 screen was not run on that pool. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-004`. + +## Unbind held-out frozen — 2026-09-27 + +The 29-row v3 application is frozen. Sample sha256 `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. It was produced by one application of `RUNE.UNBIND_SCREEN.v3` and was not hand-corrected. Operator labels are pending. Held-out precision is `NOT_COMPUTABLE`. v3 was not revised. Development data remain 52 rows. Validation-development data remain 32 rows. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not authorized and is not sealed. + +## Unbind screen evaluation lane — 2026-09-27 + +`HLX-EVAL-UNBIND-SCREEN-V3-001` evaluates the frozen 29-row v3 application. Source sample sha256 remains `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. Prediction, operator judgment, gold, admission, and settlement are separate artifacts. The blind review does not carry the predicted bucket. Operator labels are pending. Held-out precision and the confusion matrix are `NOT_COMPUTABLE`. A scored report does not authorize `HLX-EXP-2026-09-27-SELECT-005`. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen held-out scored — 2026-09-27 + +Operator labels for HLX-EVAL-UNBIND-SCREEN-V3-001 are frozen. Label sha256 `4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5`. The source sample sha256 remains `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. Resolved accuracy is 16/29. High precision is 1 and recall is 6/13. Secondary precision is 4/17 and recall is 1. Reject precision is 1 and recall is 6/12. Quarantine support is 0. Every error is false secondary: 7 operator-high and 6 operator-reject. False high and false reject are 0. revision_eligible stays false. SELECT-005 is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 hypothesis — 2026-09-28 + +The v3 held-out score stays 16/29. All 13 errors are false secondary: 7 operator-high and 6 operator-reject. False high and false reject are 0. Those 29 rows are now v4 development evidence, not a validation set. `RUNE.UNBIND_SCREEN.v4` is drafted and not encoded, applied, or authorized. It would only add coverage around the secondary basin: normalized productive numbers, multi-token personal names, organization glosses, species common names, and medical technical phrases on one side; nonliteral and conventionalized gloss evidence on the other. Existing high and reject decisions stay in place. `revision_eligible` stays false. SELECT-005 is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v4.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. v4 may only move additional secondary fall-throughs, and only by semantic evidence classes. It must not reinterpret v3 high or reject logic, redefine secondary, or train on a future validation sample. Phrase-specific exceptions are prohibited. The 113 reviewed surfaces are the later regression set. The next measurement sample must exclude them and be frozen before inspection. `revision_eligible` stays false. SELECT-005 is not authorized. The v3 sample and label hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v4` is encoded as a wrapper over a frozen v3 bucket. It inspects a row only when that bucket is secondary. Patch A may move secondary to reject. Patch B may move secondary to high. Existing high and reject decisions are not reopened. The acceptance contract is unchanged, sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. + +The 113 reviewed surfaces were replayed as a regression suite. Gate A through Gate E passed. High rows unchanged: 39. Reject rows unchanged: 28. Secondary moves: 13 to high and 6 to reject. Each move has one Patch A or Patch B evidence code. No phrase-specific rule fired. Operator conflict on those moves: 0. This replay is not a new precision estimate. + +Success criteria were frozen before the measurement draw, sha256 `4fcbebfad7797eb22393fa94a389f1ef41402612e359b63d3a3a4b031c6cf8be`. High and reject precision floors are the v3 held-out floors of 1. The false-secondary rate must be strictly below 13/29. False high and false reject are not allowed. Perfect accuracy is not required. + +The measurement sample excludes all 113 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. v4 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The v3 sample sha256 `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e` and label sha256 `4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5` are unchanged. + +## Unbind screen v4 measurement scored — 2026-09-28 + +Operator labels for the 28-row v4 measurement sample are frozen. Label sha256 `023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463`. The sample sha256 remains `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. The prediction sha256 remains `a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f`. Labels were recorded at `2026-09-28T02:17:00Z`, after the sample freeze at `2026-09-28T01:07:34Z`. Hand corrections are 0. v4 was not applied again. + +Resolved accuracy is 16/28. High precision is 1 (7/7) and recall is 7/13. Reject precision is 1 (5/5) and recall is 5/11. Secondary precision is 4/16 and recall is 4/4. Quarantine support is 0. False high is 0. False reject is 0. False secondary is 12/28, which is below the frozen floor of 13/29. Every error is a secondary fall-through. The pre-registered success criteria pass, so `revision_eligible` is true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 hypothesis — 2026-09-28 + +The v4 measurement artifact and score receipt stay frozen. Sample sha256 `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. Prediction sha256 `a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f`. Label sha256 `023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463`. Acceptance sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. Resolved accuracy remains 16/28. High precision remains 1. Reject precision remains 1. False secondary remains 12/28. `revision_eligible` on that measurement remains true. + +Those 28 rows are now v5 development evidence, not a validation set. Evidence sha256 `dcab832038c3209a54a4159b23caf4eecfb94864cef7188af28f1ae3b0ff80c0`. The twelve secondary fall-throughs are two escape routes only: referential or terminological rows that stayed secondary, and lexicalized noncompositional rows that stayed secondary. They are not a fit list. + +`RUNE.UNBIND_SCREEN.v5` is drafted and not encoded, applied, or authorized. It is a wrapper over a frozen v4 bucket. It inspects a row only when that bucket is secondary. One referential/terminological dominance test may move secondary to reject. One lexicalized noncompositionality test may move secondary to high, and only when the surface is conventionalized and the gloss is not compositionally recoverable. A lexicalized and mostly compositional surface stays secondary. Existing high and reject decisions stay in place. A separate rule for each miss is prohibited. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v5.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. The draft hypothesis sha256 is `c3d3f8081e18582158eab7a1bf154a02aed38614b9801df6a83f7eaf3875da7b`. v5 may only move additional secondary fall-throughs, and only through the two general tests named in the contract. It must not reinterpret v3 or v4 high or reject logic, redefine secondary, redesign the classifier, or train on a future validation sample. Phrase-specific exceptions are prohibited. One executable rule per observed miss is prohibited. + +The later regression replays all 141 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, and 28 from the v4 measurement. The 113-row v4 replay and the 28-row measurement sample are disjoint. Every previously correct high stays high. Every previously correct reject stays reject. Every new move originates from secondary. The next measurement sample must exclude all 141 and be frozen before inspection. v5 is applied once. Operator labels are collected independently. + +The pre-registered bar is high precision 1, reject precision 1, and a false-secondary rate strictly below 12/28. No accuracy-gain target is registered. Perfect accuracy is not required. `revision_eligible` on the v4 measurement stays true and does not authorize `HLX-EXP-2026-09-27-SELECT-005`. The v4 sample, prediction, label, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v5` is encoded as a wrapper over a frozen v4 bucket. It inspects a row only when that bucket is secondary. Referential or terminological dominance may move secondary to reject. A conventionalized surface whose gloss is not recoverable from the ordinary first senses of its constituents may move secondary to high. Existing high and reject decisions are not reopened. The acceptance contract is unchanged, sha256 `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. + +The 141 reviewed surfaces were replayed as a regression suite. High rows unchanged: 59. Reject rows unchanged: 39. Secondary moves: 4 to high and 7 to reject. Each move has the one evidence code for its patch. No phrase-specific rule fired. Operator conflict on those moves: 0. This replay is not a new precision estimate. Gate report sha256 `c70214ac1ea3f24b0b8c5590b015a64de4f5a7a632cf372f02aa9e88bd5b2977`. + +The pre-registered bar is unchanged: high precision 1, reject precision 1, and a false-secondary rate strictly below 12/28. Criteria sha256 `49e179db537a738bb7370404167c7ec1c7019fcd251d77e6a916f4426a0ecccb`. No accuracy-gain target was added. + +The measurement sample excludes all 141 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. Prediction sha256 `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. v5 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. `revision_eligible` on this screen stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 measurement scored — 2026-09-28 + +Operator labels for the 28-row v5 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f`. The row table is high 11, secondary 8, reject 9, quarantine 0, unresolved 0. The seal line said high 10, secondary 8, reject 10. The frozen artifact follows the row table. + +The sample sha256 remains `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. The prediction sha256 remains `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. Labels were recorded at `2026-09-28T03:03:27Z`, after the sample freeze at `2026-09-28T02:44:48Z`. Hand corrections are 0. v5 was not applied again. The 141-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 17/28. High precision is 6/7 and recall is 6/11. Reject precision is 5/6 and recall is 5/9. Secondary precision is 6/15 and recall is 6/8. Quarantine support is 0. False high is 1. False reject is 1. False secondary is 9/28: 5 operator-high and 4 operator-reject. Accuracy was recorded and was not part of the bar. + +False high is 1 and false reject is 1, so the outer-bucket bar fails. The false high is `keep out`, which v4 had already marked high and v5 passed through. The false reject is `on the job`, which v5 moved from secondary to reject on referential/terminological dominance. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. + +## Unbind screen v6 hypothesis — 2026-09-28 + +The v5 measurement stays scored as an outer-bucket failure. Sample sha256 `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. Prediction sha256 `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. Label sha256 `b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f`. False high is 1. False reject is 1. High precision is 6/7. Reject precision is 5/6. False secondary is 9/28. Resolved accuracy is 17/28 and is descriptive only. `revision_eligible` on v5 stays false. + +The error analysis is frozen and authorizes no fix. Analysis sha256 `ecb229dccec627af958d12537ab153d35de9e3e58d71bb8d3f0a8835077b26ac`. One inherited high overreach: `keep out` was already high at v4, and v5 passed it through against an operator secondary label. One new reject overreach: `on the job` moved from secondary to reject under referential/terminological dominance against an operator secondary label. Secondary undercoverage remains 5 operator-high rows and 4 operator-reject rows. These 28 rows are v6 development evidence, not a validation set. Evidence sha256 `daf512dd331216fecb87ec4d4890c83a03afaf78b5fc1ca6dae6e46e1a9f63a9`. + +`RUNE.UNBIND_SCREEN.v6` is drafted and not encoded, applied, or authorized. Draft sha256 `942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5`. It is a structural revision, not another secondary-only wrapper. A frozen v5 bucket is provisional. High may fall to secondary only with compositional recoverability. Reject may fall to secondary only with nonreferential lexical use. A demotion stops at secondary for that application. A provisional secondary row may still move by the two existing evidences: lexicalized noncompositional to high, and referential/terminological dominance to reject. High and reject do not swap directly. The motivating surfaces are not a required fit. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v6.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. The draft hypothesis sha256 is `942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5`. The error analysis sha256 is `ecb229dccec627af958d12537ab153d35de9e3e58d71bb8d3f0a8835077b26ac`. The development evidence sha256 is `daf512dd331216fecb87ec4d4890c83a03afaf78b5fc1ca6dae6e46e1a9f63a9`. + +The replay set is all 169 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, 28 from the v4 measurement, and 28 from the v5 measurement. The v5 measurement is disjoint from the prior 141. Previously operator-correct rows must remain operator-correct. High falls to secondary only with compositional recoverability. Reject falls to secondary only with nonreferential lexical use. Secondary rises to high only with lexicalized noncompositional evidence. Secondary falls to reject only with referential/terminological dominance. Direct movement between high and reject is prohibited. Phrase-specific rules are prohibited. + +The pre-registered measurement bar is high precision 1 and reject precision 1, with false high and false reject at 0. No false-secondary quota and no accuracy-gain target are registered. The next measurement sample must exclude all 169 reviewed surfaces and be frozen before inspection. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. The v5 sample, prediction, label, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v6` is encoded over a frozen v5 bucket. High may fall to secondary only when the gloss is recoverable from the ordinary first senses of at least two constituents and the synset has no unrelated single-word mapping. Reject may fall to secondary only when a pertainym is used as a state or relation whose gloss meets those ordinary senses, and not when the gloss is a relational designation. A demotion stops at secondary for that application. A provisional secondary row may still move by lexicalized noncompositional evidence or referential/terminological dominance. High and reject do not swap. The acceptance contract is unchanged, sha256 `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. + +The 169 reviewed surfaces were replayed. Previously correct rows lost: 0. Direct swaps between high and reject: 0. High to secondary: 0. Reject to secondary: 1. That row is `on the job`, evidence `nonreferential_lexical_use`, and the operator label is secondary. Secondary to high: 0. Secondary to reject: 0. No phrase-specific rule fired. `keep out` stays high. Its gloss is not covered by the ordinary first senses of its constituents, and the synset maps it to an unrelated word, so compositional recoverability does not fire. That miss is not a required fit. Unchanged disagreements: 16. This replay is not a precision estimate. Gate report sha256 `66e4f3859738dfaa63a21ec2e69025b568ea097c17f509ceed4277022d979ffb`. Prediction sha256 `7398254bd62253129153e7b57c99122ed7cc7157520788cc60bb0e80f0b356e8`. Diff sha256 `3af430bb9261c1cedc1bd6181f5e928e01eef22bccea0b81c3c435073fabff59`. + +The pre-registered bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor and no accuracy target are registered. Criteria sha256 `3a511ffb49f32930ead14ca2f1e2f6639122c47a7ceec9d1bc29815dd7b490dd`. + +The measurement sample excludes all 169 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. Prediction sha256 `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. v6 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. `revision_eligible` on this screen stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 measurement scored — 2026-09-28 + +Operator labels for the 28-row v6 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11`. The row table is high 12, secondary 4, reject 12, quarantine 0, unresolved 0. The seal line said high 11, secondary 4, reject 13. The frozen artifact follows the row table. + +The sample sha256 remains `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. The prediction sha256 remains `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. Labels were recorded at `2026-09-28T03:34:48Z`, after the sample freeze at `2026-09-28T03:25:47Z`. Hand corrections are 0. v6 was not applied again. The 169-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 21/28 and is descriptive only. High precision is 9/10 and recall is 9/12. Reject precision is 1 (9/9) and recall is 9/12. Secondary precision is 3/9 and recall is 3/4. Quarantine support is 0. False high is 1. False reject is 0. False secondary is 6/28: 3 operator-high and 3 operator-reject. Recall and accuracy were not part of the bar. + +A false high or false reject is present, so the outer-bucket bar fails. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. + +## Unbind screen v7 hypothesis — 2026-09-28 + +The v6 measurement stays scored as an outer-bucket failure. Sample sha256 `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. Prediction sha256 `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. Label sha256 `feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11`. High precision is 9/10. Reject precision is 1 (9/9). False high is 1. False reject is 0. False secondary is 6/28. Resolved accuracy is 21/28 and is descriptive only. `revision_eligible` on v6 stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. + +The only false high in that measurement is `to a lesser extent`. The operator bucket is secondary. v3, v4, v5, and v6 are high, and v6 recorded no evidence code. A prior inherited high disagreement, `keep out`, stays high in the 169-row replay. Its operator bucket is secondary and `compositional_recoverability` did not fire. That v6 predicate is not redefined. Secondary fall-throughs are 13/29 at v3, 12/28 at v4, 9/28 at v5, and 6/28 at v6. Coverage is not the next question. + +The error analysis is frozen and authorizes no fix. Analysis sha256 `06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766`. These 28 rows are v7 development evidence, not a validation set. Evidence sha256 `3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3`. The six false-secondary rows stay unresolved and are not a fit list: `to the letter`, `with child`, `dressed to the nines`, `union jack`, `atomic number 98`, and `law of definite proportions`. Known reviewed inventory is 197: the prior 169 plus these 28. Normalized overlap is 0. + +`RUNE.UNBIND_SCREEN.v7` is specified and not encoded, applied, or authorized. State is `SPEC_FROZEN`. Draft sha256 `c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b`. It keeps the frozen v6 phase order and replaces only the high challenge. A provisional high may fall to secondary only with `ordinary_compositional_derivation`: the WordNet sense can be derived by ordinary grammatical, syntactic, comparative, or phrasal composition without a stored conventionalized or nonliteral lexical binding. The test does not require recovery from only the ordinary first sense of each token. It asks whether lexicalized binding is required at all. A demotion stops at secondary for that application. Reject may still fall to secondary only with `nonreferential_lexical_use`. A provisional secondary row may still move by the unchanged v6 evidences. High and reject do not swap. `to a lesser extent` and `keep out` motivate the question and are not required fits, and neither is special-cased. No SELECT-005 admission surface exists. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7.ACCEPTANCE` is frozen and the screen is not encoded. State is `SPEC_FROZEN`. Acceptance sha256 `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. The draft hypothesis sha256 is `c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b`. The error analysis sha256 is `06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766`. The development evidence sha256 is `3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3`. + +The replay set is all 197 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, 28 from the v4 measurement, 28 from the v5 measurement, and 28 from the v6 measurement. The v6 measurement is disjoint from the prior 169. Previously operator-correct rows must remain operator-correct. A previously correct high stays high. High falls to secondary only with ordinary compositional derivation. The frozen v6 transitions for secondary to high, secondary to reject, and reject to secondary are unchanged. Direct movement between high and reject is prohibited. A high demotion stops at secondary and is not re-promoted in the same application. Phrase-specific rules, row-id rules, and surface-hash rules are prohibited. + +No measurement sample was drawn. The bar proposed for a later unseen sample is high precision 1 and reject precision 1, with false high and false reject at 0. No false-secondary quota, no accuracy target, and no recall target are registered. That sample must exclude all 197 reviewed normalized identities and be frozen before inspection. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. No admission surface for that experiment exists. The v6 sample, prediction, label, criteria, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7` is encoded over a frozen v6 bucket. The only new predicate is `ordinary_compositional_derivation`. It inspects a provisional high. The recorded sense must be a comparative, syntactic, or phrasal composition, and the synset must not store an unrelated single-word synonym. A gloss that merely shares constituent stems does not fire, and a metaphorical retelling does not fire. A demotion stops at secondary. Reject rows still use the v6 reject challenge. Secondary rows are not reopened, so a demotion v6 already stopped stays stopped. `compositional_recoverability` is not a v7 transition. The acceptance contract is unchanged, sha256 `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. + +The 197 reviewed surfaces were replayed as development and regression evidence, not as a generalization estimate. Previously correct rows lost: 0. Previously correct high rows demoted: 0. Direct swaps between high and reject: 0. High to secondary: 1. Reject to secondary: 0. Secondary to high: 0. Secondary to reject: 0. One row moves. `to a lesser extent` goes from high to secondary with `ordinary_compositional_derivation`, and the operator bucket is secondary. That firing is not a required fit. `keep out` stays high. No phrase-specific rule fired. This replay is not a precision estimate. Gate report sha256 `78225b68c639694bb4c17342c87320ccb44faac557a9145ea117603da0d545ed`. Prediction sha256 `24f266e16b80da602b011bf7cca13774ce9f76c0c52e81e600a1d9743d61d197`. Diff sha256 `b86ae6611a5fcd233b6e92b3def4cdd0680018902532a0a0d5fb02659ad092ae`. + +State is `REGRESSION_VERIFIED`. `measurement_eligible` stays false. The unseen sample was not drawn. The pre-registered bar remains high precision 1, reject precision 1, false high 0, and false reject 0, with no false-secondary quota, no accuracy target, and no recall target. No measurement sample was drawn. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 measurement frozen — 2026-09-28 + +The 197-row regression stays `REGRESSION_VERIFIED`. One unseen measurement sample excludes those 197 normalized identities. Normalized overlap with the reviewed set is 0. Duplicate normalized identities inside the sample are 0. The draw is deterministic and stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. No occupied cell was exhausted. Each occupied cell contributed 2 rows. + +The sample was frozen before v7 was applied. Sample sha256 `2b4414e2a6db3acf2967603e3ef4552c631803285633fbae5f3f3d50451724b1`. v7 was then applied once. Hand corrections are 0. Prediction sha256 `d201affe0c87206227bc211a6071a25e639c0ffb7cbcfdc8dac0572155a4da47`. The blind rows carry surface, part of speech, token count, and gloss. They do not carry a bucket or an evidence code. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. + +The acceptance bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor, no accuracy target, and no recall target were added. Criteria sha256 `3b22449874f2384d843e4f16cb7e5c28b6acfc83273aec44542e0bc5b0c3ef37`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 measurement scored — 2026-09-28 + +Operator labels for the 28-row v7 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `0ce39dffad49b28b5adebfbe3514fcf81f9ca39689d3eadf0695975496b6adb5`. The row table is high 14, secondary 5, reject 9, quarantine 0, unresolved 0. The seal line said high 13, secondary 5, reject 10. The frozen artifact follows the row table. + +The sample sha256 remains `2b4414e2a6db3acf2967603e3ef4552c631803285633fbae5f3f3d50451724b1`. The prediction sha256 remains `d201affe0c87206227bc211a6071a25e639c0ffb7cbcfdc8dac0572155a4da47`. Labels were recorded at `2026-09-28T04:14:27Z`, after the sample freeze at `2026-09-28T04:05:23Z`. Hand corrections are 0. v7 was not applied again. The 197-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 21/28 and is descriptive only. High precision is 10/11 and recall is 10/14. Reject precision is 7/8 and recall is 7/9. Secondary precision is 4/9 and recall is 4/5. Quarantine support is 0. False high is 1. False reject is 1. False secondary is 5/28: 3 operator-high and 2 operator-reject. Recall, accuracy, and false secondary were not part of the bar. + +A false high or false reject is present, so the outer-bucket bar fails. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on v6 stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. + +## Unbind screen lineage retired — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v3` through `RUNE.UNBIND_SCREEN.v7` stay sealed. They are a completed experimental lineage. The lineage showed that further surface-pattern patches are the wrong instrument. No `RUNE.UNBIND_SCREEN.v8` exists. Historical scores are not results for the sense-first screen. v7 remains `SCORED`, outcome `OUTER_BUCKET_FAILURE`, `revision_eligible` false. v7 was not retuned. v6 and v5 `revision_eligible` stay false. The v4 measurement `revision_eligible` stays true. Source sha256 values remain v3 `179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6`, v4 `f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377`, v5 `70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948`, v6 `59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba`, v7 `73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab`. Retirement sha256 `fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f`. + +The v7 measurement disagreements are not one failure class. The frozen error analysis separates three questions: whether the bound sense designates a referent, whether that sense requires conventionalized lexical binding, and whether that sense is compositionally recoverable. Error analysis sha256 `ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8`. The canonical v7 operator counts from the row artifact are high 14, secondary 5, reject 9. The handwritten seal does not override those rows. + +## Unbind sense screen v1 specified — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` is `SPEC_FROZEN`. It is a new screen family. It is not encoded, not applied, and no measurement sample was drawn. The unit of analysis is the specific lexical sense: surface, part of speech, WordNet synset, and the frozen first-sense gloss. Another sense of the same surface, a historical v3-v7 bucket, and a referential origin that is not the predicated sense are outside that unit. + +The five classes are `REFERENTIAL`, `LEXICALIZED_NONCOMPOSITIONAL`, `LEXICALIZED_COMPOSITIONAL`, `ORDINARY_COMPOSITIONAL`, and `AMBIGUOUS`. The procedure assigns exactly one class. Empty or unmatched sense evidence is `AMBIGUOUS`. A figurative gloss is not referential merely because the wording mentions a place or story; `road to damascus` is the protected illustration and is not a required fit. Ordinary syntax, comparison, degree, and phrasal composition are `ORDINARY_COMPOSITIONAL`; a WordNet lemma by itself is not lexicalization. `as far as possible` motivates that distinction and is not a required fit. Insufficient evidence stays `AMBIGUOUS`. + +The frozen mapping is referential to reject, lexicalized noncompositional to high, both compositional classes to secondary, and ambiguous to quarantine. Ordinary composition does not map to reject. That would hide a second classifier. The two compositional classes remain distinct before the mapping. No historical bucket may override the sense class. + +The reviewed positional inventory is 225: the prior 197 plus the 28 v7 measurement rows. Normalized overlap between those sets is 0. The 28 rows are development evidence, not validation, and not a required fit. 225 rows bind a synset offset whose first-sense gloss equals the frozen gloss. 0 rows keep the frozen gloss with no offset. A future measurement must exclude all 225 normalized identities and be frozen before inspection. The proposed pass/fail bar is high precision 1, reject precision 1, false high 0, and false reject 0. Sense-class confusion, bucket confusion, per-class support, the ambiguous rate, high recall, reject recall, and secondary precision and recall are descriptive. This pass does not make them pass/fail criteria. + +Draft hypothesis sha256 `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. Acceptance sha256 `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. Development evidence sha256 `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. Tracker sha256 `4145848a2e1368b23f28f645249a9a8c55bf113a43763d31a434f1389b9e1154`. No JSON Schema exists for this hypothesis family. The older unbind sample, label, and report schemas are a different contract and were not applied. The next named transition is `ENCODED`. This pass does not authorize it. + +## Lexeme structure screen v1 architecture — 2026-09-28 + +`RUNE.LEXEME_STRUCTURE_SCREEN.v1` is an architecture note for a sibling lane. It evaluates one orthographic lexeme for internal structure. Candidate classes are atomic, compound, affixed, blend, clipping, respelling, reduplicated, borrowed, and unknown. A single token is not evidence of an atomic lexeme. A string split is not evidence of a valid morphological decomposition. Conceptual illustrations are not a corpus and not a required fit. This lane is not mixed into positional unbind. It is not encoded. No corpus was populated, no sample was drawn, and no gold was created. Architecture sha256 `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. + +`HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind sense screen v1 procedure frozen — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` moves from `SPEC_FROZEN` to `PROCEDURE_FROZEN`. The screen is not encoded and not applied. No development replay was run. No row received a sense class. No measurement sample was drawn. + +The companion artifact is `hyperlex.unbind_sense_screen_v1_classification_procedure.v1`, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The sealed acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The sealed draft stays `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. The acceptance does not embed this procedure, so its bytes were not revised. The tracker now points at the procedure. Tracker sha256 `f9c4757b6eec558b5e1baf644bcf33c27c949807e7f00cd15df869eb6411de31`. + +WordNet membership is not evidence that a surface is a conventional lexical unit. A conventional unit requires an unrelated single-word co-lemma or a lexical pointer on the whole lemma. A productive frame requires a one-token lemma alternation stored on the synset, or a gloss that starts with the comparative or superlative operator formula. Anything else at that test is `AMBIGUOUS`. A stored unrelated equivalent is the noncompositional mapping. A derivation or pertainym from the whole lemma to one of its constituents, with no unrelated equivalent, is the compositional lexical unit. Referential designation requires an instance-hypernym pointer or a parenthetical four-digit lifespan. A geographic or religious allusion does not qualify, and neither does the capital letter in `road_to_Damascus`. Ordinary composition still maps only to secondary. Ambiguity maps to quarantine and is a normal class. + +The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. Reviewed positional inventory remains 225. v3 through v7 sources are unchanged. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `ENCODE_AUTHORIZATION`. This pass does not authorize it. + +## Unbind sense screen v1 development replay — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` is encoded and the 225-row development replay is analyzed. The frozen classification procedure is unchanged, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The sealed acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The sealed draft stays `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. The development manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. No sense class was written onto that manifest. + +The encoder follows the frozen tests. An instance-hypernym pointer or a parenthetical four-digit lifespan is referential. An unrelated single-word co-lemma is lexicalized noncompositional. A derivation or pertainym from the whole lemma back to a constituent, with no unrelated equivalent, is lexicalized compositional. A stored one-token lemma alternation, or a gloss that starts with the comparative or superlative operator formula, is ordinary compositional. Anything else is ambiguous. A yes-signal together with a no-signal is a conflict and stays ambiguous. Absence of a signal is not secondary. No gloss-keyword list, technical-term dictionary, or phrase exception was added. A high ambiguous rate was not repaired. + +Each development row was classified once from its frozen synset. The replay is development evidence, not validation. The measurement bar was not applied. Prediction sha256 `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef`. Report sha256 `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. Tracker sha256 `b3690de5c957f019224c7ea980bf6e27b25ba1f1c519347f72a327fa3ea399a2`. + +Sense classes are referential 26, lexicalized noncompositional 72, lexicalized compositional 0, ordinary compositional 25, and ambiguous 102. The ambiguous rate is 102/225. Buckets are high 72, secondary 25, reject 26, and quarantine 102. Evidence codes are referential designation 26, noncompositional semantic mapping 72, compositional lexical unit 0, productive grammatical frame 25, and insufficient record evidence 102. The determinate sources are instance hypernym 26, unrelated single-word co-lemma 72, and productive alternation 25. No lifespan marker and no grammatical-operator gloss fired on this inventory. + +Lexicalized compositional support is 0. One development row stores a pertainym from the whole lemma to a constituent token and also stores a one-token lemma alternation. The frozen conflict rule returns ambiguous. That zero records the frozen test. It is not a missing dictionary. + +Operator labels already on the manifest, joined after classification: of 101 operator-high rows, 32 stay high, 17 go to secondary, and 52 go to quarantine. Of 46 operator-secondary rows, 22 go to high, 3 stay secondary, and 21 go to quarantine. Of 77 operator-reject rows, 18 go to high, 4 go to secondary, 26 stay reject, and 29 go to quarantine. The one operator-quarantine row goes to secondary. Descriptive precision and recall, not pass/fail criteria: high 32/72 and 32/101, secondary 3/25 and 3/46, reject 26/26 and 26/77. Operator quarantine support is 1. False high is 40. False reject is 0. False secondary is 22 and is descriptive only. + +`road to damascus` is ambiguous and quarantined. The bound record has no instance pointer and no lifespan marker. `as far as possible` is ordinary compositional and secondary, from the stored far/much alternation. Neither row is a required fit. + +State is `DEVELOPMENT_ANALYZED`. The path was `PROCEDURE_FROZEN`, `ENCODE_AUTHORIZED`, `ENCODED`, `225_ROW_DEVELOPMENT_REPLAY`, `DEVELOPMENT_ANALYZED`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. v3 through v7 sources are unchanged. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `MEASUREMENT_AUTHORIZATION`. This pass does not authorize it. + +## Unbind sense screen procedure v2 frozen — 2026-09-28 + +The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v1 stays frozen, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The companion procedure `hyperlex.unbind_sense_screen_v1_classification_procedure.v2` is `PROCEDURE_V2_FROZEN`. It is not encoded and not applied. No v2 development replay was run. No row received a v2 sense class. No measurement sample was drawn. + +The v1 development replay mapped an unrelated single-word co-lemma to lexicalized noncompositional, and therefore to high. All 40 development false highs used that path. A stored whole-expression synonym can show that the phrase is a lexical unit. It does not show that the sense is semantically noncompositional. + +Procedure v2 keeps the five classes and the sealed bucket map. Before the class, it records three states: referential, lexicalized, and compositional. Each state is yes, no, unknown, or conflict. An unknown state is not a no. Referential yes, from an instance-hypernym pointer or a parenthetical four-digit lifespan, remains referential and reject. A co-lemma or a whole-lemma lexical pointer is lexicalized yes. A derivation or pertainym back to a constituent, a stored one-token alternation, or a comparative or superlative operator gloss is compositional yes. The alternation does not set lexicalized to no. The operator gloss sets lexicalized to no and compositional to yes, so when referential is not yes the class is ordinary compositional. Lexicalized yes together with compositional yes, when referential is not yes, is lexicalized compositional. That cell is reachable. The procedure does not force rows into it. + +WordNet 3.0 has no structural noncompositionality field once the co-lemma shortcut is retired. The compositional no-signal set is empty, so the lexicalized-noncompositional cell stays defined and does not fire. A later encoder must not invent a replacement signal. Missing lexicalization evidence stays unknown. On the record already frozen in procedure v1, `as far as possible` is therefore ambiguous under v2, and `road to damascus` stays ambiguous. Neither illustration is a required fit, and neither class was written onto the development manifest. An insufficient record stays ambiguous and maps to quarantine. A lower ambiguous rate is not a success criterion. Referential precision is not traded for coverage, and no technical-term gloss list was added. + +No JSON Schema document exists for procedure v2. The schema name on the artifact is not a JSON Schema file. Older unbind sample, label, and report schemas were not applied. Procedure sha256 `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. v1 error analysis sha256 `471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e`. Change note sha256 `443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c`. Tracker sha256 `af11d20ebefec5629718617ebacc07d8b4cc36c61e59e829d153b05b9397ca40`. + +The reviewed positional inventory remains 225. A future encoded replay of those same rows may report lexicalized-compositional support, the false-high count, the ambiguous rate, referential precision and coverage, the sense-class distribution, the evidence-state distribution, and the operator confusion tables. This freeze sets no accuracy threshold. Lexicalized-compositional support above zero would be a structural readiness signal, not a hard pass/fail rule. + +The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. The procedure records the shared invariant that lexicalized identity, compositionality, and semantic shift are different questions. That note was not edited, and the lexeme screen was not implemented. v3 through v7 sources are unchanged. The v1 implementation, v1 predictions, and v1 report are unchanged. Acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. Prediction sha256 remains `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef`. Report sha256 remains `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. + +The next named transition is `ENCODE_PROCEDURE_V2_AUTHORIZATION`. This pass does not authorize it. An unseen measurement stays deferred until an encoded v2 replay of these 225 rows has been examined. + +## Unbind sense screen procedure v2 development replay — 2026-09-28 + +Procedure v2 is encoded and the same 225 development rows were replayed once. State is `DEVELOPMENT_ANALYZED_V2`. The path was `PROCEDURE_V2_FROZEN`, `ENCODE_PROCEDURE_V2_AUTHORIZATION`, `ENCODED`, `225_ROW_DEVELOPMENT_REPLAY`, `DEVELOPMENT_ANALYZED_V2`. The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v2 bytes stay `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. Procedure v1 stays `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. No sense class was written onto the development manifest. No measurement sample was drawn. + +The replay asks whether the frozen WordNet record can instantiate the five classes, and `LEXICALIZED_NONCOMPOSITIONAL` in particular, without treating a single-word co-lemma as compositional NO. It does not instantiate that class. High support is 0. Compositional NO count is 0. False high is 0. A co-lemma sets lexicalized YES and leaves the row ambiguous when compositionality is unknown. Ninety-five rows are lexicalized YES. One hundred eighty-two rows are compositionally unknown. + +Sense classes are referential 27, lexicalized noncompositional 0, lexicalized compositional 16, ordinary compositional 0, and ambiguous 182. The ambiguous rate is 182/225. Buckets are high 0, secondary 16, reject 27, and quarantine 182. Of the 16 lexicalized-compositional rows, 15 are a stored one-token alternation on a lexicalized synset and 1 is a derivation or pertainym back to a constituent. That support is a readiness signal. It is not a pass/fail result. Ordinary compositional support is 0 because no row has lexicalized NO. The one grammatical-operator gloss also has an unrelated co-lemma, so lexicalized state is conflict and the class is ambiguous. + +Referential yes remains an instance-hypernym pointer. All 27 referential rows use that pointer. Descriptive reject precision is 27/27 and recall is 27/77. False reject is 0. One of those 27 was quarantined by procedure v1 because the instance pointer shared the synset with a one-token alternation. Procedure v2 keeps the instance pointer decisive and records the alternation as compositional yes. Secondary precision is 7/16 and recall is 7/46. False secondary is 9 and is descriptive only. High precision is not computable, because no row was predicted high. High recall is 0/101. + +Evidence states: referential yes 27, no 97, unknown 101. Lexicalized yes 95, no 0, unknown 129, conflict 1. Compositional yes 43, no 0, unknown 182, conflict 0. `as far as possible` is referential no, lexicalized unknown, compositional yes, and ambiguous. `road to damascus` is unknown on all three states and ambiguous. Neither illustration was written onto the manifest. + +Prediction sha256 `1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a`. Report sha256 `93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5`. Tracker sha256 `6ecb7e16bbe2ea3be2efca51c46cc970f2a5ba8ec8cd9c101eaef8398f5a0e61`. Acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The v1 prediction and report hashes stay `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef` and `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. The development manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. v3 through v7 sources and the v1 implementation are unchanged. + +These figures are development evidence. No accuracy threshold was applied. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `V2_DEVELOPMENT_RESULT_REVIEW`. This pass does not authorize it, and it does not authorize an unseen measurement. + +## Semantic evidence source v1 specified — 2026-09-28 + +The procedure-v2 development result is reviewed and frozen. WordNet 3.0 structural evidence supports referentiality, whole-expression lexicalization, and some positive compositionality. Under the frozen procedure it does not provide a deterministic structural signal for semantic noncompositionality at useful coverage. That is a source-capability limitation. Procedure v2 was not retuned. + +Observed on the 225 development rows: compositional NO 0, lexicalized noncompositional 0, high 0, ambiguous 182/225, whole-expression lexicalization 95, positive compositionality 43, referential 27. Descriptive reject precision remains 27/27. False reject is 0. False high is 0. No measurement bar was applied. Review sha256 `77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d`. Limitation sha256 `3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a`. Status is `WORDNET_STRUCTURAL_SOURCE_LIMITATION_CONFIRMED`. + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` is `SPEC_FROZEN`. It is an evidence-source evaluation framework, not an unbind classifier. The research question is which additional source can establish semantic noncompositionality independently of WordNet whole-expression lexicalization, with provenance and precision enough to become eligible input to a future sense-first screen. No source is selected. Nothing was encoded or applied. No external resource was downloaded. The only local lexical source is WordNet 3.0, and that source is the one whose limitation was just confirmed. + +Five candidate families are specified and left unselected: explicit lexical-semantic resources, a deterministic comparison of the bound sense with a composition of its parts, a curated linguistic annotation, a model-based judgment, and a later hybrid of WordNet structure plus one independent semantic source. An LLM completion is not runtime truth. A source must pass provenance, target alignment, sense alignment, label independence, reproducibility, abstention, a ban on phrase exceptions, development evaluation before integration, and isolation of any later generalization rows. The future output can be YES, NO, or UNKNOWN. UNKNOWN stays valid. No hard accuracy target is frozen. A high unknown rate is acceptable when false semantic claims stay low. Reducing ambiguity is not a reason to select a source. + +The same 225 rows remain development evidence for a future candidate test. They were not mutated and no source was run on them. False high and false reject stay the safety priority for a later integration specification. That specification does not exist yet, and no unseen-measurement bar was registered. + +A shared primitive, `RUNE.SEMANTIC_COMPOSITIONALITY`, is architecture only. The phrase lane would eventually use constituent words, phrase structure, and the bound sense. The single-lexeme lane would use morphemes or compound constituents and the bound sense. Lexicalized identity, compositionality, and semantic shift remain different questions. `RUNE.LEXEME_STRUCTURE_SCREEN.v1` was not implemented. Its architecture note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. + +No JSON Schema document exists for this source-evaluation family. Hypothesis sha256 `39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787`. Acceptance sha256 `1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94`. Evaluation plan sha256 `472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a`. Architecture sha256 `180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36`. Tracker sha256 `521eccb8bd068d4697a3ffdc06e3b44feb4eec3c8a3bd6aebcea11dd288b7dfc`. + +Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`, predictions `1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a`, and report `93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5`. Procedure v1, the sense-screen acceptance, and the 225-row manifest are unchanged. The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v2 stays `DEVELOPMENT_ANALYZED_V2`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. No procedure v3 was created. + +The next named transition is `CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. Selecting or integrating a source is a later decision. + +## MAGPIE candidate evaluation — 2026-09-28 + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` is `CANDIDATE_SOURCE_EVALUATED` for one candidate, MAGPIE. The candidate remains unselected. `selected_source` is `none`. Runtime integration is false. The encoded sense screen and procedure v2 were not modified. + +The evaluated artifact is the author corpus `MAGPIE_unfiltered.jsonl` from `https://github.com/hslh/magpie-corpus` at commit `7fa677b82b9a772dfa54bbdd0fb414412d73db3b` (2020-06-07T09:57:12Z). The file was acquired from that commit on 2026-09-28T05:50:00Z. It contains 56,622 instances and 1,756 potentially idiomatic expression types. Assigned labels are idiomatic 40,011, literal 16,168, other 436, and unclear 7. The filtered split files were not acquired and were not mixed into the evaluation. No Hugging Face package was used. No other repository named magpie was used. + +Licenses are recorded per artifact. The LREC 2020 PDF, sha256 `8247c926909772ce317d5f33cddb83caa51969eb5ef928bcbadbbcf05e39c979`, states on page 1 that the ELRA proceedings text is licensed under CC-BY-NC. That statement is the publication license. The dataset LICENSE file in the pinned commit is Creative Commons Attribution 4.0 International, sha256 `05ab88f3f9da1d05f9c5bf0a7c45c49a9007f877dd9c237a5bf668276fe04c3b`. That file is also the only code license in the commit. The evaluated jsonl sha256 is `541ee535e93d71eff85351351665115e2a9f22ad736423881da5774a93bc880e`. Provenance sha256 `bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd`. License receipt sha256 `8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972`. + +Surface comparison is orthographic. Case and separator folding produced 25 exact matches. One normalized match folds a comma: `day in day out` corresponds to the MAGPIE type `day in, day out`. `to a t` casefolds onto `to a T` and is counted as exact. Recorded variant categories such as inflection, dashes, and possessive describe occurrences of a type. They are not alternate expression strings, so variant matches are 0. Ambiguous collisions are 0. Unmatched rows are 199. Surface coverage is 26/225. + +Sense alignment requires equality between a source sense identifier and the supplied WordNet synset. The pinned instances have no synset, sense key, or gloss field. Alignment coverage is 0/225. Every row is UNKNOWN. Semantic noncompositionality is YES 0, NO 0, UNKNOWN 225. Abstention is 225/225. Evidence codes are `magpie_surface_only` 26 and `magpie_no_match` 199. A surface hit does not assign YES. Fifteen matched rows have only idiomatic instance labels, including `throw in the towel` and `dressed to the nines`, and they stay UNKNOWN. Ten matched types contain both literal and idiomatic instances, including `round the bend` and `to the letter`. Those counts stay on the match record. They are not a MIXED sense alignment and they are not semantic YES. `as luck would have it` is an exact match with only idiomatic labels, and the historical operator bucket is secondary. The idiomatic labels were not converted into YES. + +Operator labels were joined after the semantic evidence file was written. They are not runtime evidence. The descriptive proxy, stated as a proxy, treats operator HIGH as the comparison class for semantic YES and operator SECONDARY as the comparison class for semantic NO. YES precision is `NOT_COMPUTABLE`. NO precision is `NOT_COMPUTABLE`. False YES is 0. False NO is 0. Both counts are zero because no YES or NO claim was emitted. Operator REJECT is its own row in the joint table: 77 unknown. It was not read as compositional NO or as semantic YES. The joint table is HIGH 101 unknown, SECONDARY 46 unknown, REJECT 77 unknown, and quarantine 1 unknown. + +Development status is `CANDIDATE_INSUFFICIENT`. Coverage of familiar phrases is not a reason to select the source. The readiness question, non-zero semantic YES with a defensible sense alignment, is not met. Match sha256 `84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83`. Sense-alignment sha256 `037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66`. Semantic-evidence sha256 `89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577`. Evaluation sha256 `74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba`. Decision sha256 `6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d`. Tracker sha256 `c3fe6f2fd21d07b8b45d9f26ffeebbcfbe3cb68a1f5a7952dee94d9a2c995936`. + +No JSON Schema document exists for this source-evaluation family. The schema names on the artifacts name those artifacts. Older unbind sample, label, and report schemas were not applied. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. No sense class was written onto it. v3 through v7 sources are unchanged. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. A later authorization may name one curated resource whose entries carry a WordNet synset offset or sense key for the expression sense. PARSEME, STREUSLE, PIE, EPIE, and NCS were not downloaded and are not selected. + +## Korkontzelos–Manandhar candidate evaluation — 2026-09-28 + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` stays `CANDIDATE_SOURCE_EVALUATED`. A second candidate, the Korkontzelos–Manandhar 2009 compositionality set, was evaluated and not selected. `selected_source` remains `none`. Runtime integration remains false. MAGPIE stays `CANDIDATE_INSUFFICIENT`. Its artifacts were not modified. + +The pinned source is Table 1 of Korkontzelos and Manandhar, “Detecting Compositionality in Multi-Word Expressions,” ACL-IJCNLP 2009 short papers, pages 65–68, anthology `P09-2017`. Canonical URL `https://aclanthology.org/P09-2017/`. PDF sha256 `046da9fc26cfdf220e41ad914314f0703ea3c84cd86e045b20146b34189113d1`, acquired 2026-09-28T06:41:05Z. The paper states that the items were drawn from WordNet 3.0. The bibliography cites Miller 1995 for WordNet in general. That citation is not a second version number. Reconstruction used the local Princeton WordNet 3.0 data files and no other WordNet release. The suggested counts of about 60 compositional and 56 noncompositional items do not match this table. The verified inventory is 19 compositional and 19 noncompositional. A later 2010 sample by the same authors was not acquired. + +The PDF header reads `c©2009 ACL and AFNLP`. The ACL Anthology states that materials prior to 2016 are licensed under CC-BY-NC-SA 3.0 and that copies may be made for teaching and research. That is the publication license of the acquired PDF. No separate dataset license exists. The evaluation list is the table inside the paper. No code artifact was published with the table, and none was acquired. Bold, underline, and italic marks in the table report system detections. They were not read as gold labels. + +Table 1 preserves the surface and the section label. It does not preserve a synset offset, sense key, lemma key, part of speech, or gloss. Source sense identity is therefore absent on all 38 items. A lookup key folds case, underscore, and apostrophe shape, and it keeps hyphens. That key is not a sense choice. The key matches exactly one PWN 3.0 synset for 30 items: `EXACT_UNIQUE_RECONSTRUCTION`. Those 30 are the 19 compositional items and 11 noncompositional items. Eight noncompositional items match two synsets each: `black maria`, `dead end`, `dutch oven`, `goat's rue`, `green light`, `high jump`, `living rock`, and `prince Albert`. They are `AMBIGUOUS_MULTIPLE_SYNSETS`. No gloss, operator label, or frequency was used to pick one. `EXACT_SOURCE_ID` is 0. `NO_PWN3_MATCH` is 0. `VERSION_CONFLICT` is 0. On the inventory, the conservative map gives semantic YES 11, NO 19, and UNKNOWN 8. Only the unique reconstructions carry YES or NO. + +None of those 30 synsets occurs in the frozen 225 development rows. Surface keys also do not meet any development row. The join is synset identity, so the Hyperlex result is YES 0, NO 0, UNKNOWN 225. Sense-aligned coverage is 0/225. Abstention is 225/225. Evidence codes on the 225 rows are `km_no_match` 225. Operator labels were joined after the evidence file was written. Under the stated proxy, YES precision and NO precision are `NOT_COMPUTABLE`. False YES is 0. False NO is 0. The joint table is HIGH 101 unknown, SECONDARY 46 unknown, REJECT 77 unknown, and quarantine 1 unknown. Reject and quarantine were not read as compositionality. + +Development status is `CANDIDATE_INSUFFICIENT`. The inventory can reconstruct monosemous PWN 3.0 identity, which MAGPIE could not, and that reconstruction still supplies no YES row on this development set. The stop-condition finding `WORDNET_DERIVED_BUT_SENSE_IDENTITY_NOT_PRESERVED` was not frozen. Eight of 38 items are polysemous, not most of them, and 30 items do reconstruct one synset. The development sample and this table are disjoint. MAGPIE remains surface coverage 26/225, sense-aligned coverage 0/225, YES 0, NO 0, abstention 225/225. This candidate is surface coverage 0/225, sense-aligned coverage 0/225, YES 0, NO 0, abstention 225/225. Coverage is not the comparison. Neither candidate puts a sense-aligned YES on a development row. + +Provenance sha256 `88fbfa077d2394b8ce631ec700c482ad98a0f62f0ab06964f4f43a77d906f9aa`. License receipt sha256 `eb4c9406aab7f9021d346ebd24634cad1dcf0e6076f069e73b1d4b2702dbe2f8`. Inventory sha256 `3a35ddc0b2c04a5386c6112a2bb3cdf22735edbe2fd791f0c2ec542fe9184d81`. PWN 3.0 alignment sha256 `531d13cf1bdbc939fc11d9ef5864c6878a9f58210368c596c0aad08ff75e61c9`. Hyperlex semantic evidence sha256 `4e98f8f06713ffcf02549305aef140e92b9b8b2790f471f3ce1f41fe208b2f7a`. Evaluation sha256 `00a1d1f667635354e20e5002c4ece846fe3a8125a7ca12ebe09bb7e28dedd1a7`. Comparison sha256 `240b3ea468d22a80ac5e5765521cc691b80f914b7baff6bb1ae4819475f54965`. Decision sha256 `93a07e3c78b53a69965497410c34ddb52c2a5d3add3fb2f3cd3fd9ca84eb3d9f`. Tracker sha256 `b3546102410058d3596c4563604998685753a698b2bc533b353e1f039440f704`. + +No JSON Schema document exists for this source-evaluation family. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. v3 through v7, the lexeme-structure note, and the WordNet source-limitation finding are unchanged. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. A later candidate has to carry a PWN synset identity that can meet the development rows. Surface-only MWE corpora are not the next step. `CANDIDATE_SOURCE_SELECTION_REVIEW` is not the next step, because this candidate is not promising and is not selected. + +## Semantic compositionality residual candidate evaluation — 2026-09-28 + +`NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION` applies only to `RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1`. The candidate is an evaluation of a continuous residual. It is not selected, it is not integrated into `UNBIND_SENSE_SCREEN`, and it does not create procedure v3. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. No unseen measurement sample is drawn. No training gold is created. No row is admitted or settled. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. MAGPIE and the 2009 Korkontzelos–Manandhar artifacts stay frozen, and both remain `CANDIDATE_INSUFFICIENT`. + +The design was hashed before any development row was encoded. The whole sense is the supplied surface, its POS, and the frozen gloss, in the form `{surface} ({pos}): {gloss}`. A constituent uses the same form on the resolved PWN 3.0 lemma, POS, and first gloss clause. Whitespace tokenization keeps a hyphen inside one token. A frozen closed class of determiners, prepositions, pronouns, auxiliaries, and other structural tokens is ignored. A row needs two content tokens. Sense resolution accepts one derivation or pertainym pointer whose target word number is nonzero and whose lemma matches the constituent or a one-hop exception neighbor. That result is `EXACT`. With no exact target, one lexical synset across noun, verb, adjective, and adverb is `UNIQUE`. Several targets are `AMBIGUOUS`. None is `UNRESOLVED`. A target word number of zero does not name a lemma. The source word number is not a filter. An ambiguous or unresolved content constituent makes the row `UNKNOWN`. The pass does not use an arbitrary first sense, gloss similarity, an embedding to choose a sense, an operator label, or a manual sense choice. Composition is one operator, `normalized_mean_v1`: binary64 L2-normalize each constituent vector, take the mean, and L2-normalize the mean. Duplicate synsets stay, one vector per content token. The residual is `1 - cosine`, recorded with `format(value, '.10f')` and no clamp. A vector hash is sha256 of little-endian binary32 bytes. No semantic-noncompositionality threshold is chosen, and the residual is not turned into YES or NO. + +The pinned model is `sentence-transformers/all-MiniLM-L6-v2`, revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`, acquired locally at `2026-09-28T06:52:00Z` from the Hugging Face revision. The model license is Apache-2.0. Weights sha256 `53aa51172d142c89d9012cce15ae4d6cc0ca6895895114379cacb4fab128d9db`. `tokenizer.json` sha256 `be50c3628f2bf5bb5e3a7f17b1f74611b2561a3a27eeab05e5aa30f411572037`. `vocab.txt` sha256 `07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3`. Pooling is mean tokens into 384 dimensions, and the stack includes a normalize module. Encode uses CPU, float32, `normalize_embeddings=True`, batch size 1, seed 0, one thread, and eval mode. `use_deterministic_algorithms` stays false. The effective maximum sequence length is 256; overflow would abstain rather than truncate, and no row overflowed. Runtime versions are Python 3.12.3, torch `2.14.0+cpu` (Apache-2.0 with additional component licenses), sentence-transformers 6.1.0 (Apache-2.0), transformers 5.17.0 (Apache 2.0), tokenizers 0.23.2 (Apache), and NumPy 2.5.3 (BSD-3-Clause and other component licenses). ONNX, OpenVINO, and TensorFlow weight copies were not acquired. No API embedding service was called. Constituent glosses come from Princeton WordNet 3.0, whose license is separate from the model license. The encoder was run twice and the float32 hashes matched before the score file was written. + +Constituent extraction is `EXTRACTED` 166 and `UNKNOWN` 59. The 59 are rows with fewer than two content tokens. Across content constituents the resolution counts are `EXACT` 4, `UNIQUE` 60, `AMBIGUOUS` 370, and `UNRESOLVED` 70. Row abstentions are ambiguous content 145, fewer than two content tokens 59, and unresolved content 18. Three rows are `SCORED` and 222 are `UNKNOWN`. Ten representation texts were encoded. Six content tokens contain a character outside letters, digits, hyphen, and apostrophe; the extractor did not strip them. The three scored rows are operator `REJECT`: `california fern` residual `0.2139784896`, `monoamine oxidase inhibitor` residual `0.1768095281`, and `.22 caliber` residual `0.1309395496`. Reject is a referential axis, not semantic no. Their descriptive distribution is count 3, min `0.1309395496`, p25 `0.1538745388`, median `0.1768095281`, p75 `0.1953940088`, max `0.2139784896`, mean `0.1739091891`. Operator `HIGH` has 0 scored rows of 101, so the primary comparison is `NOT_COMPUTABLE`. `SECONDARY` has 0 of 46. `QUARANTINE` has 0 of 1. No threshold was fit to these labels. + +Development status is `CANDIDATE_DISTRIBUTION_FROZEN` because the preregistered rule uses that status whenever the scored count is greater than zero. The status freezes the score distribution. It does not say the residual separates noncompositionality, and the empty HIGH distribution supplies no such comparison. `selected_source` remains `none`. Candidate spec sha256 `39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d`. Provenance sha256 `2c34a7d0d480cde564bda694dbaa349550814c5fb7c647bfa3bbbc9db5e26886`. License receipt sha256 `323a5bad21ac74d6a1fd94c54dba95a0046b633cd7328dd6680972525abc0909`. Score artifact sha256 `cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7`. Evaluation sha256 `ba782622d4c68d23c53ae0ffb5f54f1e43c8cf059b44b7adb34e0eb356ef3896`. Decision sha256 `45922eba294b0b7d7238ce71c6157641259ea292f65743ba2ee1324ff9b52617`. Tracker sha256 `14d4daaab1e4740a596acaef7f4dbd9ed11edaf2182ff22f7aa339325febf591`. Every score row carries the spec hash, and the spec contains no residual. Operator labels were joined only after the score file was hashed. + +No JSON Schema document exists for this source-evaluation family. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. Sense classes on those rows stay null. v3 through v7, the lexeme-structure note, and the WordNet source-limitation finding are unchanged. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION`. This pass does not authorize it. A threshold is not supported by a HIGH distribution, because no HIGH row was scored. `CANDIDATE_SOURCE_SELECTION_REVIEW` is not the next step. `selected_source` remains `none`. + +## Residual v1 coverage limitation — 2026-09-28 + +`RESIDUAL_DEVELOPMENT_RESULT_REVIEW` is authorized for the frozen residual distribution only. `RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION` is not authorized. The residual has not been falsified. The experiment could not test HIGH against SECONDARY because constituent sense ambiguity collapsed coverage. Scored rows are 3, all operator REJECT. Unknown rows are 222. HIGH scored 0 of 101. SECONDARY scored 0 of 46. REJECT scored 3 of 77. Ambiguous content abstains 145 rows, fewer than two content tokens abstain 59, and unresolved content abstains 18. `threshold_eligible` is false. `source_selection_eligible` is false. The primary limitation is `constituent_sense_resolution`. The finding is `SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED`: the dominant blocker is deterministic constituent sense resolution, not the residual model. The model, revision, normalized mean, and `1 - cosine` residual stay frozen. No residual score was recomputed. + +## Constituent sense resolution v1 — 2026-09-28 + +`RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` starts from that limitation. The question is whether a frozen deterministic resolver can give enough content constituents a PWN 3.0 sense to make a later residual replay testable. Operator labels, residual scores, unbind buckets, phrase exceptions, manual choices, an LLM, and the residual embedding model are not inputs. The specification and procedure were hashed before the replay. The resolver then ran twice from that specification. The two artifacts matched, so determinism is `IDENTICAL`. + +Tier 1 is structural. A lexical pointer in the frozen set `! + \ ^ * & < $` whose target word number is nonzero and whose lemma matches the constituent, or one exception hop, is structural evidence. One such synset is `EXACT` by `STRUCTURAL_EXACT`. More than one is `AMBIGUOUS` and Extended Lesk does not override it. With no structural target, one lemma synset across noun, verb, adjective, and adverb is `EXACT` by `UNIQUE_LEMMA`. No synset is `UNRESOLVED`. The extra pointer symbols did not raise structural exact above the residual baseline: structural exact stays 4 and unique lemma stays 60. + +Extended Lesk v1 runs only when several lemma synsets remain. Candidate text is the full PWN 3.0 gloss, including quoted examples, plus the first gloss clause of each depth-1 synset reached by `@ + \ & ^ =`. Context is the parent frozen gloss plus the first gloss clause of other constituents in the row that tier 1 already resolved. Lesk output is not reused as context. Normalization casefolds, folds apostrophes, and splits on characters outside letters, digits, apostrophe, and hyphen. Stopwords are the residual v1 structural class. Tokens shorter than two characters are dropped. There is no stemmer and no lemmatizer. The score is the sum of squares of greedy longest contiguous overlaps, and matched tokens are not reused. Scores are integers. The minimum margin is 1, so a tie, including a zero-overlap tie, is `AMBIGUOUS`. A strict win is `RESOLVED`. The parent gloss is local context only. Parent sense is not constituent sense. + +Every one of the 504 content constituents was attempted. Status counts are `EXACT` 64, `RESOLVED` 121, `AMBIGUOUS` 249, and `UNRESOLVED` 70. Against the residual baseline of `EXACT` 4, `UNIQUE` 60, `AMBIGUOUS` 370, and `UNRESOLVED` 70, the 60 unique lemmas are the same senses under the new `EXACT` label, unresolved stays 70, and 121 of the 370 ambiguous constituents become `RESOLVED`. The other 249 stay `AMBIGUOUS`. The tie rate on Lesk attempts is 249/370. Exact proportion is 64/504. Context-resolved proportion is 121/504. Unresolved rate is 70/504. Abstention, ambiguous plus unresolved, is 319/504. There is no constituent-sense gold, so these figures are coverage, determinism, tie behavior, and provenance. They are not accuracy. + +A row is `RESIDUAL_READY` only when it has at least two content constituents and every one is `EXACT` or `RESOLVED`. That holds for 20 rows. The other 205 are `UNKNOWN`. After the resolution file was hashed, operator labels give HIGH 4 ready and 97 unknown, SECONDARY 4 ready and 42 unknown, REJECT 12 ready and 65 unknown, and quarantine 0 ready and 1 unknown. The necessary condition, at least one ready HIGH row and one ready SECONDARY row, is met. It is not sufficient. Because 249 of 370 baseline-ambiguous constituents remain AMBIGUOUS, which is more than half, the preregistered finding is `CONSTITUENT_WSD_COVERAGE_INSUFFICIENT`. No second resolver was added. A projection, not a replay, says 20 rows could enter a later residual pass: HIGH 4, SECONDARY 4, REJECT 12. The residual was not rerun. + +Resolver state is `DEVELOPMENT_ANALYZED`. Candidate status is `COVERAGE_INSUFFICIENT`. `selected_source` remains `none`. Residual state remains `CANDIDATE_DISTRIBUTION_FROZEN` with coverage limitation `CONFIRMED`. Limitation sha256 `fc8839c15a7638b2bfca1cf0548bfb4d5f433434bae0fea2944a528dd15d6142`. Review sha256 `1c1b69856dd88567167fd5c958cd8db6d39ab9ec74a4e0ed3e667a521c82e6fa`. Resolver spec sha256 `176e6219ddc3127814a25d39ad26e3571817f7ea8323d685e081d2e0fd867acb`. Procedure sha256 `9f76b64aa6aac09bd56ca9cc8a847cda12b31a58eea8c4b51426733917d54248`. Replay sha256 `a0c707ab55e02f627a698c33ddc0ca398e34aa0bafc19b13e72422bc26d97d0a`. Analysis sha256 `6d47610014f74094394355a50419fe24441103bec09458b6f25ea392fb57151c`. Decision sha256 `ff8b90ebe5d48151dc68ddbf676e1f27d4cee5e3be2730f999c9b089f1392caa`. Tracker sha256 `c72e55c8096143b8675d4aa995c5d4b19de95d256dd234d32a421ba098bc7d4f`. + +No JSON Schema document exists for this source-evaluation family. No residual threshold was created. No semantic YES or NO was emitted. No unseen sample was drawn. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. The 225-row manifest, procedure v2, MAGPIE, Korkontzelos–Manandhar, and the residual v1 artifacts are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. `RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION` is not the next step while the constituent-coverage finding stands. `selected_source` remains `none`. + +## Model-based constituent WSD candidate v1 — 2026-09-28 + +`MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION` evaluates one WordNet-native model on the 249 constituents that Extended Lesk v1 left `AMBIGUOUS`. The model is `kanishka/GlossBERT`, family GlossBERT, canonical source `https://huggingface.co/kanishka/GlossBERT`, revision `0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a`. It is a BERT sequence classifier fine-tuned on SemCor 3.0. Each decision is a score over a supplied Princeton WordNet 3.0 gloss, so the output is a candidate synset or an abstention. The card license is MIT. The checkpoint is a third-party Hugging Face upload, not the authors' Google Drive file. The original code repository `https://github.com/HSLCY/GlossBERT` is MIT. Inference is local. Device is CPU. Dtype is float32. Batch size is 1. One thread is used and mkldnn is disabled. Runtime is Python 3.12.3, torch 2.14.0+cpu, transformers 5.17.0, tokenizers 0.23.2, and numpy 2.5.3. The positive class is index 1. A non-Hyperlex polarity control fixed that index before any development constituent was scored: a financial sentence selects `noun:08420278`, and a river sentence selects `noun:09213565`. + +The candidate specification was hashed before those 249 constituents were scored. Spec sha256 `c861ff7fff11ae6a790531267229c18d6e6e0a171a9bf6c34cfb6f7e7b14498c`. Weight sha256 `60706c7618f8ccbfa7d0a6d1d1009765a7146ea5f4232924ed9f1c46d521c898`. Vocab sha256 `07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3`. Tokenizer config sha256 `09e49d0e788d25991da77d37b10eaa6a86a4e94e2127de8bedc94eb45baf2d84`. Each constituent is scored only among its frozen PWN 3.0 candidate synsets. The context is the parent surface with the target content token in double quotes. The paired text is the matched lemma, a colon, and the first gloss clause. The parent gloss is not appended. Operator labels, residual scores, and the residual embedding are not inputs. The score is the class-1 softmax probability, quantized to six decimal places, half even. The abstention rule, frozen before scoring, requires a top probability of at least 0.50 and a top-versus-second margin of at least 0.10. An equal top score stays `AMBIGUOUS`. A sense outside the candidate set would be `INVALID`. The same inference then ran a second time. The two raw artifacts matched, so determinism is `IDENTICAL`. + +Tier 3 attempted 249 constituents. It resolved 153, left 96 ambiguous, and produced 0 invalid outputs and 0 errors. The 153 resolutions use evidence `model_margin`. The 96 abstentions use evidence `model_abstention`. Exact score ties are 0. Top-score bins are 81 below 0.50, 30 from 0.50 to 0.60, 35 from 0.60 to 0.70, 26 from 0.70 to 0.80, 32 from 0.80 to 0.90, and 45 from 0.90 to 1.00. Margin bins are 50 below 0.10, 56 from 0.10 to 0.25, 56 from 0.25 to 0.50, and 87 at 0.50 or more. There is no constituent-sense gold on these rows, so the figures are coverage, validity, determinism, confidence, and abstention. They are not accuracy, precision, recall, or F1. + +Combined with the frozen earlier tiers, constituent counts are `EXACT` 64, `LESK_RESOLVED` 121, `MODEL_RESOLVED` 153, `AMBIGUOUS` 96, and `UNRESOLVED` 70. Tier 3 did not replace a structural exact, a unique lemma, or an Extended Lesk decision that had already cleared its margin. A row is projected `RESIDUAL_READY` only when it has at least two content constituents and every one is exact, Lesk-resolved, or model-resolved. That projection was hashed before operator labels were joined. Projected ready rows are 73. Unknown rows are 152. After the join, HIGH is 28 ready and 73 unknown, SECONDARY is 11 ready and 35 unknown, REJECT is 34 ready and 43 unknown, and quarantine is 0 ready and 1 unknown. Against the lexical baseline of HIGH 4, SECONDARY 4, and total 20, the deltas are +24, +7, and +53. Invalid outputs are 0 and the repeat is identical, so the preregistered coverage gate returns `CANDIDATE_PROMISING`. The gate is a readiness projection. It does not say the selected senses are correct, and it does not replay the residual. + +`selected_source` remains `none`. The model is not integrated into runtime. `RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` stays `DEVELOPMENT_ANALYZED` with lexical status `COVERAGE_INSUFFICIENT`. `RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1` stays `CANDIDATE_DISTRIBUTION_FROZEN`. `threshold_eligible` stays false. No residual score was recomputed and no residual threshold was created. No semantic YES or NO was emitted. No unseen sample was drawn. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. The resolver artifacts, residual artifacts, MAGPIE artifacts, Korkontzelos–Manandhar artifacts, procedure v2, and the 225-row manifest are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +No JSON Schema document exists for this source-evaluation family. Provenance sha256 `6d91283244a88f44477c836e6549118d5d96a36bb878bf448fcadb13ff765e11`. License receipt sha256 `67d45243959e3e76643a675d49b508a73d0f7f0bf8fbf060ce7b83c9200c621b`. Raw output sha256 `a0e508c225e6db4cdbcae701682202b1d854f3762546dbfe10b84a15e0e9e17c`. Resolution sha256 `ed945989cf4947ac84633ba2c4aa10c1ba381d2396da0b573a844f83ec367a18`. Readiness projection sha256 `c75834faf4a84d36e83246244e0aa7c6c7788c3a57cfdb7f77c7628a52023328`. Analysis sha256 `8ca0d8c8dd7e510a04daef6b3fbd78ace6d20b173e88d2a3d0c2e10fdaafe30b`. Decision sha256 `c8ed0b8acaaa415277c5f9bdbf982d075136b795b5d95fd8813d5a3010072dbb`. Tracker sha256 `6da1e9730c785d2784d23433455515989d312ed8c76f4763b3a2dd6a1f2c4b2d`. + +The next named transition is `RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION`. This pass does not authorize it. The residual is not replayed. `selected_source` remains `none`. +## Residual replay with model-resolved senses — 2026-09-28 + +`RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION` runs one residual replay on the frozen Tier 1, Tier 2, and Tier 3 constituent senses. The question is whether operator-HIGH rows then show larger frozen semantic-compositionality residuals than operator-SECONDARY rows. This is a development distribution. It is not a classifier, and it does not train a composition function. The residual model stays `sentence-transformers/all-MiniLM-L6-v2` at revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`. Settings stay CPU, float32, seed 0, batch size 1, one thread, eval mode, normalized embeddings, and maximum sequence length 256. Composition stays `normalized_mean_v1`. The residual stays `1 - cosine_similarity(whole_sense_vector, normalized_mean(constituent_sense_vectors))`, stored at 10 decimal places. Constituent extraction, structural exact, unique lemma, Extended Lesk v1, and GlossBERT are not retuned. No confidence or margin rule changes. No second embedding, pooling, distance, or weight scheme is tried. + +The integrated resolution is built from the frozen resolver replay and the frozen GlossBERT resolution, then checked against the frozen readiness projection before any vector is computed. The projection reproduces: 73 `RESIDUAL_READY` rows and 152 `UNKNOWN` rows. The integrated file is hashed before scoring. Integrated sha256 `0f5dafc3676a4071ce8c889e58958b90203589aa3e91d78111b4e3292bdd87fb`. Only ready rows are scored. Representation text, lemma choice, gloss fields, example inclusion, normalization, and the residual formula are the frozen residual v1 functions. The encoder then runs twice. Row readiness, representation texts, vector hashes, and 10-decimal scores match, so determinism is `IDENTICAL`. No ready row exceeds 256 tokens. Scored rows are 73 and scoring abstentions are 0. The score file is hashed before operator labels are read. The receipt records that sequence with `operator_labels_joined` false. Score sha256 `16c0a9eaa918ac4a6e8223cafcbf4b1918cb212769063e262cf29a279f1048f6`. Receipt sha256 `675b1b8b8f1b1e6f19c7d320e7b8fe516eae92b60e966afbce407c1f0482e736`. + +After that hash, the historical operator labels join. Ready counts are HIGH 28, SECONDARY 11, REJECT 34, and quarantine 0. REJECT stays out of the primary comparison. HIGH residuals have count 28, minimum 0.1169573790, p10 0.2930005820, p25 0.3564342144, median 0.4040327275, p75 0.4404865614, p90 0.5184032344, maximum 0.5908804826, mean 0.3943616308, and sample standard deviation 0.1017789927. SECONDARY residuals have count 11, minimum 0.1768283745, p10 0.2079496807, p25 0.2436642596, median 0.3307873412, p75 0.4000929338, p90 0.4561366947, maximum 0.5587792172, mean 0.3299980313, and sample standard deviation 0.1186719529. The HIGH mean exceeds the SECONDARY mean by 0.0643635995. The HIGH median exceeds the SECONDARY median by 0.0732453863. Mann-Whitney U for HIGH, with half credit for ties, is 207. There are 207 pairs in which the HIGH residual is larger, 101 in which it is smaller, and 0 ties, out of 308 pairs. The rank-biserial correlation is 0.344156. Descriptive ROC AUC, with HIGH as the positive class and a larger residual as the HIGH-like score, is 0.672078. The hypothesized direction `HIGH residual > SECONDARY residual` is `SUPPORTED_DIRECTION` on these point estimates. That direction is not a semantic YES or NO. + +Uncertainty uses the bootstrap frozen before the label join: seed 0 and 10000 resamples, with interpolated 2.5 and 97.5 percentiles. The AUC interval is 0.451299 to 0.870130. The mean-difference interval is -0.0155777495 to 0.1373246447. The median-difference interval is -0.0445612032 to 0.1742972287. Each interval includes a null or reversed value. The preregistered status rule does not require the interval to exclude the null, and no stronger AUC floor is added after seeing the scores. + +Tier 3 dependence, among ready rows, is HIGH with Tier 3: 24, HIGH without Tier 3: 4, SECONDARY with Tier 3: 7, and SECONDARY without Tier 3: 4. The no-Tier-3 group has HIGH n=4 and SECONDARY n=4, median difference 0.1435511609, rank-biserial 0.500000, and AUC 0.750000, direction `SUPPORTED_DIRECTION`. The Tier-3 group has HIGH n=24 and SECONDARY n=7, median difference 0.0595389352, rank-biserial 0.083333, and AUC 0.541667, also `SUPPORTED_DIRECTION`. The gap is not concentrated in the GlossBERT subgroup. The within-Tier-3 separation is small. Dropping the largest HIGH residual and the smallest SECONDARY residual does not remove the overall directional result, so the comparison is not extreme-driven. HIGH and SECONDARY are not confined to disjoint parts of speech. The only part of speech with at least three rows in each class is adverb: HIGH 5 and SECONDARY 6, median difference 0.0797654836, AUC 0.733333. Adjective, noun, and verb cells are `NOT_COMPUTABLE`. Whitespace token count has HIGH median 4 and SECONDARY median 3, with Spearman against the residual of 0.092940 for HIGH and 0.603202 for SECONDARY. Character length has HIGH median 17 and SECONDARY median 12, with Spearman -0.059115 and 0.165145. Content-constituent count is 2 for every SECONDARY row and for almost every HIGH row. Maximum candidate-synset count has Spearman 0.390077 for HIGH and 0.073060 for SECONDARY. Where GlossBERT supplies a constituent, minimum confidence and minimum margin have Spearman near 0 for HIGH. These diagnostics do not retune the residual. + +REJECT is reported separately and is not semantic compositionality NO. REJECT count is 34, minimum 0.1036592522, p10 0.1240175180, p25 0.1569692084, median 0.2174387191, p75 0.2995541264, p90 0.3474576871, maximum 0.5180572821, mean 0.2343664094, and sample standard deviation 0.0963240560. The largest HIGH residual is `on the other hand` (`adv:00119578`, row `1d19d96bfedbade746610820599e28166b4ef6c291fd4d186b95055e0cf22074`) at 0.5908804826, tiers GlossBERT then Extended Lesk. The smallest HIGH residual is `naked as a jaybird` (`adj:00458266`, row `d479d3da8dd2b87a7b30c9710abefd1e87f2706a7ac032dde17fc337bef8ac4c`) at 0.1169573790, tiers Extended Lesk then unique lemma. The largest SECONDARY residual is `at one time` (`adv:00153261`, row `0f1c15024e09410a3336e3910351d2cfe6143b0a98b88ab561b8dedcbe64a5fc`) at 0.5587792172, both constituents GlossBERT. The smallest SECONDARY residual is `sneak thief` (`noun:10616204`, row `00152611a0b327fd51c12b1c9a80ab838874fb68692c7e988e7970e41557c9bb`) at 0.1768283745, tiers Extended Lesk then unique lemma. The largest REJECT residual is `three times` (`adv:00476680`, row `0e17c819392a00600e3089202fef9b2e74036b73b6429724dfe91d14f7334b2b`) at 0.5180572821, both Extended Lesk. The smallest REJECT residual is `family ascaphidae` (`noun:01644542`, row `00181b47ec5143e4625569c7b41909401ccb940250f8138c775be0000ec3ff93`) at 0.1036592522, tiers GlossBERT then unique lemma. These rows are descriptive. They do not become phrase rules. + +All six preregistered conditions hold: readiness reproduces, the replay is deterministic, the HIGH median is larger, the rank separation is positive, AUC is above one half, and the separation is not solely one preregistered confound. Residual development state is `RESIDUAL_DEVELOPMENT_ANALYZED_V2`. Candidate status is `CANDIDATE_PROMISING`. `threshold_eligible` stays false because these 225 rows remain a reused development surface. No threshold is frozen. No row receives `semantic_noncompositional` YES or NO. The original residual state stays `CANDIDATE_DISTRIBUTION_FROZEN`. `MODEL_BASED_WSD_CANDIDATE_V1` stays `CANDIDATE_PROMISING`. `RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` stays `DEVELOPMENT_ANALYZED` with lexical status `COVERAGE_INSUFFICIENT`. `selected_source` remains `none`. There is no runtime integration. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. + +No JSON Schema document exists for this replay family. Original residual spec sha256 remains `39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d`. Original residual scores sha256 remains `cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7`. Analysis sha256 `4367930648a68c2f84a1fd8e011fa07d9f3bf111303079f3ec07688f7d5425fb`. Confound analysis sha256 `d0f2496ca7623050eb5519969abcf7c5e2d0e23e0c1961859c40cae4dcdb7021`. Decision sha256 `ed0296fe6888e7c9fe864a2c6c7ab6cecbd4f490d6b7d650e442dafc0bd0976d`. Tracker sha256 `6500394d24543d1797eb9a2a05ba31868e035ffcbc7ae568f53116d4b016e62a`. The 225-row manifest, procedure v2, GlossBERT artifacts, resolver artifacts, MAGPIE artifacts, and Korkontzelos–Manandhar artifacts are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `RESIDUAL_THRESHOLD_PREREGISTRATION_AUTHORIZATION`. This pass does not authorize it. A threshold still requires a later preregistered stage. `selected_source` remains `none`.