diff --git a/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py new file mode 100644 index 00000000..4dd12090 --- /dev/null +++ b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py @@ -0,0 +1,376 @@ +"""Deterministic constituent-sense resolution for a Hyperlex multiword row. + +Structural WordNet evidence outranks contextual overlap. Extended Lesk uses +gloss text only. Operator labels, residual scores, and embeddings are not inputs. +""" + +from __future__ import annotations + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + STRUCTURAL_TOKENS, + neighbor_keys, +) + +RULE_VERSION = "RUNE.CONSTITUENT_SENSE_RESOLUTION.v1" +EXTENDED_LESK = "EXTENDED_LESK_V1" +STRUCTURAL_EXACT = "STRUCTURAL_EXACT" +UNIQUE_LEMMA = "UNIQUE_LEMMA" +METHOD_NONE = "NONE" +EXACT = "EXACT" +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +RESIDUAL_READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +MINIMUM_MARGIN = 1 +RELATION_DEPTH = 1 +LEXICAL_POINTER_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +RELATION_EXPANSION_SYMBOLS = frozenset({"@", "+", "\\", "&", "^", "="}) +READY_STATUSES = frozenset({EXACT, RESOLVED}) + +_RESEARCH_QUESTION = ( + "Can Hyperlex deterministically resolve enough content-constituent WordNet " + "senses to construct a semantic compositional baseline, without using " + "operator labels or residual scores as evidence?" +) +_LESK_SCORE_RULE = ( + "Greedy longest contiguous token overlap. Each match adds the square of its " + "length. Matched tokens are removed. Search restarts until no token matches." +) +_TIE_RULE = ( + "Integer scores. If the top score minus the second score is below the minimum " + "margin, the constituent is AMBIGUOUS. Equal scores are a tie. No sense is chosen." +) +_STOPWORD_RULE = ( + "Tokens whose lookup form is in the residual v1 structural class are removed, " + "as are tokens shorter than two characters. No stemmer and no lemmatizer." +) +_CONTEXT_RULE = ( + "Context is the parent frozen gloss plus the first gloss clause of every other " + "content constituent in the row that structural evidence already resolved. " + "Lesk output is not context." +) +_CANDIDATE_TEXT_RULE = ( + "A candidate contributes its full PWN 3.0 gloss, including quoted examples, " + "then the first gloss clause of each depth-1 related synset." +) +_CIRCULARITY_RULE = ( + "The parent gloss is local context. It is not an embedding target. Parent sense " + "is not constituent sense." +) +_NORMALIZATION_RULE = ( + "casefold, apostrophe fold, split on characters outside letters, digits, apostrophe, and hyphen" +) +_TARGET_ZERO = "does not name a lemma" +_TIER_CONFLICT = "more than one structural target is AMBIGUOUS and blocks Lesk" +_TIER_STRUCTURAL = ( + "one lexical pointer whose target word number is nonzero and whose lemma matches " + "the constituent or one exception hop" +) +_TIER_UNIQUE = "no structural target and one lemma synset across noun, verb, adj, and adv" +_TIER_UNRESOLVED = "no structural target and no lemma synset" +_READY_REQUIRES = "at_least_two_content_constituents_and_each_is_EXACT_or_RESOLVED" +_EMPTY_ROW = ( + "Every content constituent is resolved. A row with fewer than two content " + "constituents stays UNKNOWN. A row with none emits one abstention record." +) + + +def resolver_policy() -> dict: + """Return the preregistered resolver. The dict has no development counts.""" + return { + "applied_at_freeze": False, + "candidate_text": _CANDIDATE_TEXT_RULE, + "circularity": _CIRCULARITY_RULE, + "context": _CONTEXT_RULE, + "empty_row": _EMPTY_ROW, + "encoded_at_freeze": False, + "extended_lesk": { + "examples_included_for_candidate": True, + "examples_included_for_related_synsets": False, + "method": EXTENDED_LESK, + "minimum_margin": MINIMUM_MARGIN, + "minimum_token_length": 2, + "normalization": _NORMALIZATION_RULE, + "relation_depth": RELATION_DEPTH, + "relation_symbols": sorted(RELATION_EXPANSION_SYMBOLS), + "related_gloss": "first_clause", + "score": _LESK_SCORE_RULE, + "stemming": False, + "stopwords": _STOPWORD_RULE, + "structural_tokens": sorted(STRUCTURAL_TOKENS), + "tie": _TIE_RULE, + "zero_overlap_is_a_tie": True, + }, + "forbidden_inputs": [ + "embedding_similarity", + "language_model_judge", + "manual_selection", + "operator_labels", + "phrase_exceptions", + "residual_scores", + "unbind_bucket", + ], + "measurement_eligible": False, + "research_question": _RESEARCH_QUESTION, + "row_rule": { + "ready": RESIDUAL_READY, + "ready_requires": _READY_REQUIRES, + "unknown": ROW_UNKNOWN, + }, + "rule": RULE_VERSION, + "state_at_freeze": "SPEC_FROZEN", + "status_rule": { + "baseline_ambiguous_not_resolved_gt_half": { + "finding": "CONSTITUENT_WSD_COVERAGE_INSUFFICIENT", + "next_legal_transition": "MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION", + "status": "COVERAGE_INSUFFICIENT", + }, + "else_if_residual_ready_high_and_secondary_positive": { + "finding": None, + "next_legal_transition": "RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION", + "status": "COVERAGE_NECESSARY_CONDITION_MET", + }, + "else": { + "finding": None, + "next_legal_transition": "CONSTITUENT_SENSE_RESOLUTION_REVISION_AUTHORIZATION", + "status": "NECESSARY_CONDITION_UNMET", + }, + }, + "structural_pointers": sorted(LEXICAL_POINTER_SYMBOLS), + "target_word_number_zero": _TARGET_ZERO, + "tier1": { + "conflict": _TIER_CONFLICT, + "structural": _TIER_STRUCTURAL, + "unique_lemma": _TIER_UNIQUE, + "unresolved": _TIER_UNRESOLVED, + }, + } + + +def lesk_tokens(text: str) -> list[str]: + """Normalize gloss text into overlap tokens.""" + folded = text.casefold().replace("\u2019", "'").replace("\u2018", "'").replace("`", "'") + tokens = [] + current = [] + for character in folded: + if character.isascii() and (character.isalnum() or character in "'-"): + current.append(character) + continue + if current: + tokens.append("".join(current).strip("'-")) + current = [] + if current: + tokens.append("".join(current).strip("'-")) + kept = [] + for token in tokens: + if len(token) < 2 or lookup_key(token) in STRUCTURAL_TOKENS: + continue + kept.append(token) + return kept + + +def longest_overlap(context: list[str], candidate: list[str]) -> tuple[int, int, int] | None: + """Return the leftmost longest contiguous match as context start, candidate start, length.""" + best_length = 0 + best = None + for context_start, token in enumerate(context): + for candidate_start, other in enumerate(candidate): + if other != token: + continue + length = 0 + while ( + context_start + length < len(context) + and candidate_start + length < len(candidate) + and context[context_start + length] == candidate[candidate_start + length] + ): + length += 1 + if length > best_length: + best_length = length + best = (context_start, candidate_start, length) + return best + + +def overlap_score(context: list[str], candidate: list[str]) -> int: + """Score one candidate. The same token cannot support two matches.""" + left = list(context) + right = list(candidate) + score = 0 + while left and right: + found = longest_overlap(left, right) + if found is None: + break + context_start, candidate_start, length = found + score += length * length + del left[context_start : context_start + length] + del right[candidate_start : candidate_start + length] + return score + + +def structural_synset_ids( + pointers: list[tuple[str, int, str, str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect lexical-pointer targets that name this constituent.""" + keys = neighbor_keys(constituent, exceptions) + found = [] + for symbol, target_word, synset_id, lemma in pointers: + if symbol not in LEXICAL_POINTER_SYMBOLS or target_word <= 0 or not lemma: + continue + if lookup_key(lemma) not in keys: + continue + found.append(synset_id) + return found + + +def constituent_pos(selected_synset: str | None, candidate_synsets: list[str]) -> str | None: + if selected_synset: + return selected_synset.split(":", 1)[0] + poses = {synset_id.split(":", 1)[0] for synset_id in candidate_synsets} + if len(poses) == 1: + return next(iter(poses)) + return None + + +def resolve_constituent_sense( + structural_ids: list[str], + lexical_ids: list[str], + lesk_scores: list[tuple[str, int]] | None, +) -> dict: + """Resolve one constituent. Structural evidence blocks contextual overlap.""" + structural = list(dict.fromkeys(structural_ids)) + lexical = list(dict.fromkeys(lexical_ids)) + if len(structural) == 1: + return _decision( + EXACT, + STRUCTURAL_EXACT, + structural[0], + None, + None, + None, + "structural_exact_pointer", + [f"pointer_target:{structural[0]}"], + ) + if len(structural) > 1: + return _decision( + AMBIGUOUS, + STRUCTURAL_EXACT, + None, + None, + None, + None, + "structural_pointer_conflict", + [f"pointer_target:{synset_id}" for synset_id in structural], + ) + if len(lexical) == 1: + return _decision( + EXACT, + UNIQUE_LEMMA, + lexical[0], + None, + None, + None, + "unique_lemma_synset", + ["lexical_candidates:1"], + ) + if not lexical: + return _decision( + UNRESOLVED, + METHOD_NONE, + None, + None, + None, + None, + "no_candidate_synset", + [], + ) + if lesk_scores is None: + raise RuntimeError("ambiguous constituent has no Lesk scores") + scored = {synset_id: score for synset_id, score in lesk_scores} + if set(scored) != set(lexical) or len(lesk_scores) != len(scored): + raise RuntimeError("Lesk scores do not match the candidate synsets") + if any(not isinstance(score, int) for score in scored.values()): + raise RuntimeError("Lesk scores must be integers") + ordered = sorted(scored.items(), key=lambda item: (-item[1], item[0])) + top_id, top_score = ordered[0] + _second_id, second_score = ordered[1] + margin = top_score - second_score + if margin < MINIMUM_MARGIN: + code = "extended_lesk_tie" if margin == 0 else "extended_lesk_margin" + return _decision( + AMBIGUOUS, + EXTENDED_LESK, + None, + top_score, + second_score, + margin, + code, + [f"lesk_candidates:{len(ordered)}"], + ) + return _decision( + RESOLVED, + EXTENDED_LESK, + top_id, + top_score, + second_score, + margin, + "extended_lesk_margin", + [f"lesk_candidates:{len(ordered)}"], + ) + + +def _decision(status, method, selected, top_score, second_score, margin, code, support) -> dict: + return { + "margin": margin, + "primary_evidence_code": code, + "resolution_method": method, + "resolution_status": status, + "second_score": second_score, + "selected_synset": selected, + "supporting_evidence": list(support), + "top_score": top_score, + } + + +def row_resolution_status(content_count: int, statuses: list[str]) -> str: + if content_count < 2 or len(statuses) != content_count: + return ROW_UNKNOWN + if all(status in READY_STATUSES for status in statuses): + return RESIDUAL_READY + return ROW_UNKNOWN + + +def coverage_decision( + *, + baseline_ambiguous: int, + baseline_ambiguous_not_resolved: int, + residual_ready_high: int, + residual_ready_secondary: int, +) -> dict: + """Apply the preregistered coverage rule. The counts are not a threshold search.""" + if min(baseline_ambiguous, baseline_ambiguous_not_resolved, residual_ready_high, residual_ready_secondary) < 0: + raise RuntimeError("negative coverage count") + insufficient = baseline_ambiguous > 0 and baseline_ambiguous_not_resolved * 2 > baseline_ambiguous + necessary = residual_ready_high > 0 and residual_ready_secondary > 0 + if insufficient: + status = "COVERAGE_INSUFFICIENT" + finding = "CONSTITUENT_WSD_COVERAGE_INSUFFICIENT" + transition = "MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION" + elif necessary: + status = "COVERAGE_NECESSARY_CONDITION_MET" + finding = None + transition = "RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION" + else: + status = "NECESSARY_CONDITION_UNMET" + finding = None + transition = "CONSTITUENT_SENSE_RESOLUTION_REVISION_AUTHORIZATION" + return { + "candidate_status": status, + "coverage_finding": finding, + "necessary_condition_met": necessary, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "DEVELOPMENT_ANALYZED", + } diff --git a/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py new file mode 100644 index 00000000..efd0deed --- /dev/null +++ b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py @@ -0,0 +1,685 @@ +"""Review residual-v1 coverage, then resolve constituent senses on the 225 rows. + +The resolver specification is written before any constituent is resolved. Operator +labels are read only after the resolution artifact is hashed. Residual scores are +not recomputed. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter, defaultdict +from pathlib import Path + +from hyperlexical.constituent_sense_resolution_v1 import ( + RELATION_EXPANSION_SYMBOLS, + RULE_VERSION, + coverage_decision, + lesk_tokens, + overlap_score, + resolve_constituent_sense, + resolver_policy, + row_resolution_status, + structural_synset_ids, +) +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + neighbor_keys, + resolve_constituent, +) +from hyperlexical.semantic_compositionality_residual_replay import EXPECTED as RESIDUAL_EXPECTED +from hyperlexical.unbind_sense_screen_v1 import load_exceptions, load_wordnet, parse_data_line + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +REVIEW_PATH = SOURCE / "RESIDUAL_V1_DEVELOPMENT_RESULT_REVIEW.json" +LIMITATION_PATH = SOURCE / "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION.json" +SPEC_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_SPEC.json" +PROCEDURE_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_PROCEDURE.json" +REPLAY_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_REPLAY.jsonl" +ANALYSIS_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_DEVELOPMENT_ANALYSIS.json" +DECISION_PATH = SOURCE / "CONSTITUENT_SENSE_RESOLUTION_V1_DECISION.json" + +RESIDUAL_SPEC = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +RESIDUAL_PROVENANCE = SOURCE / "RESIDUAL_SOURCE_PROVENANCE.json" +RESIDUAL_SCORES = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +RESIDUAL_EVALUATION = SOURCE / "RESIDUAL_DEVELOPMENT_EVALUATION.json" +RESIDUAL_DECISION = SOURCE / "RESIDUAL_CANDIDATE_DECISION.json" +RESIDUAL_LICENSE = SOURCE / "RESIDUAL_LICENSE_RECEIPT.json" + +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +OPERATOR_COUNTS = {"HIGH": 101, "SECONDARY": 46, "REJECT": 77, "QUARANTINE": 1, "UNRESOLVED": 0} +POS_NAME = {"n": "noun", "v": "verb", "a": "adj", "r": "adv", "s": "adj"} +FILES = {"noun": "data.noun", "verb": "data.verb", "adj": "data.adj", "adv": "data.adv"} +BASELINE_COUNTS = {"AMBIGUOUS": 370, "EXACT": 4, "UNIQUE": 60, "UNRESOLVED": 70} +ROW_FIELDS = ("row_id", "surface", "pos", "gloss", "synset_offset", "synset_pos", "sense_class") + +EXPECTED = dict(RESIDUAL_EXPECTED) +EXPECTED.update( + { + TRACKER: "14d4daaab1e4740a596acaef7f4dbd9ed11edaf2182ff22f7aa339325febf591", + RESIDUAL_SPEC: "39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d", + RESIDUAL_PROVENANCE: "2c34a7d0d480cde564bda694dbaa349550814c5fb7c647bfa3bbbc9db5e26886", + RESIDUAL_SCORES: "cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7", + RESIDUAL_EVALUATION: "ba782622d4c68d23c53ae0ffb5f54f1e43c8cf059b44b7adb34e0eb356ef3896", + RESIDUAL_DECISION: "45922eba294b0b7d7238ce71c6157641259ea292f65743ba2ee1324ff9b52617", + RESIDUAL_LICENSE: "323a5bad21ac74d6a1fd94c54dba95a0046b633cd7328dd6680972525abc0909", + HYPERLEX / "scripts/shadow/hyperlexical/semantic_compositionality_residual.py": "795b433413e86f05ea91186cfa85a68915f18bc6515e70a5ce5cfa6e1b8847d8", + HYPERLEX / "scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py": "36091d4cde5d7580d66ca2f9d8c2a568c996d10cef5d294ba1ed736505ff21ee", + HYPERLEX / "tests/shadow/test_semantic_compositionality_residual.py": "13df0dab6867f9adcf94605df7da5e779696f5e140ae1386c3d0b6600de6c07f", + } +) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def load_sealed_rows() -> list[dict]: + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if len(manifest["rows"]) != 225: + refuse("manifest row count drifted") + sealed = [] + for row in manifest["rows"]: + if row.get("sense_class") is not None: + refuse("development row has a sense class") + if row.get("pos") != row.get("synset_pos"): + refuse("row POS and synset POS differ") + sealed.append({field: row[field] for field in ROW_FIELDS}) + return sealed + + +def build_catalog() -> tuple[dict, dict]: + synsets, first_glosses = load_wordnet(WORDNET) + full = {} + for pos, name in FILES.items(): + for line in (WORDNET / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parsed = parse_data_line(line) + if parsed is None: + continue + synset, _first = parsed + full[(pos, synset.offset)] = " ".join(line.partition("|")[2].split()) + by_id = {} + index = defaultdict(list) + for (pos, offset), synset in synsets.items(): + if pos not in FILES: + continue + synset_id = f"{pos}:{offset}" + if synset_id in by_id: + continue + if (pos, offset) not in full: + refuse(f"missing gloss {synset_id}") + by_id[synset_id] = { + "gloss_first": " ".join(first_glosses[(pos, offset)].split()), + "gloss_full": full[(pos, offset)], + "lemmas": list(synset.lemmas), + "pointers": list(synset.pointers), + "pos": pos, + "synset_id": synset_id, + } + for lemma in synset.lemmas: + index[lookup_key(lemma)].append(synset_id) + for key, identifiers in index.items(): + index[key] = sorted(set(identifiers)) + return by_id, dict(index) + + +def pointer_tuples(record: dict | None, by_id: dict) -> list[tuple[str, int, str, str]]: + if record is None: + return [] + rows = [] + for pointer in record["pointers"]: + pos = POS_NAME.get(pointer.pos) + target_id = f"{pos}:{pointer.offset}" if pos else f"unknown:{pointer.offset}" + target = by_id.get(target_id) + if target is None or pointer.target < 1 or pointer.target > len(target["lemmas"]): + lemma = "" + else: + lemma = target["lemmas"][pointer.target - 1] + rows.append((pointer.symbol, pointer.target, target_id, lemma)) + return rows + + +def related_clauses(record: dict, by_id: dict) -> list[str]: + found = [] + seen = set() + ordered = sorted( + record["pointers"], + key=lambda pointer: (pointer.symbol, pointer.pos, pointer.offset, pointer.source, pointer.target), + ) + for pointer in ordered: + if pointer.symbol not in RELATION_EXPANSION_SYMBOLS: + continue + pos = POS_NAME.get(pointer.pos) + if pos is None: + continue + target_id = f"{pos}:{pointer.offset}" + if target_id in seen or target_id == record["synset_id"]: + continue + target = by_id.get(target_id) + if target is None: + continue + seen.add(target_id) + found.append(target["gloss_first"]) + return found + + +def candidate_tokens(record: dict, by_id: dict) -> list[str]: + tokens = lesk_tokens(record["gloss_full"]) + for clause in related_clauses(record, by_id): + tokens.extend(lesk_tokens(clause)) + return tokens + + +def lemma_supported(record: dict, constituent: str, exceptions: dict[str, set[str]]) -> bool: + keys = neighbor_keys(constituent, exceptions) + return any(lookup_key(lemma) in keys for lemma in record["lemmas"]) + + +def resolve_once(rows: list[dict], by_id: dict, index: dict, exceptions: dict[str, set[str]], spec_sha: str, procedure_sha: str) -> list[dict]: + records = [] + for row in rows: + parent_synset = f"{row['synset_pos']}:{row['synset_offset']}" + parent = by_id.get(parent_synset) + pointers = pointer_tuples(parent, by_id) + extraction = extract_constituents(row["surface"]) + content = extraction["content_constituents"] + if not content: + records.append( + { + "baseline_resolution_status": None, + "candidate_glosses": [], + "candidate_synsets": [], + "constituent_index": None, + "constituent_pos": None, + "constituent_surface": None, + "margin": None, + "parent_row_id": row["row_id"], + "parent_surface": row["surface"], + "parent_synset": parent_synset, + "primary_evidence_code": "fewer_than_two_content_constituents", + "procedure_sha256": procedure_sha, + "resolution_method": "NONE", + "resolution_status": None, + "row_resolution_status": "UNKNOWN", + "second_score": None, + "selected_synset": None, + "spec_sha256": spec_sha, + "supporting_evidence": [], + "top_score": None, + } + ) + continue + prepared = [] + for index_in_row, constituent in enumerate(content): + structural = structural_synset_ids(pointers, constituent, exceptions) + lexical = lexical_synset_ids(index, constituent, exceptions) + baseline = resolve_constituent( + exact_synset_ids(pointers, constituent, exceptions), + lexical, + ) + prepared.append( + { + "baseline": baseline, + "constituent": constituent, + "lexical": lexical, + "structural": structural, + } + ) + exact_gloss = {} + for index_in_row, item in enumerate(prepared): + structural = list(dict.fromkeys(item["structural"])) + lexical = list(dict.fromkeys(item["lexical"])) + selected = structural[0] if len(structural) == 1 else lexical[0] if not structural and len(lexical) == 1 else None + if selected is None: + continue + record = by_id.get(selected) + if record is None or not lemma_supported(record, item["constituent"], exceptions): + refuse(f"structural selection is not a constituent lemma: {selected}") + exact_gloss[index_in_row] = record["gloss_first"] + statuses = [] + emitted = [] + for index_in_row, item in enumerate(prepared): + structural = list(dict.fromkeys(item["structural"])) + lexical = list(dict.fromkeys(item["lexical"])) + lesk_scores = None + support_extra = [] + if not structural and len(lexical) > 1: + others = [exact_gloss[other] for other in range(len(prepared)) if other != index_in_row and other in exact_gloss] + context = lesk_tokens(row["gloss"]) + for clause in others: + context.extend(lesk_tokens(clause)) + lesk_scores = [] + for synset_id in lexical: + candidate = by_id.get(synset_id) + if candidate is None: + refuse(f"missing candidate synset {synset_id}") + if not lemma_supported(candidate, item["constituent"], exceptions): + refuse(f"candidate does not match constituent {synset_id}") + lesk_scores.append((synset_id, overlap_score(context, candidate_tokens(candidate, by_id)))) + support_extra.append(f"context_tokens:{len(context)}") + decision = resolve_constituent_sense(structural, lexical, lesk_scores) + if decision["selected_synset"] is not None: + chosen = by_id.get(decision["selected_synset"]) + if chosen is None or not lemma_supported(chosen, item["constituent"], exceptions): + refuse(f"selected synset fails the lemma check: {decision['selected_synset']}") + candidates = sorted(set(structural) | set(lexical)) + glosses = [] + for synset_id in candidates: + candidate = by_id.get(synset_id) + if candidate is None: + refuse(f"missing candidate gloss {synset_id}") + glosses.append(candidate["gloss_full"]) + poses = {synset_id.split(":", 1)[0] for synset_id in candidates} + if decision["selected_synset"]: + pos = decision["selected_synset"].split(":", 1)[0] + elif len(poses) == 1: + pos = next(iter(poses)) + else: + pos = None + statuses.append(decision["resolution_status"]) + support = list(decision["supporting_evidence"]) + support_extra + emitted.append( + { + "baseline_resolution_status": item["baseline"], + "candidate_glosses": glosses, + "candidate_synsets": candidates, + "constituent_index": index_in_row, + "constituent_pos": pos, + "constituent_surface": item["constituent"], + "margin": decision["margin"], + "parent_row_id": row["row_id"], + "parent_surface": row["surface"], + "parent_synset": parent_synset, + "primary_evidence_code": decision["primary_evidence_code"], + "procedure_sha256": procedure_sha, + "resolution_method": decision["resolution_method"], + "resolution_status": decision["resolution_status"], + "second_score": decision["second_score"], + "selected_synset": decision["selected_synset"], + "spec_sha256": spec_sha, + "supporting_evidence": support, + "top_score": decision["top_score"], + } + ) + status = row_resolution_status(len(content), statuses) + for item in emitted: + item["row_resolution_status"] = status + records.append(item) + return records + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def main() -> None: + check_sealed() + evaluation = json.loads(RESIDUAL_EVALUATION.read_text(encoding="utf-8")) + if evaluation["scored_count"] != 3 or evaluation["unknown_count"] != 222: + refuse("frozen residual evaluation counts drifted") + if evaluation["high_comparison"] != "NOT_COMPUTABLE": + refuse("frozen HIGH comparison drifted") + if evaluation["semantic_noncompositionality_threshold"] is not None or evaluation["emits_yes_no"] is not False: + refuse("frozen residual evaluation carries a threshold or a yes/no claim") + observed = { + "ambiguous_content_rows": 145, + "fewer_than_two_content_tokens": 59, + "scored_HIGH": 0, + "scored_REJECT": 3, + "scored_SECONDARY": 0, + "scored_rows": 3, + "unknown_rows": 222, + "unresolved_content_rows": 18, + } + if evaluation["abstention_reason_counts"] != { + "ambiguous_content_constituent": 145, + "fewer_than_two_content_constituents": 59, + "unresolved_content_constituent": 18, + }: + refuse("frozen abstention counts drifted") + limitation = { + "conclusion": { + "primary_limitation": "constituent_sense_resolution", + "source_selection_eligible": False, + "threshold_eligible": False, + }, + "finding": "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED", + "interpretation": "The dominant blocker for semantic-compositionality residual evaluation is deterministic constituent sense resolution, not the residual model itself.", + "json_schema_document": None, + "observed": observed, + "residual_candidate_decision_sha256": EXPECTED[RESIDUAL_DECISION], + "residual_candidate_spec_sha256": EXPECTED[RESIDUAL_SPEC], + "residual_development_evaluation_sha256": EXPECTED[RESIDUAL_EVALUATION], + "residual_development_scores_sha256": EXPECTED[RESIDUAL_SCORES], + "residual_model_name": "sentence-transformers/all-MiniLM-L6-v2", + "residual_model_revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", + "residual_source_provenance_sha256": EXPECTED[RESIDUAL_PROVENANCE], + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "schema": "hyperlex.semantic_residual_v1_coverage_limitation.v1", + "threshold_authorization": "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + "threshold_authorization_granted": False, + } + limitation_sha = write_json(LIMITATION_PATH, limitation) + review = { + "authorized_transition": "RESIDUAL_DEVELOPMENT_RESULT_REVIEW", + "coverage_limitation": "CONFIRMED", + "coverage_limitation_sha256": limitation_sha, + "finding": "SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED", + "json_schema_document": None, + "next_research_track": RULE_VERSION, + "next_research_track_state_at_review": "SPEC_FROZEN", + "residual_artifacts_modified": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "schema": "hyperlex.residual_v1_development_result_review.v1", + "selected_source": "none", + "source_selection_eligible": False, + "threshold_authorization_granted": False, + "threshold_eligible": False, + } + review_sha = write_json(REVIEW_PATH, review) + policy = resolver_policy() + spec = { + "coverage_limitation_sha256": limitation_sha, + "development_result_review_sha256": review_sha, + "json_schema_document": None, + "policy": policy, + "residual_model_unchanged": { + "composition": "normalized_mean_v1", + "distance": "one_minus_cosine_v1", + "model_name": "sentence-transformers/all-MiniLM-L6-v2", + "model_revision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", + }, + "residual_replay_authorized": False, + "residual_scores_included": False, + "schema": "hyperlex.constituent_sense_resolution_v1_spec.v1", + "selected_source": "none", + } + spec_text = json.dumps(spec, sort_keys=True) + if "operator_bucket" in spec_text or "0.2139784896" in spec_text: + refuse("resolver spec contains a label or a residual score") + spec_sha = write_json(SPEC_PATH, spec) + procedure = { + "json_schema_document": None, + "policy": policy, + "rule": RULE_VERSION, + "schema": "hyperlex.constituent_sense_resolution_v1_procedure.v1", + "spec_sha256": spec_sha, + } + procedure_sha = write_json(PROCEDURE_PATH, procedure) + rows = load_sealed_rows() + exceptions = load_exceptions(WORDNET) + first_catalog = build_catalog() + second_catalog = build_catalog() + first = resolve_once(rows, first_catalog[0], first_catalog[1], exceptions, spec_sha, procedure_sha) + second = resolve_once(rows, second_catalog[0], second_catalog[1], exceptions, spec_sha, procedure_sha) + first_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in first) + second_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in second) + if first_text != second_text: + refuse("NOT_DETERMINISTIC") + baseline_counts = Counter( + row["baseline_resolution_status"] for row in first if row["constituent_index"] is not None + ) + for name, expected_count in BASELINE_COUNTS.items(): + if baseline_counts[name] != expected_count: + refuse(f"baseline {name} is {baseline_counts[name]}") + if any("operator_bucket" in row for row in first): + refuse("resolution row carries an operator bucket") + replay_sha = write_jsonl(REPLAY_PATH, first) + if sha256(SPEC_PATH) != spec_sha or sha256(PROCEDURE_PATH) != procedure_sha: + refuse("resolution mutated the frozen spec") + + rejoined = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during resolution") + buckets = {row["row_id"]: row["operator_bucket"] for row in rejoined["rows"]} + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + attempts = [row for row in first if row["constituent_index"] is not None] + status_counts = Counter(row["resolution_status"] for row in attempts) + method_counts = Counter(row["resolution_method"] for row in attempts) + code_counts = Counter(row["primary_evidence_code"] for row in attempts) + baseline_ambiguous_rows = [row for row in attempts if row["baseline_resolution_status"] == "AMBIGUOUS"] + not_resolved = sum(row["resolution_status"] in {"AMBIGUOUS", "UNRESOLVED"} for row in baseline_ambiguous_rows) + lesk_attempts = [row for row in attempts if row["resolution_method"] == "EXTENDED_LESK_V1"] + ties = sum(row["primary_evidence_code"] == "extended_lesk_tie" for row in lesk_attempts) + row_status = {} + for row in first: + row_status[row["parent_row_id"]] = row["row_resolution_status"] + if len(row_status) != 225: + refuse("row status does not cover 225 rows") + ready_by_operator = {name: {"RESIDUAL_READY": 0, "UNKNOWN": 0} for name in OPERATORS} + for row_id, status in row_status.items(): + ready_by_operator[buckets[row_id]][status] += 1 + ready_total = sum(item["RESIDUAL_READY"] for item in ready_by_operator.values()) + unknown_total = sum(item["UNKNOWN"] for item in ready_by_operator.values()) + outcome = coverage_decision( + baseline_ambiguous=BASELINE_COUNTS["AMBIGUOUS"], + baseline_ambiguous_not_resolved=not_resolved, + residual_ready_high=ready_by_operator["HIGH"]["RESIDUAL_READY"], + residual_ready_secondary=ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + ) + attempt_count = len(attempts) + analysis = { + "abstention_rate_fraction": fraction(status_counts["AMBIGUOUS"] + status_counts["UNRESOLVED"], attempt_count), + "baseline_ambiguous_not_resolved": not_resolved, + "baseline_counts": {name: BASELINE_COUNTS[name] for name in ("EXACT", "UNIQUE", "AMBIGUOUS", "UNRESOLVED")}, + "candidate_status": outcome["candidate_status"], + "change_versus_residual_v1": { + "baseline_AMBIGUOUS": BASELINE_COUNTS["AMBIGUOUS"], + "baseline_EXACT": BASELINE_COUNTS["EXACT"], + "baseline_UNIQUE": BASELINE_COUNTS["UNIQUE"], + "baseline_UNRESOLVED": BASELINE_COUNTS["UNRESOLVED"], + "new_AMBIGUOUS": status_counts["AMBIGUOUS"], + "new_EXACT": status_counts["EXACT"], + "new_RESOLVED": status_counts["RESOLVED"], + "new_UNRESOLVED": status_counts["UNRESOLVED"], + }, + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "determinism": "IDENTICAL", + "development_result_review_sha256": review_sha, + "exact_fraction": fraction(status_counts["EXACT"], attempt_count), + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "high_unknown": ready_by_operator["HIGH"]["UNKNOWN"], + "json_schema_document": None, + "method_counts": dict(sorted(method_counts.items())), + "necessary_condition_met": outcome["necessary_condition_met"], + "operator_labels_joined_after_resolution_artifact_was_hashed": True, + "operator_labels_used_during_resolution": False, + "primary_evidence_code_counts": dict(sorted(code_counts.items())), + "procedure_sha256": procedure_sha, + "projection": { + "REJECT_rows": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "HIGH_rows": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "QUARANTINE_rows": ready_by_operator["QUARANTINE"]["RESIDUAL_READY"], + "SECONDARY_rows": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "residual_replay_candidate_rows": ready_total, + "residual_replay_performed": False, + }, + "quarantine_residual_ready": ready_by_operator["QUARANTINE"]["RESIDUAL_READY"], + "quarantine_unknown": ready_by_operator["QUARANTINE"]["UNKNOWN"], + "reject_residual_ready": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "reject_unknown": ready_by_operator["REJECT"]["UNKNOWN"], + "replay_sha256": replay_sha, + "residual_ready_by_operator": ready_by_operator, + "residual_ready_rows": ready_total, + "residual_scores_recomputed": False, + "resolution_counts": { + "AMBIGUOUS": status_counts["AMBIGUOUS"], + "EXACT": status_counts["EXACT"], + "RESOLVED": status_counts["RESOLVED"], + "UNRESOLVED": status_counts["UNRESOLVED"], + }, + "resolved_fraction": fraction(status_counts["RESOLVED"], attempt_count), + "row_count": 225, + "rule": RULE_VERSION, + "schema": "hyperlex.constituent_sense_resolution_v1_development_analysis.v1", + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "secondary_unknown": ready_by_operator["SECONDARY"]["UNKNOWN"], + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "spec_sha256": spec_sha, + "tie_rate_fraction": fraction(ties, len(lesk_attempts)) if lesk_attempts else "0/0", + "total_constituent_attempts": attempt_count, + "unknown_rows": unknown_total, + "unresolved_rate_fraction": fraction(status_counts["UNRESOLVED"], attempt_count), + "yes_no_emitted": False, + } + analysis_sha = write_json(ANALYSIS_PATH, analysis) + decision = { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "development_result_review_sha256": review_sha, + "encoded": True, + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "necessary_condition_met": outcome["necessary_condition_met"], + "next_legal_transition": outcome["next_legal_transition"], + "next_transition_authorized": False, + "procedure_sha256": procedure_sha, + "replay_sha256": replay_sha, + "residual_replay_performed": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "residual_threshold_created": False, + "rule": RULE_VERSION, + "runtime_integration": False, + "schema": "hyperlex.constituent_sense_resolution_v1_decision.v1", + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "select_005_authorized": False, + "selected_source": "none", + "spec_sha256": spec_sha, + "state": "DEVELOPMENT_ANALYZED", + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, decision) + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_coverage_limitation"] = "CONFIRMED" + tracker["residual_threshold_eligible"] = False + tracker["residual_source_selection_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["residual_development_result_review_sha256"] = review_sha + tracker["residual_coverage_limitation_sha256"] = limitation_sha + tracker["constituent_sense_resolution_rule"] = RULE_VERSION + tracker["constituent_sense_resolution_state"] = "DEVELOPMENT_ANALYZED" + tracker["constituent_sense_resolution_candidate_status"] = outcome["candidate_status"] + tracker["constituent_sense_resolution_coverage_finding"] = outcome["coverage_finding"] + tracker["constituent_sense_resolution_encoded"] = True + tracker["constituent_sense_resolution_runtime_applied"] = False + tracker["constituent_sense_resolution_spec_sha256"] = spec_sha + tracker["constituent_sense_resolution_procedure_sha256"] = procedure_sha + tracker["constituent_sense_resolution_replay_sha256"] = replay_sha + tracker["constituent_sense_resolution_analysis_sha256"] = analysis_sha + tracker["constituent_sense_resolution_decision_sha256"] = decision_sha + tracker["constituent_sense_resolution_residual_ready_rows"] = ready_total + tracker["constituent_sense_resolution_high_ready"] = ready_by_operator["HIGH"]["RESIDUAL_READY"] + tracker["constituent_sense_resolution_secondary_ready"] = ready_by_operator["SECONDARY"]["RESIDUAL_READY"] + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = outcome["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "coverage_finding": outcome["coverage_finding"], + "coverage_limitation_sha256": limitation_sha, + "decision_sha256": decision_sha, + "determinism": "IDENTICAL", + "high_residual_ready": ready_by_operator["HIGH"]["RESIDUAL_READY"], + "high_unknown": ready_by_operator["HIGH"]["UNKNOWN"], + "method_counts": analysis["method_counts"], + "necessary_condition_met": outcome["necessary_condition_met"], + "next_legal_transition": outcome["next_legal_transition"], + "procedure_sha256": procedure_sha, + "projection": analysis["projection"], + "reject_residual_ready": ready_by_operator["REJECT"]["RESIDUAL_READY"], + "replay_sha256": replay_sha, + "residual_ready_rows": ready_total, + "resolution_counts": analysis["resolution_counts"], + "review_sha256": review_sha, + "secondary_residual_ready": ready_by_operator["SECONDARY"]["RESIDUAL_READY"], + "secondary_unknown": ready_by_operator["SECONDARY"]["UNKNOWN"], + "spec_sha256": spec_sha, + "tie_rate_fraction": analysis["tie_rate_fraction"], + "total_constituent_attempts": attempt_count, + "tracker_sha256": tracker_sha, + "unknown_rows": unknown_total, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/km_candidate_evaluation.py b/scripts/shadow/hyperlexical/km_candidate_evaluation.py new file mode 100644 index 00000000..10ed0031 --- /dev/null +++ b/scripts/shadow/hyperlexical/km_candidate_evaluation.py @@ -0,0 +1,142 @@ +"""Development-only alignment for the Korkontzelos–Manandhar candidate. + +A compositionality label supports semantic evidence only when the item +is tied to one PWN 3.0 synset by a source identifier or by a unique +lemma reconstruction. Multiple synsets stay unresolved. Gloss text, +operator labels, and surface overlap are not tie breaks. +""" + +from __future__ import annotations + +EXACT_SOURCE_ID = "EXACT_SOURCE_ID" +EXACT_UNIQUE_RECONSTRUCTION = "EXACT_UNIQUE_RECONSTRUCTION" +AMBIGUOUS_MULTIPLE_SYNSETS = "AMBIGUOUS_MULTIPLE_SYNSETS" +NO_PWN3_MATCH = "NO_PWN3_MATCH" +VERSION_CONFLICT = "VERSION_CONFLICT" +UNKNOWN = "UNKNOWN" + +SUPPORTING = frozenset({EXACT_SOURCE_ID, EXACT_UNIQUE_RECONSTRUCTION}) + +YES = "YES" +NO = "NO" + +CODE_NONCOMPOSITIONAL = "km_exact_noncompositional" +CODE_COMPOSITIONAL = "km_exact_compositional" +CODE_AMBIGUOUS = "km_ambiguous_synset" +CODE_NONE = "km_no_match" +CODE_VERSION = "km_version_conflict" +CODE_UNRESOLVED = "km_source_id_unresolved" + +NONCOMPOSITIONAL = "NONCOMPOSITIONAL" +COMPOSITIONAL = "COMPOSITIONAL" + + +def lookup_key(text: str) -> str: + """Orthographic key for a WordNet lemma. Hyphens stay in the key.""" + folded = text.casefold() + folded = folded.replace("\u2019", "'").replace("\u2018", "'").replace("`", "'") + folded = folded.replace("_", " ") + return "_".join(folded.split()) + + +def alignment_status(source_identifier: str | None, candidate_synset_ids: list[str]) -> str: + """Classify sense identity. The candidate list is not ranked.""" + identifiers = list(dict.fromkeys(candidate_synset_ids)) + if source_identifier: + if len(identifiers) == 1 and identifiers[0] == source_identifier: + return EXACT_SOURCE_ID + if source_identifier not in identifiers: + return VERSION_CONFLICT + return AMBIGUOUS_MULTIPLE_SYNSETS + if len(identifiers) == 1: + return EXACT_UNIQUE_RECONSTRUCTION + if not identifiers: + return NO_PWN3_MATCH + return AMBIGUOUS_MULTIPLE_SYNSETS + + +def semantic_noncompositional(source_label: str, status: str) -> str: + if status not in SUPPORTING: + return UNKNOWN + if source_label == NONCOMPOSITIONAL: + return YES + if source_label == COMPOSITIONAL: + return NO + return UNKNOWN + + +def evidence_code(status: str, semantic: str) -> str: + if semantic == YES and status in SUPPORTING: + return CODE_NONCOMPOSITIONAL + if semantic == NO and status in SUPPORTING: + return CODE_COMPOSITIONAL + if status == AMBIGUOUS_MULTIPLE_SYNSETS: + return CODE_AMBIGUOUS + if status == VERSION_CONFLICT: + return CODE_VERSION + if status == UNKNOWN: + return CODE_UNRESOLVED + return CODE_NONE + + +def align_item( + source_label: str, + source_identifier: str | None, + candidate_synset_ids: list[str], +) -> dict: + status = alignment_status(source_identifier, candidate_synset_ids) + semantic = semantic_noncompositional(source_label, status) + aligned = candidate_synset_ids[0] if status in SUPPORTING and len(set(candidate_synset_ids)) == 1 else None + if status in SUPPORTING and aligned is None: + raise RuntimeError("supporting alignment has no single synset") + if status not in SUPPORTING and semantic != UNKNOWN: + raise RuntimeError("non-aligned item assigned semantic evidence") + return { + "aligned_pwn30_synset": aligned, + "alignment_status": status, + "primary_evidence_code": evidence_code(status, semantic), + "semantic_noncompositional": semantic, + } + + +def join_synset(synset: str, exact: dict[str, dict], ambiguous: dict[str, list[str]]) -> dict: + """Join a Hyperlex synset. The surface is not an argument.""" + exact_item = exact.get(synset) + ambiguous_ids = list(ambiguous.get(synset, [])) + if exact_item is not None and ambiguous_ids: + status = AMBIGUOUS_MULTIPLE_SYNSETS + semantic = UNKNOWN + item_id = None + aligned = None + elif exact_item is not None: + status = exact_item["alignment_status"] + semantic = exact_item["semantic_noncompositional"] + item_id = exact_item["source_item_id"] + aligned = exact_item["aligned_pwn30_synset"] + elif len(ambiguous_ids) == 1: + status = AMBIGUOUS_MULTIPLE_SYNSETS + semantic = UNKNOWN + item_id = ambiguous_ids[0] + aligned = None + elif ambiguous_ids: + status = AMBIGUOUS_MULTIPLE_SYNSETS + semantic = UNKNOWN + item_id = None + aligned = None + else: + status = UNKNOWN + semantic = UNKNOWN + item_id = None + aligned = None + code = CODE_NONE + if semantic != UNKNOWN and status not in SUPPORTING: + raise RuntimeError("join assigned semantic evidence without exact alignment") + if status != UNKNOWN: + code = evidence_code(status, semantic) + return { + "alignment_status": status, + "candidate_aligned_synset": aligned, + "candidate_source_item_id": item_id, + "primary_evidence_code": code, + "semantic_noncompositional": semantic, + } diff --git a/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py b/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py new file mode 100644 index 00000000..16fc450e --- /dev/null +++ b/scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py @@ -0,0 +1,553 @@ +"""Evaluate the 2009 Korkontzelos–Manandhar Table 1 on the 225 development rows. + +The pass does not select the source, integrate it, or draw a measurement sample. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter, defaultdict +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.km_candidate_evaluation import align_item, join_synset, lookup_key +from hyperlexical.unbind_sense_screen_v1 import load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +PDF = LEDGER / "acquisition/sources/korkontzelos-manandhar-2009/P09-2017.pdf" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "KM_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "KM_LICENSE_RECEIPT.json" +INVENTORY_PATH = SOURCE / "KM_RAW_EVALUATION_INVENTORY.jsonl" +ALIGNMENT_PATH = SOURCE / "KM_PWN30_ALIGNMENT.jsonl" +SEMANTIC_PATH = SOURCE / "KM_HYPERLEX_SEMANTIC_EVIDENCE.jsonl" +EVALUATION_PATH = SOURCE / "KM_DEVELOPMENT_EVALUATION.json" +COMPARISON_PATH = SOURCE / "KM_MAGPIE_COMPARISON.json" +DECISION_PATH = SOURCE / "KM_CANDIDATE_DECISION.json" + +PUBLICATION = "P09-2017" +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +SEMANTIC_STATES = ("YES", "NO", "UNKNOWN") +ALIGNMENT_STATES = ( + "EXACT_SOURCE_ID", + "EXACT_UNIQUE_RECONSTRUCTION", + "AMBIGUOUS_MULTIPLE_SYNSETS", + "NO_PWN3_MATCH", + "VERSION_CONFLICT", + "UNKNOWN", +) +TABLE = ( + ("KM2009-NC-01", "NONCOMPOSITIONAL", "agony aunt"), + ("KM2009-NC-02", "NONCOMPOSITIONAL", "black maria"), + ("KM2009-NC-03", "NONCOMPOSITIONAL", "dead end"), + ("KM2009-NC-04", "NONCOMPOSITIONAL", "dutch oven"), + ("KM2009-NC-05", "NONCOMPOSITIONAL", "fish finger"), + ("KM2009-NC-06", "NONCOMPOSITIONAL", "fool\u2019s paradise"), + ("KM2009-NC-07", "NONCOMPOSITIONAL", "goat\u2019s rue"), + ("KM2009-NC-08", "NONCOMPOSITIONAL", "green light"), + ("KM2009-NC-09", "NONCOMPOSITIONAL", "high jump"), + ("KM2009-NC-10", "NONCOMPOSITIONAL", "joint chiefs"), + ("KM2009-NC-11", "NONCOMPOSITIONAL", "lip service"), + ("KM2009-NC-12", "NONCOMPOSITIONAL", "living rock"), + ("KM2009-NC-13", "NONCOMPOSITIONAL", "monkey puzzle"), + ("KM2009-NC-14", "NONCOMPOSITIONAL", "motor pool"), + ("KM2009-NC-15", "NONCOMPOSITIONAL", "prince Albert"), + ("KM2009-NC-16", "NONCOMPOSITIONAL", "stocking stuffer"), + ("KM2009-NC-17", "NONCOMPOSITIONAL", "sweet bay"), + ("KM2009-NC-18", "NONCOMPOSITIONAL", "teddy boy"), + ("KM2009-NC-19", "NONCOMPOSITIONAL", "think tank"), + ("KM2009-CO-01", "COMPOSITIONAL", "box white oak"), + ("KM2009-CO-02", "COMPOSITIONAL", "cartridge brass"), + ("KM2009-CO-03", "COMPOSITIONAL", "common iguana"), + ("KM2009-CO-04", "COMPOSITIONAL", "closed chain"), + ("KM2009-CO-05", "COMPOSITIONAL", "eastern pipistrel"), + ("KM2009-CO-06", "COMPOSITIONAL", "field mushroom"), + ("KM2009-CO-07", "COMPOSITIONAL", "hard candy"), + ("KM2009-CO-08", "COMPOSITIONAL", "king snake"), + ("KM2009-CO-09", "COMPOSITIONAL", "labor camp"), + ("KM2009-CO-10", "COMPOSITIONAL", "lemon tree"), + ("KM2009-CO-11", "COMPOSITIONAL", "life form"), + ("KM2009-CO-12", "COMPOSITIONAL", "parenthesis-free notation"), + ("KM2009-CO-13", "COMPOSITIONAL", "parking brake"), + ("KM2009-CO-14", "COMPOSITIONAL", "petit juror"), + ("KM2009-CO-15", "COMPOSITIONAL", "relational adjective"), + ("KM2009-CO-16", "COMPOSITIONAL", "taxonomic category"), + ("KM2009-CO-17", "COMPOSITIONAL", "telephone service"), + ("KM2009-CO-18", "COMPOSITIONAL", "tea table"), + ("KM2009-CO-19", "COMPOSITIONAL", "upland cotton"), +) + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + SOURCE / "MAGPIE_SOURCE_PROVENANCE.json": "bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd", + SOURCE / "MAGPIE_LICENSE_RECEIPT.json": "8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972", + SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl": "84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83", + SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl": "037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66", + SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl": "89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577", + SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json": "74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba", + SOURCE / "MAGPIE_CANDIDATE_DECISION.json": "6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "c3fe6f2fd21d07b8b45d9f26ffeebbcfbe3cb68a1f5a7952dee94d9a2c995936", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + PDF: "046da9fc26cfdf220e41ad914314f0703ea3c84cd86e045b20146b34189113d1", + WORDNET / "data.noun": "489f145e0f68877c0be5bd0eb4117adaaac52f38f6204eb8d85dbe2158b614cc", + WORDNET / "data.verb": "29cc96ed80c9f47d94fe75e332a9df80f4b1c737205f92d2f433d63c6da2ab51", + WORDNET / "data.adj": "f24b635368be441501c9b8001e9271fd3b30b203f00d91e332979e6f8fe35646", + WORDNET / "data.adv": "e66dbbda0e0359e41b7f225bff71dd0c263dc7c66c1b61abc9ba334973d92979", + WORDNET / "README": "adad8d28ddea1db05b67ba1ac23506b025d29e0bcbf23bb35dde346089d8808d", +} + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def lemma_index(root: Path) -> dict[str, list[dict]]: + synsets, glosses = load_wordnet(root) + grouped: dict[str, dict[str, dict]] = defaultdict(dict) + for (pos, offset), synset in synsets.items(): + if pos not in {"noun", "verb", "adj", "adv"}: + continue + synset_id = f"{pos}:{offset}" + gloss = " ".join(glosses[(pos, offset)].split()) + bucket = grouped[synset_id] + if not bucket: + bucket["gloss"] = gloss + bucket["lemmas"] = [] + bucket["synset"] = synset_id + for lemma in synset.lemmas: + if lemma not in bucket["lemmas"]: + bucket["lemmas"].append(lemma) + by_key: dict[str, list[dict]] = defaultdict(list) + for record in grouped.values(): + record["lemmas"] = sorted(record["lemmas"]) + for lemma in record["lemmas"]: + by_key[lookup_key(lemma)].append(record) + for key, records in by_key.items(): + unique = {item["synset"]: item for item in records} + by_key[key] = [unique[name] for name in sorted(unique)] + return by_key + + +def candidate_status(yes_count: int, yes_exact: int) -> str: + if yes_count == 0: + return "CANDIDATE_INSUFFICIENT" + if yes_exact != yes_count: + return "CANDIDATE_REJECTED" + return "CANDIDATE_PROMISING" + + +def main() -> None: + magpie_module = HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation.py" + if not magpie_module.is_file(): + refuse("MAGPIE evaluator is missing") + for path, expected in EXPECTED.items(): + if expected is None: + continue + if sha256(path) != expected: + refuse(f"sealed artifact changed before evaluation: {path}") + readme = (WORDNET / "README").read_text(encoding="utf-8", errors="replace") + if "WordNet 3.0" not in readme: + refuse("local WordNet README does not identify release 3.0") + labels = Counter(label for _item_id, label, _surface in TABLE) + if labels["NONCOMPOSITIONAL"] != 19 or labels["COMPOSITIONAL"] != 19 or len(TABLE) != 38: + refuse("recovered table does not match the published 19/19 split") + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + pdf_mtime = datetime.fromtimestamp(PDF.stat().st_mtime, timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + by_key = lemma_index(WORDNET) + inventory = [] + alignments = [] + for item_id, label, surface in TABLE: + records = by_key.get(lookup_key(surface), []) + synset_ids = [record["synset"] for record in records] + aligned = align_item(label, None, synset_ids) + inventory.append({ + "schema": "hyperlex.km_raw_evaluation_item.v1", + "source_context": None, + "source_item_id": item_id, + "source_label": label, + "source_page_or_location": "P09-2017 Table 1, proceedings page 67", + "surface": surface, + }) + alignments.append({ + "aligned_pwn30_synset": aligned["aligned_pwn30_synset"], + "alignment_status": aligned["alignment_status"], + "candidate_glosses": [record["gloss"] for record in records], + "candidate_lemmas": sorted({lemma for record in records for lemma in record["lemmas"]}), + "candidate_synsets": synset_ids, + "exact_sense_identity_preserved_by_source": False, + "json_schema_document": None, + "lookup_key": lookup_key(surface), + "primary_evidence_code": aligned["primary_evidence_code"], + "schema": "hyperlex.km_pwn30_alignment_row.v1", + "semantic_noncompositional": aligned["semantic_noncompositional"], + "source_gloss": None, + "source_item_id": item_id, + "source_label": label, + "source_lemma_key": None, + "source_pos": None, + "source_sense_key": None, + "source_synset_identifier": None, + "source_synset_offset": None, + "surface": surface, + "wordnet_version_in_source": "3.0", + }) + exact = {} + ambiguous: dict[str, list[str]] = defaultdict(list) + for row in alignments: + if row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}: + synset = row["aligned_pwn30_synset"] + if synset in exact: + refuse("two exact items share one synset") + exact[synset] = row + elif row["alignment_status"] == "AMBIGUOUS_MULTIPLE_SYNSETS": + for synset in row["candidate_synsets"]: + ambiguous[synset].append(row["source_item_id"]) + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225 or any(row.get("sense_class") is not None for row in rows): + refuse("development manifest changed") + semantic_rows = [] + surface_key_hits = [] + inventory_keys = {lookup_key(surface) for _item_id, _label, surface in TABLE} + for row in rows: + synset = f"{row['synset_pos']}:{row['synset_offset']}" + joined = join_synset(synset, exact, ambiguous) + if lookup_key(row["surface"]) in inventory_keys: + surface_key_hits.append(row["row_id"]) + semantic_rows.append({ + "alignment_status": joined["alignment_status"], + "candidate_aligned_synset": joined["candidate_aligned_synset"], + "candidate_source": "KORKONTZELOS_MANANDHAR", + "candidate_source_item_id": joined["candidate_source_item_id"], + "hyperlex_synset": synset, + "json_schema_document": None, + "pos": row["pos"], + "primary_evidence_code": joined["primary_evidence_code"], + "provenance": { + "gloss_used_to_choose_synset": False, + "operator_label_used": False, + "publication": PUBLICATION, + "publication_pdf_sha256": EXPECTED[PDF], + "surface_join": False, + "wordnet_release": "3.0", + }, + "row_id": row["row_id"], + "schema": "hyperlex.km_hyperlex_semantic_evidence_row.v1", + "semantic_noncompositional": joined["semantic_noncompositional"], + "surface": row["surface"], + }) + alignment_counts = Counter(row["alignment_status"] for row in alignments) + label_counts = Counter(row["source_label"] for row in alignments) + inventory_sha = write_jsonl(INVENTORY_PATH, inventory) + alignment_sha = write_jsonl(ALIGNMENT_PATH, alignments) + semantic_sha = write_jsonl(SEMANTIC_PATH, semantic_rows) + frozen = [json.loads(line) for line in SEMANTIC_PATH.read_text(encoding="utf-8").splitlines() if line] + if any("operator_bucket" in row for row in frozen): + refuse("operator label entered semantic evidence") + by_operator = {name: Counter() for name in OPERATORS} + semantic_counts = Counter() + code_counts = Counter() + false_yes = [] + false_no = [] + for manifest_row, evidence_row in zip(rows, frozen): + operator = manifest_row["operator_bucket"] + semantic = evidence_row["semantic_noncompositional"] + if operator not in by_operator: + refuse(f"unexpected operator bucket: {operator}") + by_operator[operator][semantic] += 1 + semantic_counts[semantic] += 1 + code_counts[evidence_row["primary_evidence_code"]] += 1 + if semantic == "YES" and operator != "HIGH": + false_yes.append(evidence_row["row_id"]) + if semantic == "NO" and operator == "HIGH": + false_no.append(evidence_row["row_id"]) + yes_count = semantic_counts["YES"] + no_count = semantic_counts["NO"] + unknown_count = semantic_counts["UNKNOWN"] + yes_exact = sum(1 for row in frozen if row["semantic_noncompositional"] == "YES" and row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}) + sense_covered = sum(1 for row in frozen if row["alignment_status"] in {"EXACT_SOURCE_ID", "EXACT_UNIQUE_RECONSTRUCTION"}) + polysemous = alignment_counts["AMBIGUOUS_MULTIPLE_SYNSETS"] + unique = alignment_counts["EXACT_UNIQUE_RECONSTRUCTION"] + identity_not_preserved = unique == 0 and polysemous > (len(TABLE) / 2) + status = "CANDIDATE_INSUFFICIENT" if identity_not_preserved else candidate_status(yes_count, yes_exact) + provenance = { + "acquisition_method": "HTTPS GET of the ACL Anthology PDF", + "acquisition_timestamp_utc": pdf_mtime, + "acquisition_url": "https://aclanthology.org/P09-2017.pdf", + "authors": ["Ioannis Korkontzelos", "Suresh Manandhar"], + "canonical_publication_url": "https://aclanthology.org/P09-2017/", + "evaluated_at": when, + "historical_count_hint_not_used": {"compositional": 60, "noncompositional": 56}, + "json_schema_document": None, + "later_naacl_2010_sample_acquired": False, + "publication": "Detecting Compositionality in Multi-Word Expressions", + "publication_year": 2009, + "schema": "hyperlex.km_source_provenance.v1", + "source_artifact": "P09-2017.pdf", + "source_artifact_sha256": EXPECTED[PDF], + "source_name": "Korkontzelos-Manandhar WordNet-derived MWE compositionality evaluation set", + "source_preserves_synset_identifier": False, + "table": "Table 1, proceedings pages 65-68, table printed on page 67", + "venue": "ACL-IJCNLP 2009 short papers", + "verified_inventory_counts": {"COMPOSITIONAL": 19, "NONCOMPOSITIONAL": 19, "items": 38}, + "version": "ACL Anthology P09-2017", + "wordnet_data_sha256": { + "data.adj": EXPECTED[WORDNET / "data.adj"], + "data.adv": EXPECTED[WORDNET / "data.adv"], + "data.noun": EXPECTED[WORDNET / "data.noun"], + "data.verb": EXPECTED[WORDNET / "data.verb"], + }, + "wordnet_release_used_for_reconstruction": "Princeton WordNet 3.0", + "wordnet_version_stated_in_paper": "3.0", + } + receipt = { + "anthology_distribution_license": "CC-BY-NC-SA-3.0", + "anthology_distribution_license_basis": "ACL Anthology footer: materials prior to 2016 are licensed under CC-BY-NC-SA 3.0, and permission is granted to make copies for teaching and research.", + "code_license": None, + "code_license_note": "No code artifact was published with Table 1 and none was acquired.", + "data_license": None, + "data_license_note": "No separate dataset license exists. The evaluation list is Table 1 of the paper.", + "json_schema_document": None, + "license_status": "ESTABLISHED_FOR_RESEARCH_EXTRACTION", + "pdf_copyright_notice": "c\u00a92009 ACL and AFNLP", + "publication_license": "CC-BY-NC-SA-3.0", + "schema": "hyperlex.km_license_receipt.v1", + "source_name": "KORKONTZELOS_MANANDHAR", + "source_version": "P09-2017", + "used": "Table 1 surfaces and the two section labels only", + "not_used_from_table_typography": "Bold, underline, and italic marks report system detections, not gold labels.", + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + evaluation = { + "abstention_count": unknown_count, + "abstention_rate_fraction": fraction(unknown_count, 225), + "alignment_counts": {name: alignment_counts[name] for name in ALIGNMENT_STATES}, + "candidate": "KORKONTZELOS_MANANDHAR", + "candidate_inventory_size": 38, + "compositional_source_count": label_counts["COMPOSITIONAL"], + "development_manifest_sha256": EXPECTED[EVIDENCE_MANIFEST], + "development_rows": 225, + "diagnostics_are_descriptive": True, + "evidence_code_counts": dict(sorted(code_counts.items())), + "false_no_count": len(false_no), + "false_yes_count": len(false_yes), + "inventory_sha256": inventory_sha, + "json_schema_document": None, + "mapping_assumptions": { + "false_no": "A semantic NO whose operator bucket is HIGH. Operator HIGH is not gold semantic YES.", + "false_yes": "A semantic YES whose operator bucket is not HIGH. Operator REJECT is not compositional evidence.", + "no_precision": "Among semantic NO rows, the fraction whose operator bucket is SECONDARY. SECONDARY is not defined as compositional NO.", + "operator_quarantine": "Quarantine is not compositionality evidence.", + "operator_reject": "Reject stays in its own joint-table row. It is a different axis from compositionality.", + "yes_precision": "Among semantic YES rows, the fraction whose operator bucket is HIGH. The correlation is a proxy, not an identity.", + }, + "no_precision": "NOT_COMPUTABLE" if no_count == 0 else fraction(by_operator["SECONDARY"]["NO"], no_count), + "no_support": no_count, + "noncompositional_source_count": label_counts["NONCOMPOSITIONAL"], + "operator_by_semantic": { + name: {state: by_operator[name][state] for state in SEMANTIC_STATES} + for name in OPERATORS + }, + "operator_labels_joined_after_semantic_evidence_was_written": True, + "operator_labels_used_as_runtime_evidence": False, + "pwn30_alignment_sha256": alignment_sha, + "schema": "hyperlex.km_development_evaluation.v1", + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_aligned_coverage_count": sense_covered, + "sense_aligned_coverage_fraction": fraction(sense_covered, 225), + "source_version": "P09-2017", + "surface_key_overlap_count": len(surface_key_hits), + "surface_key_overlap_not_used_as_join": True, + "unknown_support": unknown_count, + "yes_precision": "NOT_COMPUTABLE" if yes_count == 0 else fraction(by_operator["HIGH"]["YES"], yes_count), + "yes_support": yes_count, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + magpie_path = SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json" + magpie = json.loads(magpie_path.read_text(encoding="utf-8")) + comparison = { + "json_schema_document": None, + "korkontzelos_manandhar": { + "abstention": evaluation["abstention_rate_fraction"], + "no_support": no_count, + "sense_aligned_coverage": evaluation["sense_aligned_coverage_fraction"], + "surface_coverage": fraction(len(surface_key_hits), 225), + "target_quality": "Human compositional versus noncompositional labels on a WordNet 3.0 MWE sample. The paper stores no synset id. A unique PWN 3.0 lemma reconstructs one synset for monosemous items.", + "yes_support": yes_count, + }, + "magpie": { + "abstention": magpie["abstention_rate_fraction"], + "development_evaluation_sha256": EXPECTED[magpie_path], + "no_support": magpie["no_support"], + "sense_aligned_coverage": magpie["sense_alignment_coverage_fraction"], + "surface_coverage": magpie["surface_coverage_fraction"], + "target_quality": "Contextual literal versus idiomatic labels. The artifact stores no WordNet sense identifier.", + "yes_support": magpie["yes_support"], + }, + "purpose": "Compare sense-aligned support. Raw surface coverage is not the ranking.", + "schema": "hyperlex.km_magpie_comparison.v1", + "sense_alignment_barrier": { + "korkontzelos_manandhar_inventory_unique_reconstructions": unique, + "korkontzelos_manandhar_hyperlex_sense_aligned_rows": sense_covered, + "magpie_hyperlex_sense_aligned_rows": 0, + }, + } + comparison_sha = write_json(COMPARISON_PATH, comparison) + decision = { + "candidate": "KORKONTZELOS_MANANDHAR", + "comparison_sha256": comparison_sha, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "inventory_sha256": inventory_sha, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "next_transition_authorized": False, + "pwn30_alignment_sha256": alignment_sha, + "readiness_question_met": yes_count > 0 and yes_exact == yes_count, + "runtime_integration": False, + "schema": "hyperlex.km_candidate_decision.v1", + "select_005_authorized": False, + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_identity_not_preserved_finding": identity_not_preserved, + "sense_identity_not_preserved_finding_name": "WORDNET_DERIVED_BUT_SENSE_IDENTITY_NOT_PRESERVED" if identity_not_preserved else None, + "source_provenance_sha256": provenance_sha, + "source_version": "P09-2017", + "state": "CANDIDATE_SOURCE_EVALUATED", + "why": "Table 1 stores surfaces and labels, not synset ids. Unique PWN 3.0 lemmas reconstruct 30 synsets, and 8 items are polysemous and stay UNKNOWN. None of the reconstructed synsets occur in the 225 development rows, so Hyperlex YES support is 0.", + } + decision_sha = write_json(DECISION_PATH, decision) + for path, expected in EXPECTED.items(): + if path == TRACKER or expected is None: + continue + if sha256(path) != expected: + refuse(f"evaluation mutated {path}") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["evaluated_candidates"] = ["MAGPIE", "KORKONTZELOS_MANANDHAR"] + tracker["magpie_evaluation_status"] = "CANDIDATE_INSUFFICIENT" + tracker["km_evaluation_status"] = status + tracker["latest_evaluated_candidate"] = "KORKONTZELOS_MANANDHAR" + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["km_source_provenance_sha256"] = provenance_sha + tracker["km_license_receipt_sha256"] = receipt_sha + tracker["km_raw_inventory_sha256"] = inventory_sha + tracker["km_pwn30_alignment_sha256"] = alignment_sha + tracker["km_hyperlex_semantic_evidence_sha256"] = semantic_sha + tracker["km_development_evaluation_sha256"] = evaluation_sha + tracker["km_magpie_comparison_sha256"] = comparison_sha + tracker["km_candidate_decision_sha256"] = decision_sha + tracker["km_source_version"] = "P09-2017" + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + for path, expected in EXPECTED.items(): + if path == TRACKER or expected is None: + continue + if sha256(path) != expected: + refuse(f"tracker update mutated {path}") + print(json.dumps({ + "alignment_counts": evaluation["alignment_counts"], + "candidate_decision_sha256": decision_sha, + "comparison_sha256": comparison_sha, + "development_evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "inventory_sha256": inventory_sha, + "license_receipt_sha256": receipt_sha, + "no_support": no_count, + "operator_by_semantic": evaluation["operator_by_semantic"], + "pwn30_alignment_sha256": alignment_sha, + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_aligned_coverage_fraction": evaluation["sense_aligned_coverage_fraction"], + "sense_identity_not_preserved_finding": identity_not_preserved, + "source_provenance_sha256": provenance_sha, + "tracker_sha256": tracker_sha, + "unknown_support": unknown_count, + "yes_support": yes_count, + "label_counts": dict(label_counts), + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py b/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py new file mode 100644 index 00000000..fa66b52a --- /dev/null +++ b/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py @@ -0,0 +1,363 @@ +"""Development-only matcher for the MAGPIE candidate evaluation. + +Surface identity is orthographic. A surface hit does not assign +semantic noncompositionality. Sense alignment requires identifier +equality with the supplied synset. This module does not read operator +labels, gloss text, or a model judgment. +""" + +from __future__ import annotations + +from collections import Counter +from collections.abc import Iterable +from decimal import Decimal + +EXACT = "EXACT" +NORMALIZED = "NORMALIZED" +VARIANT = "VARIANT" +NONE = "NONE" +AMBIGUOUS = "AMBIGUOUS" +SURFACE_MATCHES = frozenset({EXACT, NORMALIZED, VARIANT, NONE, AMBIGUOUS}) + +ALIGNED_IDIOMATIC = "ALIGNED_IDIOMATIC" +ALIGNED_LITERAL = "ALIGNED_LITERAL" +MIXED = "MIXED" +CONFLICT = "CONFLICT" +UNKNOWN = "UNKNOWN" +ALIGNMENTS = frozenset({ALIGNED_IDIOMATIC, ALIGNED_LITERAL, MIXED, CONFLICT, UNKNOWN}) + +YES = "YES" +NO = "NO" + +CODE_ALIGNED_IDIOMATIC = "magpie_aligned_idiomatic" +CODE_ALIGNED_LITERAL = "magpie_aligned_literal" +CODE_MIXED = "magpie_mixed_usage" +CODE_SURFACE = "magpie_surface_only" +CODE_CONFLICT = "magpie_sense_conflict" +CODE_NONE = "magpie_no_match" + +_SENSE_FIELDS = frozenset({"synset", "synset_offset", "sense_key", "wordnet_offset"}) +_APOSTROPHES = str.maketrans( + { + "\u2019": "'", + "\u2018": "'", + "\u02bc": "'", + "\u2032": "'", + "`": "'", + "\u00b4": "'", + } +) +_PUNCTUATION = str.maketrans( + { + "-": " ", + "\u2010": " ", + "\u2011": " ", + "\u2013": " ", + "\u2014": " ", + ".": " ", + ",": " ", + ";": " ", + ":": " ", + "!": " ", + "?": " ", + '"': " ", + "(": " ", + ")": " ", + "[": " ", + "]": " ", + "{": " ", + "}": " ", + "/": " ", + "\\": " ", + } +) + + +def exact_key(text: str) -> str: + """Casefold and collapse separators. Punctuation stays in the key.""" + folded = text.casefold().replace("_", " ") + return " ".join(folded.split()) + + +def normalized_key(text: str) -> str: + """Exact key plus apostrophe folding and punctuation removal. + + Apostrophes are kept. Possessive pronouns are not rewritten, and no + stem or inflection is restored. + """ + folded = exact_key(text).translate(_APOSTROPHES).translate(_PUNCTUATION) + return " ".join(folded.split()) + + +def _confidence_key(value: object) -> str: + if isinstance(value, bool) or not isinstance(value, (int, float, Decimal)): + raise ValueError("confidence is not numeric") + if isinstance(value, Decimal): + return format(value, "f") + if isinstance(value, int): + return str(value) + return format(Decimal(str(value)), "f") + + +def _sense_token(instance: dict) -> str | None: + found = [ + field + for field in sorted(_SENSE_FIELDS) + if field in instance and instance[field] not in (None, "") + ] + if not found: + return None + if len(found) == 1: + return str(instance[found[0]]) + return "|".join(f"{field}={instance[field]}" for field in found) + + +def _label_bucket(label: object) -> str: + if label == "i": + return "idiomatic" + if label == "l": + return "literal" + return "unresolved" + + +class TypeUsage: + def __init__(self, idiom: str) -> None: + self.idiom = idiom + self.literal_instance_count = 0 + self.idiomatic_instance_count = 0 + self.unresolved_instance_count = 0 + self.source_instance_ids: list[int] = [] + self.variant_types: Counter[str] = Counter() + self.confidence_histogram: Counter[str] = Counter() + self.judgment_counts: list[int] = [] + self.sense_ids: list[str] = [] + self.instances_with_sense_id = 0 + self.instance_count = 0 + + def add(self, instance: dict) -> None: + identifier = instance["id"] + if isinstance(identifier, bool) or not isinstance(identifier, int): + raise ValueError("instance id is not an int") + bucket = _label_bucket(instance["label"]) + if bucket == "idiomatic": + self.idiomatic_instance_count += 1 + elif bucket == "literal": + self.literal_instance_count += 1 + else: + self.unresolved_instance_count += 1 + self.source_instance_ids.append(identifier) + variant = instance["variant_type"] + if not isinstance(variant, str) or not variant: + raise ValueError("variant_type is missing") + self.variant_types[variant] += 1 + self.confidence_histogram[_confidence_key(instance["confidence"])] += 1 + if "judgment_count" in instance and instance["judgment_count"] is not None: + judgment = instance["judgment_count"] + if isinstance(judgment, bool) or not isinstance(judgment, int): + raise ValueError("judgment_count is not an int") + self.judgment_counts.append(judgment) + token = _sense_token(instance) + self.instance_count += 1 + if token is not None: + self.instances_with_sense_id += 1 + self.sense_ids.append(token) + + def finish(self) -> None: + self.source_instance_ids.sort() + if len(self.source_instance_ids) != len(set(self.source_instance_ids)): + raise ValueError("duplicate instance id inside one expression") + + +class CorpusIndex: + def __init__(self) -> None: + self.types: dict[str, TypeUsage] = {} + self.by_exact: dict[str, list[str]] = {} + self.by_normalized: dict[str, list[str]] = {} + self.field_names: set[str] = set() + self.instance_count = 0 + self.records_bound_sense = False + + @property + def type_count(self) -> int: + return len(self.types) + + +def build_index(instances: Iterable[dict]) -> CorpusIndex: + index = CorpusIndex() + seen_ids: set[int] = set() + for instance in instances: + index.field_names.update(instance.keys()) + idiom = instance["idiom"] + if not isinstance(idiom, str) or not idiom.strip(): + raise ValueError("idiom string is empty") + identifier = instance["id"] + if identifier in seen_ids: + raise ValueError("duplicate instance id") + seen_ids.add(identifier) + usage = index.types.get(idiom) + if usage is None: + usage = TypeUsage(idiom) + index.types[idiom] = usage + usage.add(instance) + index.instance_count += 1 + if index.field_names & _SENSE_FIELDS: + index.records_bound_sense = True + exact_groups: dict[str, list[str]] = {} + normal_groups: dict[str, list[str]] = {} + for idiom, usage in index.types.items(): + usage.finish() + exact_groups.setdefault(exact_key(idiom), []).append(idiom) + key = normalized_key(idiom) + if key: + normal_groups.setdefault(key, []).append(idiom) + for key, idioms in exact_groups.items(): + index.by_exact[key] = sorted(idioms) + for key, idioms in normal_groups.items(): + index.by_normalized[key] = sorted(idioms) + return index + + +def _confidence_summary(usage: TypeUsage | None) -> dict | None: + if usage is None: + return None + summary = { + "histogram": dict(sorted(usage.confidence_histogram.items())), + "instance_count": usage.instance_count, + } + if usage.judgment_counts: + summary["judgment_count_maximum"] = max(usage.judgment_counts) + summary["judgment_count_minimum"] = min(usage.judgment_counts) + return summary + + +def match_surface(surface: str, index: CorpusIndex) -> dict: + """Return one surface relation. Variant metadata is not a second idiom.""" + exact_hits = index.by_exact.get(exact_key(surface), []) + if len(exact_hits) == 1: + relation = EXACT + idiom = exact_hits[0] + candidates: list[str] = [] + elif len(exact_hits) > 1: + relation = AMBIGUOUS + idiom = None + candidates = list(exact_hits) + else: + normal_hits = index.by_normalized.get(normalized_key(surface), []) + if len(normal_hits) == 1: + relation = NORMALIZED + idiom = normal_hits[0] + candidates = [] + elif len(normal_hits) > 1: + relation = AMBIGUOUS + idiom = None + candidates = list(normal_hits) + else: + relation = NONE + idiom = None + candidates = [] + usage = index.types[idiom] if idiom is not None else None + attributed = usage is not None + return { + "ambiguous_candidates": candidates, + "annotation_confidence": _confidence_summary(usage), + "attributed": attributed, + "idiomatic_instance_count": usage.idiomatic_instance_count if usage else None, + "literal_instance_count": usage.literal_instance_count if usage else None, + "matched_magpie_expression": idiom, + "source_instance_ids": list(usage.source_instance_ids) if usage else [], + "surface_match": relation, + "unresolved_instance_count": usage.unresolved_instance_count if usage else None, + "usage": usage, + "variant_types": dict(sorted(usage.variant_types.items())) if usage else {}, + } + + +def align_sense(match: dict, supplied_synset: str, index: CorpusIndex) -> tuple[str, str]: + """Align only by equality of a source sense identifier to the synset. + + Mixed literal and idiomatic counts do not by themselves create MIXED. + The supplied gloss is not an argument. + """ + relation = match["surface_match"] + if relation == NONE: + return UNKNOWN, "no_surface_match" + if relation == AMBIGUOUS: + return UNKNOWN, "ambiguous_surface" + if relation == VARIANT: + return UNKNOWN, "variant_without_sense_identifier" + usage: TypeUsage = match["usage"] + if not index.records_bound_sense or usage.instances_with_sense_id == 0: + return UNKNOWN, "no_bound_sense_identifier" + if usage.instances_with_sense_id != usage.instance_count: + return UNKNOWN, "partial_sense_identifier" + identifiers = set(usage.sense_ids) + if supplied_synset in identifiers and len(identifiers) > 1: + return CONFLICT, "sense_identifier_conflict" + if identifiers != {supplied_synset}: + return UNKNOWN, "sense_identifier_mismatch" + literal = usage.literal_instance_count + idiomatic = usage.idiomatic_instance_count + unresolved = usage.unresolved_instance_count + if idiomatic > 0 and literal == 0 and unresolved == 0: + return ALIGNED_IDIOMATIC, "aligned_idiomatic" + if literal > 0 and idiomatic == 0 and unresolved == 0: + return ALIGNED_LITERAL, "aligned_literal" + if idiomatic > 0 and literal > 0: + return MIXED, "mixed_usage_on_aligned_sense" + return UNKNOWN, "unresolved_labels_on_aligned_sense" + + +def evidence_code(surface_match: str, alignment: str) -> str: + if alignment == ALIGNED_IDIOMATIC: + return CODE_ALIGNED_IDIOMATIC + if alignment == ALIGNED_LITERAL: + return CODE_ALIGNED_LITERAL + if alignment == MIXED: + return CODE_MIXED + if alignment == CONFLICT: + return CODE_CONFLICT + if surface_match == NONE: + return CODE_NONE + return CODE_SURFACE + + +def semantic_noncompositional(alignment: str) -> str: + if alignment == ALIGNED_IDIOMATIC: + return YES + if alignment == ALIGNED_LITERAL: + return NO + return UNKNOWN + + +def evaluate_row( + surface: str, + synset: str, + index: CorpusIndex, + source_version: str, + source_artifact_hash: str, +) -> dict: + """One development row. Operator labels and glosses are not inputs.""" + match = match_surface(surface, index) + alignment, basis = align_sense(match, synset, index) + code = evidence_code(match["surface_match"], alignment) + semantic = semantic_noncompositional(alignment) + if match["surface_match"] != NONE and semantic == YES and alignment != ALIGNED_IDIOMATIC: + raise RuntimeError("surface match assigned YES without aligned idiomatic evidence") + return { + "alignment_basis": basis, + "ambiguous_candidates": match["ambiguous_candidates"], + "annotation_confidence": match["annotation_confidence"], + "idiomatic_instance_count": match["idiomatic_instance_count"], + "literal_instance_count": match["literal_instance_count"], + "matched_magpie_expression": match["matched_magpie_expression"], + "primary_evidence_code": code, + "semantic_noncompositional": semantic, + "sense_alignment": alignment, + "source_artifact_hash": source_artifact_hash, + "source_instance_ids": match["source_instance_ids"], + "source_name": "MAGPIE", + "source_version": source_version, + "surface_match": match["surface_match"], + "unresolved_instance_count": match["unresolved_instance_count"], + "variant_types": match["variant_types"], + } diff --git a/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py b/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py new file mode 100644 index 00000000..30c64e18 --- /dev/null +++ b/scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py @@ -0,0 +1,576 @@ +"""Evaluate the pinned MAGPIE author corpus on the 225 development rows. + +This is candidate-source evidence research. It does not select MAGPIE, +integrate it, draw a measurement sample, or authorize SELECT-005. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from decimal import Decimal +from pathlib import Path + +from hyperlexical.magpie_candidate_evaluation import build_index, evaluate_row + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +MAGPIE = ( + LEDGER + / "acquisition/sources/magpie-corpus/7fa677b82b9a772dfa54bbdd0fb414412d73db3b" +) +COMMIT = "7fa677b82b9a772dfa54bbdd0fb414412d73db3b" +UNFILTERED = MAGPIE / "MAGPIE_unfiltered.jsonl" +LICENSE = MAGPIE / "LICENSE" +README = MAGPIE / "README.md" +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "MAGPIE_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "MAGPIE_LICENSE_RECEIPT.json" +MATCHES_PATH = SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl" +ALIGNMENT_PATH = SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl" +SEMANTIC_PATH = SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl" +EVALUATION_PATH = SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json" +DECISION_PATH = SOURCE / "MAGPIE_CANDIDATE_DECISION.json" + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "521eccb8bd068d4697a3ffdc06e3b44feb4eec3c8a3bd6aebcea11dd288b7dfc", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + UNFILTERED: "541ee535e93d71eff85351351665115e2a9f22ad736423881da5774a93bc880e", + LICENSE: "05ab88f3f9da1d05f9c5bf0a7c45c49a9007f877dd9c237a5bf668276fe04c3b", + README: "bc9d281e2348a780b91929e44de0e58bc26a822d69835b0e6dd8a666754af6db", +} +BLOB = { + UNFILTERED: "24a6ef5cc6b226859956a903395caddb81c09493", + LICENSE: "56902ed9fa665f4aeebdb8cc3df5743a6de02542", + README: "f20506c021aa1484ea3ece67dedde20c202bbf53", +} +EXPECTED_INSTANCES = 56622 +EXPECTED_TYPES = 1756 +EXPECTED_LABELS = {"?": 7, "i": 40011, "l": 16168, "o": 436} +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +SEMANTIC_STATES = ("YES", "NO", "UNKNOWN") +SURFACE_STATES = ("EXACT", "NORMALIZED", "VARIANT", "NONE", "AMBIGUOUS") +ALIGNMENTS = ("ALIGNED_IDIOMATIC", "ALIGNED_LITERAL", "MIXED", "CONFLICT", "UNKNOWN") + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def git_blob_sha1(path: Path) -> str: + data = path.read_bytes() + return hashlib.sha1(f"blob {len(data)}\0".encode() + data).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def iso_mtime(path: Path) -> str: + stamp = datetime.fromtimestamp(path.stat().st_mtime, timezone.utc) + return stamp.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str: + return f"{numerator}/{denominator}" + + +def load_instances(path: Path) -> tuple[list[dict], Counter]: + rows = [] + labels: Counter[str] = Counter() + with path.open(encoding="utf-8") as handle: + for line in handle: + if not line.strip(): + continue + instance = json.loads(line, parse_float=Decimal) + rows.append(instance) + labels[str(instance.get("label"))] += 1 + return rows, labels + + +def license_established(text: str) -> bool: + if "Attribution 4.0 International" not in text: + return False + if "NonCommercial" in text or "Non-Commercial" in text: + return False + return True + + +def project_match(row: dict, found: dict) -> dict: + return { + "ambiguous_candidates": found["ambiguous_candidates"], + "annotation_confidence": found["annotation_confidence"], + "idiomatic_instance_count": found["idiomatic_instance_count"], + "literal_instance_count": found["literal_instance_count"], + "matched_magpie_expression": found["matched_magpie_expression"], + "pos": row["pos"], + "row_id": row["row_id"], + "schema": "hyperlex.magpie_development_match_row.v1", + "source_instance_ids": found["source_instance_ids"], + "source_name": "MAGPIE", + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + "unresolved_instance_count": found["unresolved_instance_count"], + "variant_types": found["variant_types"], + } + + +def project_alignment(row: dict, found: dict) -> dict: + return { + "alignment_basis": found["alignment_basis"], + "gloss_used": False, + "json_schema_document": None, + "matched_magpie_expression": found["matched_magpie_expression"], + "operator_label_used": False, + "pos": row["pos"], + "row_id": row["row_id"], + "schema": "hyperlex.magpie_sense_alignment_row.v1", + "sense_alignment": found["sense_alignment"], + "sense_alignment_rule": "identifier_equality_only", + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + } + + +def project_evidence(row: dict, found: dict, provenance: dict) -> dict: + return { + "idiomatic_instance_count": found["idiomatic_instance_count"], + "json_schema_document": None, + "literal_instance_count": found["literal_instance_count"], + "matched_magpie_expression": found["matched_magpie_expression"], + "pos": row["pos"], + "primary_evidence_code": found["primary_evidence_code"], + "provenance": provenance, + "row_id": row["row_id"], + "schema": "hyperlex.magpie_semantic_evidence_row.v1", + "semantic_noncompositional": found["semantic_noncompositional"], + "sense_alignment": found["sense_alignment"], + "source_artifact_hash": found["source_artifact_hash"], + "source_instance_ids": found["source_instance_ids"], + "source_name": "MAGPIE", + "source_version": found["source_version"], + "surface": row["surface"], + "surface_match": found["surface_match"], + "synset": found["synset"], + "unresolved_instance_count": found["unresolved_instance_count"], + } + + +def candidate_status(yes_count: int, yes_aligned: int) -> str: + if yes_count == 0: + return "CANDIDATE_INSUFFICIENT" + if yes_aligned != yes_count: + return "CANDIDATE_REJECTED" + return "CANDIDATE_PROMISING" + + +def main() -> None: + for path, expected in EXPECTED.items(): + if sha256(path) != expected: + refuse(f"sealed artifact changed before evaluation: {path}") + for path, expected in BLOB.items(): + if git_blob_sha1(path) != expected: + refuse(f"git blob does not match hslh/magpie-corpus: {path.name}") + license_text = LICENSE.read_text(encoding="utf-8") + established = license_established(license_text) + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + artifact_hash = EXPECTED[UNFILTERED] + publication_pdf_sha = "8247c926909772ce317d5f33cddb83caa51969eb5ef928bcbadbbcf05e39c979" + provenance = { + "acquisition_timestamp_utc": { + "LICENSE": iso_mtime(LICENSE), + "MAGPIE_unfiltered.jsonl": iso_mtime(UNFILTERED), + "README.md": iso_mtime(README), + }, + "acquisition_urls": { + "LICENSE": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/LICENSE", + "MAGPIE_unfiltered.jsonl": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/MAGPIE_unfiltered.jsonl", + "README.md": f"https://raw.githubusercontent.com/hslh/magpie-corpus/{COMMIT}/README.md", + }, + "artifact_filenames": ["LICENSE", "MAGPIE_unfiltered.jsonl", "README.md"], + "artifact_hashes_sha256": { + "LICENSE": EXPECTED[LICENSE], + "MAGPIE_unfiltered.jsonl": artifact_hash, + "README.md": EXPECTED[README], + }, + "canonical_source_repository": "https://github.com/hslh/magpie-corpus", + "dataset_variant": "author corpus, MAGPIE_unfiltered.jsonl", + "filtered_splits_acquired": False, + "filtered_splits_not_used": { + "MAGPIE_filtered_split_random.jsonl": { + "acquired": False, + "git_blob_sha1": "59c24bffbccb38d912930d761abc0e8d7846e128", + "sha256": None, + }, + "MAGPIE_filtered_split_typebased.jsonl": { + "acquired": False, + "git_blob_sha1": "722546a2df509b5f50556445fdabb13179019b85", + "sha256": None, + }, + }, + "git_blob_sha1": {path.name: digest for path, digest in BLOB.items()}, + "hugging_face_package_used": False, + "publication": { + "anthology_url": "https://aclanthology.org/2020.lrec-1.35/", + "authors": ["Hessel Haagsma", "Johan Bos", "Malvina Nissim"], + "pdf_sha256": publication_pdf_sha, + "pdf_url": "https://aclanthology.org/2020.lrec-1.35.pdf", + "title": "MAGPIE: A Large Corpus of Potentially Idiomatic Expressions", + "venue": "LREC 2020", + }, + "repository_full_name": "hslh/magpie-corpus", + "schema": "hyperlex.magpie_source_provenance.v1", + "source_name": "MAGPIE — A Large Corpus of Potentially Idiomatic Expressions", + "unrelated_magpie_repository_used": False, + "version_or_commit": COMMIT, + "version_recorded_at": "2020-06-07T09:57:12Z", + } + receipt = { + "code_license": "CC-BY-4.0", + "code_license_note": "The pinned commit has one LICENSE file and no separate software license.", + "dataset_artifact_license": "CC-BY-4.0" if established else "SOURCE_LICENSE_UNRESOLVED", + "dataset_license_basis": "LICENSE file in hslh/magpie-corpus at the pinned commit, corroborated by the GitHub license API SPDX CC-BY-4.0. The jsonl itself has no license field.", + "evaluated_artifact": "MAGPIE_unfiltered.jsonl", + "hugging_face_dataset_card_consulted": False, + "json_schema_document": None, + "license_status": "ESTABLISHED" if established else "SOURCE_LICENSE_UNRESOLVED", + "publication_license": "CC-BY-NC", + "publication_license_basis": "Page 1 of the anthology PDF states that the ELRA proceedings text is licensed under CC-BY-NC. That statement is not the dataset license.", + "publication_pdf_sha256": publication_pdf_sha, + "schema": "hyperlex.magpie_license_receipt.v1", + "source_name": "MAGPIE", + "source_version": COMMIT, + } + if not established: + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + decision = { + "candidate": "MAGPIE", + "evaluation_status": "SOURCE_LICENSE_UNRESOLVED", + "license_receipt_sha256": receipt_sha, + "runtime_integration": False, + "schema": "hyperlex.magpie_candidate_decision.v1", + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + } + write_json(DECISION_PATH, decision) + refuse("SOURCE_LICENSE_UNRESOLVED") + + instances, labels = load_instances(UNFILTERED) + if labels != Counter(EXPECTED_LABELS): + refuse(f"label census does not match the pinned corpus: {dict(labels)}") + index = build_index(instances) + if index.instance_count != EXPECTED_INSTANCES or index.type_count != EXPECTED_TYPES: + refuse("instance or type count does not match the pinned corpus") + if index.records_bound_sense: + refuse("pinned corpus unexpectedly carries a sense identifier") + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225: + refuse("development manifest is not 225 rows") + if any(row.get("sense_class") is not None for row in rows): + refuse("development manifest already has a sense class") + if manifest.get("sense_classes_assigned") is not False: + refuse("development manifest sense_classes_assigned flag changed") + + row_provenance = { + "alignment_basis_field": "alignment_basis", + "artifact": "MAGPIE_unfiltered.jsonl", + "artifact_sha256": artifact_hash, + "commit": COMMIT, + "gloss_used": False, + "operator_label_used": False, + "repository": "https://github.com/hslh/magpie-corpus", + "sense_alignment_rule": "identifier_equality_only", + "surface_match_assigns_semantic_yes": False, + "variant_type_is_not_an_alternate_expression": True, + } + matches = [] + alignments = [] + semantic_rows = [] + for row in rows: + synset = f"{row['synset_pos']}:{row['synset_offset']}" + found = evaluate_row(row["surface"], synset, index, COMMIT, artifact_hash) + found["synset"] = synset + item_provenance = dict(row_provenance) + item_provenance["alignment_basis"] = found["alignment_basis"] + matches.append(project_match(row, found)) + alignments.append(project_alignment(row, found)) + semantic_rows.append(project_evidence(row, found, item_provenance)) + + provenance["evaluated_at"] = when + provenance["instance_count"] = index.instance_count + provenance["label_counts"] = dict(sorted(labels.items())) + provenance["records_bound_sense_identifier"] = False + provenance["type_count"] = index.type_count + provenance["variant_match_available"] = False + provenance_sha = write_json(PROVENANCE_PATH, provenance) + receipt_sha = write_json(LICENSE_PATH, receipt) + matches_sha = write_jsonl(MATCHES_PATH, matches) + alignment_sha = write_jsonl(ALIGNMENT_PATH, alignments) + semantic_sha = write_jsonl(SEMANTIC_PATH, semantic_rows) + + frozen_semantic = [ + json.loads(line) + for line in SEMANTIC_PATH.read_text(encoding="utf-8").splitlines() + if line + ] + if [row["row_id"] for row in frozen_semantic] != [row["row_id"] for row in rows]: + refuse("semantic evidence row order drifted") + if any("operator_bucket" in row for row in frozen_semantic): + refuse("operator label entered the semantic evidence") + + by_operator = {name: Counter() for name in OPERATORS} + semantic_counts = Counter() + surface_counts = Counter() + alignment_counts = Counter() + code_counts = Counter() + false_yes = [] + false_no = [] + mixed_usage_rows = [] + mixed_expressions = set() + sense_conflict_rows = [] + for manifest_row, evidence_row in zip(rows, frozen_semantic): + operator = manifest_row["operator_bucket"] + semantic = evidence_row["semantic_noncompositional"] + if operator not in by_operator: + refuse(f"unexpected operator bucket: {operator}") + by_operator[operator][semantic] += 1 + semantic_counts[semantic] += 1 + surface_counts[evidence_row["surface_match"]] += 1 + alignment_counts[evidence_row["sense_alignment"]] += 1 + code_counts[evidence_row["primary_evidence_code"]] += 1 + if semantic == "YES" and operator != "HIGH": + false_yes.append(evidence_row["row_id"]) + if semantic == "NO" and operator == "HIGH": + false_no.append(evidence_row["row_id"]) + literal = evidence_row["literal_instance_count"] + idiomatic = evidence_row["idiomatic_instance_count"] + if literal and idiomatic: + mixed_usage_rows.append(evidence_row["row_id"]) + mixed_expressions.add(evidence_row["matched_magpie_expression"]) + if evidence_row["sense_alignment"] == "CONFLICT": + sense_conflict_rows.append(evidence_row["row_id"]) + + yes_count = semantic_counts["YES"] + no_count = semantic_counts["NO"] + unknown_count = semantic_counts["UNKNOWN"] + yes_aligned = sum( + 1 + for row in frozen_semantic + if row["semantic_noncompositional"] == "YES" and row["sense_alignment"] == "ALIGNED_IDIOMATIC" + ) + covered = surface_counts["EXACT"] + surface_counts["NORMALIZED"] + surface_counts["VARIANT"] + sense_covered = sum(alignment_counts[name] for name in ALIGNMENTS if name != "UNKNOWN") + status = candidate_status(yes_count, yes_aligned) + mapping_assumptions = { + "false_no": "A semantic NO whose operator bucket is HIGH. This is a descriptive disagreement under an explicit proxy. Operator HIGH is not gold semantic YES.", + "false_yes": "A semantic YES whose operator bucket is not HIGH. This is a descriptive disagreement under an explicit proxy. It is not computed by treating REJECT as compositional.", + "no_precision": "Among semantic NO rows, the fraction whose operator bucket is SECONDARY. SECONDARY is not defined as compositional NO.", + "operator_reject": "Operator REJECT is reported as its own row in the joint table. It is not mapped to semantic YES or NO.", + "yes_precision": "Among semantic YES rows, the fraction whose operator bucket is HIGH. Operator HIGH is not defined as semantic noncompositionality.", + } + evaluation = { + "abstention_count": unknown_count, + "abstention_rate_fraction": fraction(unknown_count, 225), + "candidate": "MAGPIE", + "coverage_is_not_success": True, + "development_manifest_sha256": EXPECTED[EVIDENCE_MANIFEST], + "development_rows": 225, + "diagnostics_are_descriptive": True, + "evidence_code_counts": dict(sorted(code_counts.items())), + "false_no_count": len(false_no), + "false_no_row_ids": false_no, + "false_yes_count": len(false_yes), + "false_yes_row_ids": false_yes, + "high_unknown_rate_is_acceptable": True, + "json_schema_document": None, + "mapping_assumptions": mapping_assumptions, + "mixed_usage_expression_count": len(mixed_expressions), + "mixed_usage_is_not_sense_alignment_mixed": True, + "mixed_usage_row_count": len(mixed_usage_rows), + "no_precision": "NOT_COMPUTABLE" if no_count == 0 else fraction(by_operator["SECONDARY"]["NO"], no_count), + "no_support": no_count, + "operator_by_semantic": { + name: {state: by_operator[name][state] for state in SEMANTIC_STATES} + for name in OPERATORS + }, + "operator_labels_joined_after_semantic_evidence_was_written": True, + "operator_labels_used_as_runtime_evidence": False, + "schema": "hyperlex.magpie_development_evaluation.v1", + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_alignment_counts": {name: alignment_counts[name] for name in ALIGNMENTS}, + "sense_alignment_coverage_count": sense_covered, + "sense_alignment_coverage_fraction": fraction(sense_covered, 225), + "sense_conflict_count": len(sense_conflict_rows), + "source_artifact_hash": artifact_hash, + "source_version": COMMIT, + "surface_ambiguous_count": surface_counts["AMBIGUOUS"], + "surface_coverage_count": covered, + "surface_coverage_excludes_ambiguous": True, + "surface_coverage_fraction": fraction(covered, 225), + "surface_match_counts": {name: surface_counts[name] for name in SURFACE_STATES}, + "surface_none_count": surface_counts["NONE"], + "unknown_support": unknown_count, + "yes_precision": "NOT_COMPUTABLE" if yes_count == 0 else fraction(by_operator["HIGH"]["YES"], yes_count), + "yes_support": yes_count, + "yes_support_with_aligned_idiomatic": yes_aligned, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + decision = { + "candidate": "MAGPIE", + "candidate_family": "C_curated_linguistic_resource", + "coverage_was_not_treated_as_success": True, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "next_transition_authorized": False, + "not_selected_reason": "Semantic YES support is zero. MAGPIE annotates idiomatic and literal uses of a potentially idiomatic expression. The pinned unfiltered artifact has no WordNet synset, sense key, or gloss, so the supplied Hyperlex sense cannot be identified by equality. Surface coverage is not sense alignment.", + "readiness_question_met": False, + "recommended_next_candidate": "A later authorization may name one curated resource whose entries carry a WordNet synset offset or sense key for the expression sense. PARSEME, STREUSLE, PIE, EPIE, and NCS were not downloaded and are not selected.", + "runtime_integration": False, + "schema": "hyperlex.magpie_candidate_decision.v1", + "select_005_authorized": False, + "selected_source": "none", + "sense_alignment_sha256": alignment_sha, + "source_provenance_sha256": provenance_sha, + "source_version": COMMIT, + "state": "CANDIDATE_SOURCE_EVALUATED", + } + decision_sha = write_json(DECISION_PATH, decision) + + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("evaluation wrote into the development manifest") + reloaded = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if any(row.get("sense_class") is not None for row in reloaded["rows"]): + refuse("evaluation wrote a sense class") + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"evaluation mutated {path}") + + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = "SEMANTIC_EVIDENCE_SOURCE_SPEC_FROZEN" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_candidate"] = "MAGPIE" + tracker["semantic_evidence_source_evaluation_status"] = status + tracker["semantic_evidence_source_selected"] = "none" + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["magpie_source_provenance_sha256"] = provenance_sha + tracker["magpie_license_receipt_sha256"] = receipt_sha + tracker["magpie_development_matches_sha256"] = matches_sha + tracker["magpie_sense_alignment_sha256"] = alignment_sha + tracker["magpie_semantic_evidence_sha256"] = semantic_sha + tracker["magpie_development_evaluation_sha256"] = evaluation_sha + tracker["magpie_candidate_decision_sha256"] = decision_sha + tracker["magpie_source_version"] = COMMIT + tracker["magpie_source_artifact_sha256"] = artifact_hash + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"tracker update mutated {path}") + print(json.dumps({ + "candidate_decision_sha256": decision_sha, + "development_evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "license_receipt_sha256": receipt_sha, + "no_support": no_count, + "operator_by_semantic": evaluation["operator_by_semantic"], + "selected_source": "none", + "semantic_evidence_sha256": semantic_sha, + "sense_alignment_coverage_fraction": evaluation["sense_alignment_coverage_fraction"], + "sense_alignment_sha256": alignment_sha, + "source_provenance_sha256": provenance_sha, + "surface_coverage_fraction": evaluation["surface_coverage_fraction"], + "surface_match_counts": evaluation["surface_match_counts"], + "tracker_sha256": tracker_sha, + "unknown_support": unknown_count, + "yes_support": yes_count, + "false_no_count": len(false_no), + "false_yes_count": len(false_yes), + "mixed_usage_expression_count": len(mixed_expressions), + "mixed_usage_row_count": len(mixed_usage_rows), + "evidence_code_counts": evaluation["evidence_code_counts"], + "matches_sha256": matches_sha, + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py new file mode 100644 index 00000000..8259c276 --- /dev/null +++ b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py @@ -0,0 +1,498 @@ +"""Tier-3 model decision for constituent senses that Extended Lesk left tied. + +The function selects only among supplied PWN 3.0 candidates, or it abstains. +It does not see operator labels, residual scores, or a residual embedding. +""" + +from __future__ import annotations + +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + STRUCTURAL_TOKENS, + extract_constituents, + neighbor_keys, +) + +RULE_VERSION = "MODEL_BASED_WSD_CANDIDATE_V1" +MODEL_NAME = "kanishka/GlossBERT" +MODEL_FAMILY = "GlossBERT" +MODEL_REVISION = "0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a" +CANONICAL_SOURCE = "https://huggingface.co/kanishka/GlossBERT" +LICENSE_NAME = "MIT" +POSITIVE_CLASS_INDEX = 1 +MAX_TOKENS = 512 +SCORE_QUANTUM = Decimal("0.000001") +MINIMUM_CONFIDENCE = Decimal("0.50") +MINIMUM_MARGIN = Decimal("0.10") +BASELINE_HIGH_READY = 4 +BASELINE_SECONDARY_READY = 4 +BASELINE_TOTAL_READY = 20 +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +INVALID = "INVALID" +ERROR = "ERROR" +EXACT = "EXACT" +LESK_RESOLVED = "LESK_RESOLVED" +MODEL_RESOLVED = "MODEL_RESOLVED" +UNRESOLVED = "UNRESOLVED" +RESIDUAL_READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +IDENTICAL = "IDENTICAL" +PROMISING = "CANDIDATE_PROMISING" +INSUFFICIENT = "CANDIDATE_INSUFFICIENT" +REJECTED = "CANDIDATE_REJECTED" +NOT_DETERMINISTIC = "NOT_DETERMINISTIC" +NOT_COMPUTABLE = "NOT_COMPUTABLE" +LICENSE_UNRESOLVED = "SOURCE_LICENSE_UNRESOLVED" +READY_CONSTITUENT = frozenset({EXACT, LESK_RESOLVED, MODEL_RESOLVED}) +NEXT_PROMISING = "RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION" +NEXT_REVISION = "MODEL_BASED_WSD_CANDIDATE_REVISION_AUTHORIZATION" +NEXT_DETERMINISM = "MODEL_BASED_WSD_DETERMINISM_REVIEW_AUTHORIZATION" +NEXT_RUNTIME = "MODEL_BASED_WSD_RUNTIME_REVIEW_AUTHORIZATION" +NEXT_LICENSE = "MODEL_BASED_WSD_LICENSE_REVIEW_AUTHORIZATION" + +_RESEARCH_QUESTION = ( + "Can one pinned WordNet-native WSD model resolve enough Extended Lesk ties " + "to make a later semantic-residual replay meaningful, without claiming accuracy?" +) +_INPUT_RULE = ( + "Sent-CLS. The context is the parent surface with the target content token in " + "double quotes. The paired text is the matched lemma, a colon, and the first " + "PWN 3.0 gloss clause. The parent gloss is not appended. Structural tokens " + "locate the quoted token and are not a score." +) +_CANDIDATE_RULE = ( + "Candidates are the frozen Extended Lesk synsets for that constituent. " + "Each gloss uses one matching lemma. Sense keys come from the local index.sense." +) +_SCORE_RULE = ( + "The score is the softmax probability of class index 1, quantized to six " + "decimal places, half even. Class index 1 is the gloss-fits class." +) +_SELECTION_RULE = ( + "Select the unique highest score among the candidate synsets. Do not break a " + "tie by synset order. A selected synset must be one of the candidates." +) +_CONFIDENCE_RULE = ( + "Confidence is the quantized top probability. It is recorded when the model " + "returns scores. Overflow and execution errors leave it empty." +) +_ABSTENTION_RULE = ( + "Abstain unless the top probability is at least one half and the top-versus-second " + "margin is at least one tenth. Do not truncate a pair longer than 512 tokens. " + "If any pair overflows, abstain the constituent." +) +_TIE_RULE = ( + "Equal quantized top scores are AMBIGUOUS. Near ties follow the margin rule. " + "No sense is chosen by list order." +) +_MAPPING_RULE = ( + "The checkpoint is SemCor 3.0 and WordNet 3.0. There is no cross-version map. " + "A winner with zero or several matching sense keys is AMBIGUOUS. Do not migrate a sense by hand." +) +_CIRCULARITY_RULE = ( + "Do not choose a constituent sense because a residual embedding is near the " + "parent gloss. This candidate does not read residual scores or residual vectors." +) +_POOL_RULE = ( + "Tier 3 sees only constituents that Extended Lesk v1 left AMBIGUOUS. " + "It does not replace structural exact, unique lemma, or a Lesk decision that cleared its margin." +) +_GATE_RULE = ( + "Promising only when ready HIGH rows, ready SECONDARY rows, and ready rows " + "are each strictly above the frozen lexical baseline, with no invalid output, " + "no execution error, and identical repeats. This is coverage, not accuracy." +) +_READY_RULE = ( + "A row is projected ready only when at least two content constituents exist " + "and every one is exact, Lesk-resolved, or model-resolved." +) +_NO_GOLD_RULE = ( + "There is no independent constituent-sense gold. Do not report accuracy, " + "precision, recall, or F1." +) + + +def candidate_policy() -> dict: + """Return the preregistered model rule. Outcome counts are not included.""" + return { + "abstention": { + "minimum_margin": "0.10", + "minimum_positive_probability": "0.50", + "on_overflow": "ABSTAIN", + "prose": _ABSTENTION_RULE, + "truncate": False, + }, + "applied_to_hyperlex_at_freeze": False, + "candidate_senses": _CANDIDATE_RULE, + "canonical_source": CANONICAL_SOURCE, + "circularity": _CIRCULARITY_RULE, + "confidence": _CONFIDENCE_RULE, + "determinism": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "inference_mode": True, + "inter_op_threads": 1, + "intra_op_threads": 1, + "manual_seed": 0, + "mkldnn": False, + "score_quantum": "0.000001", + "use_deterministic_algorithms": False, + }, + "device": "cpu", + "dtype": "float32", + "encoded_at_freeze": False, + "evaluation_only": True, + "forbidden_inputs": [ + "idiomaticity_expectation", + "korkontzelos_manandhar_labels", + "magpie_labels", + "measurement_labels", + "operator_labels", + "residual_embeddings", + "residual_scores", + "semantic_yes_no", + "unbind_bucket", + ], + "input_construction": _INPUT_RULE, + "license": LICENSE_NAME, + "max_tokens": MAX_TOKENS, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "no_accuracy_claim": _NO_GOLD_RULE, + "no_gold_constituent_senses": True, + "pool": _POOL_RULE, + "positive_class_index": POSITIVE_CLASS_INDEX, + "pwn_mapping": { + "cross_version_map": False, + "manual_migration": False, + "nonunique_sense_key": AMBIGUOUS, + "prose": _MAPPING_RULE, + "source_wordnet": "PWN3.0", + "target_wordnet": "PWN3.0", + }, + "readiness": _READY_RULE, + "readiness_gate": { + "baseline_high_ready": BASELINE_HIGH_READY, + "baseline_secondary_ready": BASELINE_SECONDARY_READY, + "baseline_total_ready": BASELINE_TOTAL_READY, + "comparison": "strictly_greater", + "error_count_must_be": 0, + "invalid_output_count_must_be": 0, + "prose": _GATE_RULE, + "required_determinism": IDENTICAL, + }, + "research_question": _RESEARCH_QUESTION, + "residual_replay_authorized": False, + "rule": RULE_VERSION, + "runtime_integration": False, + "scoring": _SCORE_RULE, + "selected_source": "none", + "selection": _SELECTION_RULE, + "state_at_freeze": "SPEC_FROZEN", + "tie": { + "equal_top_scores": AMBIGUOUS, + "order_break": False, + "prose": _TIE_RULE, + }, + "tokenizer_revision": MODEL_REVISION, + } + + +def format_probability(value: float) -> str: + """Quantize one positive-class probability at the frozen quantum.""" + if value != value or value in {float("inf"), float("-inf")}: + raise RuntimeError("probability is not finite") + number = Decimal(value).quantize(SCORE_QUANTUM, rounding=ROUND_HALF_EVEN) + if number < 0 or number > 1: + raise RuntimeError("probability is outside the unit interval") + return format(number, "f") + + +def quoted_context(surface: str, constituent_index: int, constituent_surface: str) -> str: + """Quote the content token Extended Lesk already indexed. Structural tokens stay bare.""" + extraction = extract_constituents(surface) + content = extraction["content_constituents"] + if constituent_index < 0 or constituent_index >= len(content): + raise RuntimeError("constituent index is outside the content list") + if content[constituent_index] != constituent_surface: + raise RuntimeError("constituent surface does not match the frozen index") + seen = -1 + tokens = [] + for token in extraction["surface_tokens"]: + if lookup_key(token) in STRUCTURAL_TOKENS: + tokens.append(token) + continue + seen += 1 + if seen == constituent_index: + tokens.append('"' + token + '"') + else: + tokens.append(token) + if seen != len(content) - 1: + raise RuntimeError("quoted context did not land on the constituent") + return " ".join(tokens) + + +def gloss_lemma(lemmas: list[str], constituent: str, exceptions: dict[str, set[str]]) -> str | None: + """Pick the matching WordNet lemma. Display underscores as spaces.""" + keys = neighbor_keys(constituent, exceptions) + matched = [lemma for lemma in lemmas if lookup_key(lemma) in keys] + if not matched: + return None + own = lookup_key(constituent) + preferred = [lemma for lemma in matched if lookup_key(lemma) == own] or matched + chosen = sorted(preferred, key=lambda lemma: (lookup_key(lemma), lemma))[0] + return chosen.replace("_", " ") + + +def candidate_gloss_text(lemma: str, gloss_clause: str) -> str: + """Build the gloss side of a Sent-CLS pair.""" + return lemma + ": " + gloss_clause + + +def _decimal_score(text: str) -> Decimal: + value = Decimal(text) + if value != value.quantize(SCORE_QUANTUM): + raise RuntimeError("score is not at the frozen precision") + if value < 0 or value > 1: + raise RuntimeError("score is outside the unit interval") + return value + + +def _blank(status: str, code: str, scores: list[dict], confidence: str | None, margin: str | None) -> dict: + return { + "model_candidate_scores": scores, + "model_confidence": confidence, + "model_margin": margin, + "model_resolution_status": status, + "primary_evidence_code": code, + "selected_sense_key": None, + "selected_synset": None, + } + + +def resolve_model_scores( + candidate_synsets: list[str], + candidate_sense_keys: dict[str, list[str]], + positive_probabilities: dict[str, str] | None, + *, + overflow: bool = False, + error: str | None = None, +) -> dict: + """Apply the frozen abstention rule. The arguments are scores, not labels.""" + candidates = list(candidate_synsets) + if error: + return _blank(ERROR, "model_error", [], None, None) + if overflow: + return _blank(AMBIGUOUS, "context_overflow", [], None, None) + if len(candidates) < 2 or len(set(candidates)) != len(candidates): + return _blank(INVALID, "candidate_inventory_invalid", [], None, None) + if positive_probabilities is None or set(positive_probabilities) != set(candidates): + return _blank(INVALID, "invalid_model_output", [], None, None) + scored = {synset_id: _decimal_score(positive_probabilities[synset_id]) for synset_id in candidates} + rows = [] + for synset_id in sorted(scored): + keys = list(candidate_sense_keys.get(synset_id, [])) + rows.append( + { + "positive_probability": format(scored[synset_id], "f"), + "sense_keys": sorted(keys), + "synset": synset_id, + } + ) + ordered = sorted(scored.values(), reverse=True) + top = ordered[0] + second = ordered[1] + margin = top - second + confidence = format(top, "f") + margin_text = format(margin, "f") + winners = [synset_id for synset_id, score in scored.items() if score == top] + if len(winners) != 1: + return _blank(AMBIGUOUS, "model_score_tie", rows, confidence, margin_text) + if top < MINIMUM_CONFIDENCE or margin < MINIMUM_MARGIN: + return _blank(AMBIGUOUS, "model_abstention", rows, confidence, margin_text) + selected = winners[0] + if selected not in candidates: + return _blank(INVALID, "invalid_model_output", rows, confidence, margin_text) + keys = sorted(candidate_sense_keys.get(selected, [])) + if len(keys) != 1: + return _blank(AMBIGUOUS, "pwn30_sense_key_not_unique", rows, confidence, margin_text) + return { + "model_candidate_scores": rows, + "model_confidence": confidence, + "model_margin": margin_text, + "model_resolution_status": RESOLVED, + "primary_evidence_code": "model_margin", + "selected_sense_key": keys[0], + "selected_synset": selected, + } + + +def overlay_status(frozen_status: str, frozen_method: str, model_status: str | None) -> str: + """Combine one frozen tier with an optional tier-3 status. Tier 3 cannot replace a decision.""" + if frozen_status == EXACT: + if model_status is not None: + raise RuntimeError("tier 3 overrode an exact constituent") + return EXACT + if frozen_status == RESOLVED: + if model_status is not None: + raise RuntimeError("tier 3 overrode a resolved constituent") + return LESK_RESOLVED + if frozen_status == UNRESOLVED: + if model_status is not None: + raise RuntimeError("tier 3 overrode an unresolved constituent") + return UNRESOLVED + if frozen_status != AMBIGUOUS: + raise RuntimeError("unknown frozen constituent status") + if frozen_method != "EXTENDED_LESK_V1": + if model_status is not None: + raise RuntimeError("tier 3 overrode a non-lesk constituent") + return AMBIGUOUS + if model_status is None: + raise RuntimeError("ambiguous lesk constituent has no tier 3 result") + if model_status == RESOLVED: + return MODEL_RESOLVED + if model_status in {AMBIGUOUS, INVALID, ERROR}: + return model_status + raise RuntimeError("unknown model status") + + +def project_row_status(statuses: list[str]) -> str: + """Project one row. Fewer than two content constituents stays unknown.""" + if len(statuses) < 2 or any(status not in READY_CONSTITUENT for status in statuses): + return ROW_UNKNOWN + return RESIDUAL_READY + + +def coverage_gate( + *, + high_ready: int, + secondary_ready: int, + total_ready: int, + invalid_output_count: int, + error_count: int, + determinism: str, +) -> dict: + """Apply the preregistered coverage gate. The thresholds are not fit to this run.""" + counts = (high_ready, secondary_ready, total_ready, invalid_output_count, error_count) + if min(counts) < 0: + raise RuntimeError("negative coverage count") + if determinism != IDENTICAL: + status = NOT_DETERMINISTIC + transition = NEXT_DETERMINISM + elif invalid_output_count > 0 or error_count > 0: + status = REJECTED + transition = NEXT_REVISION + elif ( + high_ready > BASELINE_HIGH_READY + and secondary_ready > BASELINE_SECONDARY_READY + and total_ready > BASELINE_TOTAL_READY + and invalid_output_count == 0 + and error_count == 0 + and determinism == IDENTICAL + ): + status = PROMISING + transition = NEXT_PROMISING + else: + status = INSUFFICIENT + transition = NEXT_REVISION + return { + "candidate_status": status, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "CANDIDATE_EVALUATED", + } + + +def _confidence_bin(value: Decimal) -> str: + if value < Decimal("0.50"): + return "below_0.50" + if value < Decimal("0.60"): + return "0.50_to_0.60" + if value < Decimal("0.70"): + return "0.60_to_0.70" + if value < Decimal("0.80"): + return "0.70_to_0.80" + if value < Decimal("0.90"): + return "0.80_to_0.90" + return "0.90_to_1.00" + + +def _margin_bin(value: Decimal) -> str: + if value == 0: + return "exact_tie" + if value < Decimal("0.10"): + return "below_0.10" + if value < Decimal("0.25"): + return "0.10_to_0.25" + if value < Decimal("0.50"): + return "0.25_to_0.50" + return "0.50_or_more" + + +def summarize_confidence(rows: list[dict]) -> dict: + """Summarize tier-3 scores. Operator buckets are not an argument.""" + confidence_bins = { + "0.50_to_0.60": 0, + "0.60_to_0.70": 0, + "0.70_to_0.80": 0, + "0.80_to_0.90": 0, + "0.90_to_1.00": 0, + "below_0.50": 0, + } + margin_bins = { + "0.10_to_0.25": 0, + "0.25_to_0.50": 0, + "0.50_or_more": 0, + "below_0.10": 0, + "exact_tie": 0, + } + by_count: dict[str, dict[str, int]] = {} + resolved = 0 + abstained = 0 + invalid = 0 + errors = 0 + for row in rows: + status = row["model_resolution_status"] + if status == RESOLVED: + resolved += 1 + elif status == AMBIGUOUS: + abstained += 1 + elif status == INVALID: + invalid += 1 + elif status == ERROR: + errors += 1 + else: + raise RuntimeError("unknown model status in the confidence summary") + key = str(len(row["candidate_pwn30_synsets"])) + bucket = by_count.setdefault( + key, + {"abstained": 0, "attempts": 0, "error": 0, "invalid": 0, "resolved": 0}, + ) + bucket["attempts"] += 1 + if status == RESOLVED: + bucket["resolved"] += 1 + elif status == AMBIGUOUS: + bucket["abstained"] += 1 + elif status == INVALID: + bucket["invalid"] += 1 + else: + bucket["error"] += 1 + if row["model_confidence"] is not None: + confidence_bins[_confidence_bin(Decimal(row["model_confidence"]))] += 1 + if row["model_margin"] is not None: + margin_bins[_margin_bin(Decimal(row["model_margin"]))] += 1 + return { + "abstained": abstained, + "by_candidate_count": dict(sorted(by_count.items(), key=lambda item: int(item[0]))), + "confidence_bins": confidence_bins, + "error": errors, + "invalid": invalid, + "margin_bins": margin_bins, + "resolved": resolved, + } diff --git a/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py new file mode 100644 index 00000000..f81eabc1 --- /dev/null +++ b/scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py @@ -0,0 +1,885 @@ +"""Evaluate one pinned GlossBERT checkpoint on Extended Lesk ties. + +The specification is hashed before any Hyperlex constituent is scored. Operator +labels are read only after the readiness projection is hashed. Residual scores +are not loaded and are not recomputed. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from collections import Counter, defaultdict +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["NUMEXPR_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.constituent_sense_resolution_v1_replay import ( + ANALYSIS_PATH as RESOLVER_ANALYSIS_PATH, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + DECISION_PATH as RESOLVER_DECISION_PATH, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EXPECTED as RESOLVER_EXPECTED, +) +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EVENTS, + EVIDENCE_MANIFEST, + HYPERLEX, + LEDGER_FILE, + LIMITATION_PATH, + OPERATOR_COUNTS, + OPERATORS, + PROCEDURE_PATH as RESOLVER_PROCEDURE_PATH, + REPLAY_PATH as RESOLVER_REPLAY_PATH, + REVIEW_PATH, + SOURCE, + SPEC_PATH as RESOLVER_SPEC_PATH, + TRACKER, + WORDNET, + build_catalog, + load_sealed_rows, + refuse, + sha256, + write_json, + write_jsonl, +) +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.model_based_wsd_candidate_v1 import ( + BASELINE_HIGH_READY, + BASELINE_SECONDARY_READY, + BASELINE_TOTAL_READY, + CANONICAL_SOURCE, + IDENTICAL, + LICENSE_NAME, + MAX_TOKENS, + MODEL_FAMILY, + MODEL_NAME, + MODEL_REVISION, + POSITIVE_CLASS_INDEX, + REJECTED, + RULE_VERSION, + candidate_gloss_text, + candidate_policy, + coverage_gate, + format_probability, + gloss_lemma, + overlay_status, + project_row_status, + quoted_context, + resolve_model_scores, + summarize_confidence, +) +from hyperlexical.semantic_compositionality_residual import neighbor_keys +from hyperlexical.unbind_sense_screen_v1 import load_exceptions + +MODEL_DIR = ( + Path("/home/morpheus/hlx-private/eval-reserve-20260926/acquisition/sources/glossbert") + / MODEL_REVISION +) +CODE_LICENSE = ( + Path("/home/morpheus/hlx-private/eval-reserve-20260926/acquisition/sources/glossbert") + / "ORIGINAL_CODE_MIT_LICENSE.txt" +) +SPEC_PATH = SOURCE / "MODEL_BASED_WSD_CANDIDATE_SPEC.json" +PROVENANCE_PATH = SOURCE / "MODEL_BASED_WSD_PROVENANCE.json" +LICENSE_PATH = SOURCE / "MODEL_BASED_WSD_LICENSE_RECEIPT.json" +RAW_PATH = SOURCE / "MODEL_BASED_WSD_RAW_OUTPUT.jsonl" +RESOLUTION_PATH = SOURCE / "MODEL_BASED_WSD_RESOLUTION.jsonl" +PROJECTION_PATH = SOURCE / "MODEL_BASED_WSD_READINESS_PROJECTION.json" +ANALYSIS_PATH = SOURCE / "MODEL_BASED_WSD_DEVELOPMENT_ANALYSIS.json" +DECISION_PATH = SOURCE / "MODEL_BASED_WSD_CANDIDATE_DECISION.json" + +CURRENT_TRACKER = "c72e55c8096143b8675d4aa995c5d4b19de95d256dd234d32a421ba098bc7d4f" +WEIGHTS = { + "config.json": "70f859c899543d9c8f822e64201e1a77530f0de12aabb0daf04b9570f538b638", + "pytorch_model.bin": "60706c7618f8ccbfa7d0a6d1d1009765a7146ea5f4232924ed9f1c46d521c898", + "README.md": "27577f2e0742d2e91ccad99778cc15fe521386de53d7bfe8acedfa0a78f8c186", + "special_tokens_map.json": "303df45a03609e4ead04bc3dc1536d0ab19b5358db685b6f3da123d05ec200e3", + "tokenizer_config.json": "09e49d0e788d25991da77d37b10eaa6a86a4e94e2127de8bedc94eb45baf2d84", + "vocab.txt": "07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3", +} +CODE_LICENSE_SHA = "333e83b83ae30a1c9af100ea742de8ab34ac9d6d8c8aae52e0bbec47cf9df80f" +RUNTIME = { + "numpy": "2.5.3", + "python": "3.12.3", + "tokenizers": "0.23.2", + "torch": "2.14.0+cpu", + "transformers": "5.17.0", +} +FINANCIAL_CONTEXT = 'She deposited the check at the "bank" yesterday.' +RIVER_CONTEXT = 'They picnicked on the grassy "bank" of the river.' +FINANCIAL_SYNSET = "noun:08420278" +RIVER_SYNSET = "noun:09213565" +POOL_ATTEMPTS = 249 +SS_POS = {"1": "noun", "2": "verb", "3": "adj", "4": "adv", "5": "adj"} + +_THIRD_PARTY = ( + "The Hugging Face card is a third-party upload. It is not the authors' " + "Google Drive checkpoint. The card declares MIT. The original GlossBERT " + "code repository is MIT. No API key is required." +) +_CODE_REPOSITORY = "https://github.com/HSLCY/GlossBERT" + +EXPECTED = dict(RESOLVER_EXPECTED) +EXPECTED[TRACKER] = CURRENT_TRACKER +EXPECTED[REVIEW_PATH] = "1c1b69856dd88567167fd5c958cd8db6d39ab9ec74a4e0ed3e667a521c82e6fa" +EXPECTED[LIMITATION_PATH] = "fc8839c15a7638b2bfca1cf0548bfb4d5f433434bae0fea2944a528dd15d6142" +EXPECTED[RESOLVER_SPEC_PATH] = "176e6219ddc3127814a25d39ad26e3571817f7ea8323d685e081d2e0fd867acb" +EXPECTED[RESOLVER_PROCEDURE_PATH] = "9f76b64aa6aac09bd56ca9cc8a847cda12b31a58eea8c4b51426733917d54248" +EXPECTED[RESOLVER_REPLAY_PATH] = "a0c707ab55e02f627a698c33ddc0ca398e34aa0bafc19b13e72422bc26d97d0a" +EXPECTED[RESOLVER_ANALYSIS_PATH] = "6d47610014f74094394355a50419fe24441103bec09458b6f25ea392fb57151c" +EXPECTED[RESOLVER_DECISION_PATH] = "ff8b90ebe5d48151dc68ddbf676e1f27d4cee5e3be2730f999c9b089f1392caa" +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py"] = ( + "927fb5c6f627d5488ecbc66779595812e8c2154ad47d6a07db1e377639f36fc6" +) +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/constituent_sense_resolution_v1_replay.py"] = ( + "d8cd1c1719d9764cb0a77ad51a2d305268181b42294cd1adddd3ff9d4ea57e0b" +) +EXPECTED[HYPERLEX / "tests/shadow/test_constituent_sense_resolution_v1.py"] = ( + "104ac0edc643b76df4ca1fe905e30ed96c7417dddfc14010b23d3b3c613adeb9" +) + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def load_jsonl(path: Path) -> list[dict]: + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line] + + +def load_sense_index(path: Path) -> dict[str, list[dict]]: + index = defaultdict(list) + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + parts = line.split() + if len(parts) < 2 or "%" not in parts[0]: + continue + sense_key, offset = parts[0], parts[1] + lemma, _, rest = sense_key.partition("%") + pos = SS_POS.get(rest.split(":", 1)[0]) + if pos is None: + continue + index[f"{pos}:{offset.zfill(8)}"].append({"lemma": lemma, "sense_key": sense_key}) + return dict(index) + + +def sense_keys_for(synset_id: str, constituent: str, exceptions: dict, sense_index: dict) -> list[str]: + allowed = neighbor_keys(constituent, exceptions) + found = [ + item["sense_key"] + for item in sense_index.get(synset_id, []) + if lookup_key(item["lemma"]) in allowed + ] + return sorted(set(found)) + + +def assert_weights() -> dict: + if not MODEL_DIR.is_dir(): + refuse("glossbert directory is missing") + found = {} + for name, expected in WEIGHTS.items(): + path = MODEL_DIR / name + digest = sha256(path) + if digest != expected: + refuse(f"weight hash mismatch: {name}") + found[name] = digest + if not CODE_LICENSE.is_file() or sha256(CODE_LICENSE) != CODE_LICENSE_SHA: + refuse("original code license file mismatch") + readme = (MODEL_DIR / "README.md").read_text(encoding="utf-8") + if "license: mit" not in readme.lower(): + refuse("model card does not declare mit") + return found + + +def license_payload(weights: dict) -> dict: + return { + "api_required": False, + "artifact_license_file_present": False, + "canonical_source": CANONICAL_SOURCE, + "card_license": "mit", + "code_license_sha256": CODE_LICENSE_SHA, + "datasets": ["SemCor3.0"], + "evaluation_permitted": True, + "json_schema_document": None, + "license_name": LICENSE_NAME, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "original_code_license": "MIT", + "original_code_repository": _CODE_REPOSITORY, + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "schema": "hyperlex.model_based_wsd_candidate_v1_license_receipt.v1", + "source_license_unresolved": False, + "third_party_note": _THIRD_PARTY, + "third_party_upload": True, + "vocab_txt_sha256": weights["vocab.txt"], + } + + +def spec_payload(weights: dict, license_sha: str) -> dict: + return { + "artifacts": { + "config_json_sha256": weights["config.json"], + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "readme_md_sha256": weights["README.md"], + "special_tokens_map_json_sha256": weights["special_tokens_map.json"], + "tokenizer_config_json_sha256": weights["tokenizer_config.json"], + "vocab_txt_sha256": weights["vocab.txt"], + }, + "expected_tier3_attempts": POOL_ATTEMPTS, + "json_schema_document": None, + "license_receipt_sha256": license_sha, + "polarity_control": { + "financial_context": FINANCIAL_CONTEXT, + "financial_synset": FINANCIAL_SYNSET, + "hyperlex_rows_used": False, + "river_context": RIVER_CONTEXT, + "river_synset": RIVER_SYNSET, + }, + "policy": candidate_policy(), + "prior_resolver_replay_sha256": EXPECTED[RESOLVER_REPLAY_PATH], + "residual_scores_included": False, + "runtime": dict(RUNTIME), + "schema": "hyperlex.model_based_wsd_candidate_v1_spec.v1", + "selected_source": "none", + } + + +def prepare_pool(frozen: list[dict], by_id: dict, exceptions: dict, sense_index: dict) -> list[dict]: + attempts = [row for row in frozen if row["constituent_index"] is not None] + counts = Counter(row["resolution_status"] for row in attempts) + if counts["EXACT"] != 64 or counts["RESOLVED"] != 121 or counts["AMBIGUOUS"] != 249 or counts["UNRESOLVED"] != 70: + refuse("frozen resolver counts drifted") + pool = [] + for row in attempts: + if row["resolution_status"] != "AMBIGUOUS": + continue + if row["resolution_method"] != "EXTENDED_LESK_V1": + refuse("ambiguous constituent is not an extended lesk tie") + pairs = [] + for synset_id in row["candidate_synsets"]: + record = by_id.get(synset_id) + if record is None: + refuse(f"missing candidate synset {synset_id}") + lemma = gloss_lemma(record["lemmas"], row["constituent_surface"], exceptions) + if lemma is None: + refuse(f"candidate lemma missing for {synset_id}") + pairs.append( + { + "gloss_text": candidate_gloss_text(lemma, record["gloss_first"]), + "sense_keys": sense_keys_for(synset_id, row["constituent_surface"], exceptions, sense_index), + "synset": synset_id, + } + ) + pool.append( + { + "candidate_pwn30_synsets": list(row["candidate_synsets"]), + "candidate_sense_keys": {item["synset"]: list(item["sense_keys"]) for item in pairs}, + "constituent_index": row["constituent_index"], + "constituent_pos": row["constituent_pos"], + "constituent_surface": row["constituent_surface"], + "context": quoted_context(row["parent_surface"], row["constituent_index"], row["constituent_surface"]), + "pairs": pairs, + "parent_row_id": row["parent_row_id"], + "parent_surface": row["parent_surface"], + "parent_synset": row["parent_synset"], + "prior_lesk_margin": row["margin"], + "prior_lesk_second_score": row["second_score"], + "prior_lesk_top_score": row["top_score"], + "prior_resolution_status": "AMBIGUOUS", + } + ) + if len(pool) != POOL_ATTEMPTS: + refuse(f"tier 3 pool is {len(pool)}") + return pool + + +def import_runtime(): + import numpy + import torch + import tokenizers + import transformers + from transformers import AutoTokenizer, BertForSequenceClassification + + versions = { + "numpy": numpy.__version__, + "python": ".".join(str(part) for part in sys.version_info[:3]), + "tokenizers": tokenizers.__version__, + "torch": torch.__version__, + "transformers": transformers.__version__, + } + if versions != RUNTIME: + refuse(f"runtime drift: {versions}") + torch.set_num_threads(1) + torch.set_num_interop_threads(1) + torch.backends.mkldnn.enabled = False + torch.use_deterministic_algorithms(False) + torch.manual_seed(0) + tokenizer = AutoTokenizer.from_pretrained(MODEL_DIR, local_files_only=True) + model = BertForSequenceClassification.from_pretrained(MODEL_DIR, local_files_only=True) + model.eval() + model.to("cpu") + if model.num_labels != 2: + refuse("classifier head is not binary") + return torch, tokenizer, model + + +def score_pair(torch, tokenizer, model, context: str, gloss_text: str) -> dict: + encoded = tokenizer(context, gloss_text, truncation=False, padding=False, return_tensors="pt") + length = int(encoded["input_ids"].shape[-1]) + if length > MAX_TOKENS: + return {"overflow": True, "positive_probability": None, "token_count": length} + with torch.inference_mode(): + logits = model(**encoded).logits[0] + probability = torch.softmax(logits, dim=-1)[POSITIVE_CLASS_INDEX] + return { + "overflow": False, + "positive_probability": format_probability(float(probability)), + "token_count": length, + } + + +def score_context(torch, tokenizer, model, context: str, pairs: list[dict]) -> dict: + scored = [] + overflow = False + error = None + try: + for pair in pairs: + result = score_pair(torch, tokenizer, model, context, pair["gloss_text"]) + overflow = overflow or result["overflow"] + scored.append( + { + "gloss_text": pair["gloss_text"], + "overflow": result["overflow"], + "positive_probability": result["positive_probability"], + "sense_keys": list(pair["sense_keys"]), + "synset": pair["synset"], + "token_count": result["token_count"], + } + ) + except Exception as exc: + error = type(exc).__name__ + scored = [] + overflow = False + return {"error": error, "overflow": overflow, "scored_pairs": scored} + + +def polarity_control(torch, tokenizer, model, by_id: dict, sense_index: dict) -> dict: + synsets = sorted( + { + synset_id + for synset_id, items in sense_index.items() + if any(item["sense_key"].startswith("bank%1:") and item["lemma"] == "bank" for item in items) + } + ) + pairs = [] + for synset_id in synsets: + record = by_id.get(synset_id) + if record is None: + refuse(f"missing polarity synset {synset_id}") + pairs.append( + { + "gloss_text": candidate_gloss_text("bank", record["gloss_first"]), + "sense_keys": [item["sense_key"] for item in sense_index[synset_id] if item["lemma"] == "bank"], + "synset": synset_id, + } + ) + report = {"hyperlex_rows_used": False} + for name, context, expected in ( + ("financial", FINANCIAL_CONTEXT, FINANCIAL_SYNSET), + ("river", RIVER_CONTEXT, RIVER_SYNSET), + ): + scored = score_context(torch, tokenizer, model, context, pairs) + probabilities = { + row["synset"]: row["positive_probability"] + for row in scored["scored_pairs"] + if row["positive_probability"] is not None + } + decision = resolve_model_scores( + [pair["synset"] for pair in pairs], + {pair["synset"]: sorted(pair["sense_keys"]) for pair in pairs}, + None if scored["error"] or scored["overflow"] else probabilities, + overflow=scored["overflow"], + error=scored["error"], + ) + report[name] = { + "expected_synset": expected, + "selected_synset": decision["selected_synset"], + "status": decision["model_resolution_status"], + "top_probability": decision["model_confidence"], + } + if decision["model_resolution_status"] != "RESOLVED" or decision["selected_synset"] != expected: + report["passed"] = False + return report + report["passed"] = True + report["positive_class_index"] = POSITIVE_CLASS_INDEX + return report + + +def infer_pass(torch, tokenizer, model, pool: list[dict], spec_sha: str) -> list[dict]: + torch.manual_seed(0) + rows = [] + for index, item in enumerate(pool): + scored = score_context(torch, tokenizer, model, item["context"], item["pairs"]) + rows.append( + { + "candidate_pwn30_synsets": list(item["candidate_pwn30_synsets"]), + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "context": item["context"], + "error": scored["error"], + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "overflow": scored["overflow"], + "pairs": scored["scored_pairs"], + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "record_kind": "hyperlex_tier3_constituent", + "spec_sha256": spec_sha, + } + ) + if index % 25 == 0: + print(f"scored {index}", file=sys.stderr, flush=True) + return rows + + +def resolution_rows(pool: list[dict], raw_rows: list[dict]) -> list[dict]: + resolved = [] + for item, raw in zip(pool, raw_rows, strict=True): + probabilities = None + if not raw["error"] and not raw["overflow"]: + probabilities = {pair["synset"]: pair["positive_probability"] for pair in raw["pairs"]} + decision = resolve_model_scores( + item["candidate_pwn30_synsets"], + item["candidate_sense_keys"], + probabilities, + overflow=raw["overflow"], + error=raw["error"], + ) + if decision["selected_synset"] is not None and decision["selected_synset"] not in item["candidate_pwn30_synsets"]: + refuse("selected synset left the candidate set") + resolved.append( + { + "candidate_pwn30_synsets": list(item["candidate_pwn30_synsets"]), + "candidate_sense_keys": item["candidate_sense_keys"], + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "model_candidate_scores": decision["model_candidate_scores"], + "model_confidence": decision["model_confidence"], + "model_margin": decision["model_margin"], + "model_name": MODEL_NAME, + "model_resolution_status": decision["model_resolution_status"], + "model_revision": MODEL_REVISION, + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "primary_evidence_code": decision["primary_evidence_code"], + "prior_lesk_margin": item["prior_lesk_margin"], + "prior_lesk_second_score": item["prior_lesk_second_score"], + "prior_lesk_top_score": item["prior_lesk_top_score"], + "prior_resolution_status": "AMBIGUOUS", + "selected_sense_key": decision["selected_sense_key"], + "selected_synset": decision["selected_synset"], + "spec_sha256": raw["spec_sha256"], + } + ) + return resolved + + +def project_rows(frozen: list[dict], resolutions: list[dict]) -> tuple[list[dict], dict]: + tier3 = {(row["parent_row_id"], row["constituent_index"]): row for row in resolutions} + if len(tier3) != len(resolutions): + refuse("duplicate tier 3 keys") + grouped = defaultdict(list) + for row in frozen: + grouped[row["parent_row_id"]].append(row) + projected = [] + combined = Counter() + for sealed in load_sealed_rows(): + items = [row for row in grouped[sealed["row_id"]] if row["constituent_index"] is not None] + items.sort(key=lambda row: row["constituent_index"]) + statuses = [] + for item in items: + key = (item["parent_row_id"], item["constituent_index"]) + model_status = None + if key in tier3: + if item["resolution_status"] != "AMBIGUOUS" or item["resolution_method"] != "EXTENDED_LESK_V1": + refuse("tier 3 key is not an extended lesk tie") + model_status = tier3[key]["model_resolution_status"] + status = overlay_status(item["resolution_status"], item["resolution_method"], model_status) + statuses.append(status) + combined[status] += 1 + projected.append( + { + "constituent_statuses": statuses, + "content_count": len(statuses), + "parent_row_id": sealed["row_id"], + "projected_status": project_row_status(statuses), + } + ) + if len(projected) != 225: + refuse("projection does not cover 225 rows") + if sum(combined.values()) != 504: + refuse("combined constituent count is not 504") + counts = Counter(row["projected_status"] for row in projected) + if counts["RESIDUAL_READY"] + counts["UNKNOWN"] != 225: + refuse("projected row count drifted") + return projected, { + "combined_constituent_counts": { + "AMBIGUOUS": combined["AMBIGUOUS"], + "ERROR": combined["ERROR"], + "EXACT": combined["EXACT"], + "INVALID": combined["INVALID"], + "LESK_RESOLVED": combined["LESK_RESOLVED"], + "MODEL_RESOLVED": combined["MODEL_RESOLVED"], + "UNRESOLVED": combined["UNRESOLVED"], + }, + "row_counts": {"RESIDUAL_READY": counts["RESIDUAL_READY"], "UNKNOWN": counts["UNKNOWN"]}, + } + + +def update_tracker(outcome: dict, hashes: dict) -> str: + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if sha256(TRACKER) != CURRENT_TRACKER: + refuse("tracker hash drifted before update") + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = CURRENT_TRACKER + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["model_based_wsd_rule"] = RULE_VERSION + tracker["model_based_wsd_state"] = "CANDIDATE_EVALUATED" + tracker["model_based_wsd_candidate_status"] = outcome["candidate_status"] + tracker["model_based_wsd_model_name"] = MODEL_NAME + tracker["model_based_wsd_model_revision"] = MODEL_REVISION + tracker["model_based_wsd_runtime_integration"] = False + tracker["model_based_wsd_spec_sha256"] = hashes.get("spec_sha256") + tracker["model_based_wsd_provenance_sha256"] = hashes.get("provenance_sha256") + tracker["model_based_wsd_license_receipt_sha256"] = hashes.get("license_sha256") + tracker["model_based_wsd_raw_output_sha256"] = hashes.get("raw_sha256") + tracker["model_based_wsd_resolution_sha256"] = hashes.get("resolution_sha256") + tracker["model_based_wsd_readiness_projection_sha256"] = hashes.get("projection_sha256") + tracker["model_based_wsd_analysis_sha256"] = hashes.get("analysis_sha256") + tracker["model_based_wsd_decision_sha256"] = hashes.get("decision_sha256") + tracker["model_based_wsd_residual_ready_rows"] = outcome.get("total_ready") + tracker["model_based_wsd_high_ready"] = outcome.get("high_ready") + tracker["model_based_wsd_secondary_ready"] = outcome.get("secondary_ready") + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_threshold_eligible"] = False + tracker["residual_source_selection_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["constituent_sense_resolution_state"] = "DEVELOPMENT_ANALYZED" + tracker["constituent_sense_resolution_candidate_status"] = "COVERAGE_INSUFFICIENT" + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = outcome["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + return tracker_sha + + +def write_decision(outcome: dict, hashes: dict, analysis_sha: str | None) -> str: + payload = { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "determinism": outcome.get("determinism"), + "high_ready": outcome.get("high_ready"), + "json_schema_document": None, + "license_receipt_sha256": hashes.get("license_sha256"), + "measurement_eligible": False, + "measurement_sample_drawn": False, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "next_legal_transition": outcome["next_legal_transition"], + "next_transition_authorized": False, + "projection_sha256": hashes.get("projection_sha256"), + "provenance_sha256": hashes.get("provenance_sha256"), + "raw_output_sha256": hashes.get("raw_sha256"), + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "residual_state": "CANDIDATE_DISTRIBUTION_FROZEN", + "residual_threshold_created": False, + "resolution_sha256": hashes.get("resolution_sha256"), + "rule": RULE_VERSION, + "runtime_integration": False, + "schema": "hyperlex.model_based_wsd_candidate_v1_decision.v1", + "secondary_ready": outcome.get("secondary_ready"), + "select_005_authorized": False, + "selected_source": "none", + "spec_sha256": hashes.get("spec_sha256"), + "state": "CANDIDATE_EVALUATED", + "threshold_eligible": False, + "total_ready": outcome.get("total_ready"), + "yes_no_emitted": False, + } + return write_json(DECISION_PATH, payload) + + +def main() -> None: + check_sealed() + weights = assert_weights() + license_sha = write_json(LICENSE_PATH, license_payload(weights)) + spec = spec_payload(weights, license_sha) + spec_text = json.dumps(spec, sort_keys=True) + if "0.2139784896" in spec_text or "operator_bucket" in spec_text: + refuse("spec contains a residual score or an operator field") + spec_sha = write_json(SPEC_PATH, spec) + if sha256(SPEC_PATH) != spec_sha: + refuse("spec hash drifted at freeze") + print(f"SPEC_FROZEN {spec_sha}", file=sys.stderr, flush=True) + sealed_rows = load_sealed_rows() + surfaces = {row["surface"] for row in sealed_rows} + if FINANCIAL_CONTEXT in surfaces or RIVER_CONTEXT in surfaces: + refuse("polarity control collides with a development surface") + frozen = load_jsonl(RESOLVER_REPLAY_PATH) + if sha256(RESOLVER_REPLAY_PATH) != EXPECTED[RESOLVER_REPLAY_PATH]: + refuse("resolver replay changed while loading") + exceptions = load_exceptions(WORDNET) + by_id, _index = build_catalog() + sense_index = load_sense_index(WORDNET / "index.sense") + pool = prepare_pool(frozen, by_id, exceptions, sense_index) + torch, tokenizer, model = import_runtime() + polarity = polarity_control(torch, tokenizer, model, by_id, sense_index) + hashes = {"license_sha256": license_sha, "spec_sha256": spec_sha} + if not polarity["passed"]: + provenance = { + "hyperlex_scoring_started": False, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "polarity_control": polarity, + "schema": "hyperlex.model_based_wsd_candidate_v1_provenance.v1", + "spec_sha256": spec_sha, + } + hashes["provenance_sha256"] = write_json(PROVENANCE_PATH, provenance) + outcome = { + "candidate_status": REJECTED, + "determinism": None, + "next_legal_transition": "MODEL_BASED_WSD_CANDIDATE_REVISION_AUTHORIZATION", + } + hashes["analysis_sha256"] = None + hashes["decision_sha256"] = write_decision(outcome, hashes, None) + update_tracker(outcome, hashes) + print(json.dumps({"candidate_status": REJECTED, "spec_sha256": spec_sha}, sort_keys=True)) + return + provenance = { + "canonical_source": CANONICAL_SOURCE, + "device": "cpu", + "dtype": "float32", + "hyperlex_scoring_started": False, + "json_schema_document": None, + "license_receipt_sha256": license_sha, + "model_family": MODEL_FAMILY, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "polarity_control_passed": True, + "polarity_financial_winner": polarity["financial"]["selected_synset"], + "polarity_river_winner": polarity["river"]["selected_synset"], + "positive_class_index": POSITIVE_CLASS_INDEX, + "residual_embeddings_used": False, + "runtime": dict(RUNTIME), + "schema": "hyperlex.model_based_wsd_candidate_v1_provenance.v1", + "spec_sha256": spec_sha, + "third_party_upload": True, + "weights": { + "pytorch_model_bin_sha256": weights["pytorch_model.bin"], + "vocab_txt_sha256": weights["vocab.txt"], + }, + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + hashes["provenance_sha256"] = provenance_sha + if sha256(SPEC_PATH) != spec_sha: + refuse("spec changed after provenance") + print("HYPERLEX_SCORING_START", file=sys.stderr, flush=True) + first = infer_pass(torch, tokenizer, model, pool, spec_sha) + print("PASS_1_DONE", file=sys.stderr, flush=True) + second = infer_pass(torch, tokenizer, model, pool, spec_sha) + print("PASS_2_DONE", file=sys.stderr, flush=True) + first_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in first) + second_text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in second) + if first_text != second_text: + mismatch = { + "determinism": "NOT_DETERMINISTIC", + "first_sha256": hashlib.sha256(first_text.encode("utf-8")).hexdigest(), + "record_kind": "determinism_mismatch", + "second_sha256": hashlib.sha256(second_text.encode("utf-8")).hexdigest(), + "spec_sha256": spec_sha, + } + hashes["raw_sha256"] = write_jsonl(RAW_PATH, [mismatch]) + outcome = { + "candidate_status": "NOT_DETERMINISTIC", + "determinism": "NOT_DETERMINISTIC", + "next_legal_transition": "MODEL_BASED_WSD_DETERMINISM_REVIEW_AUTHORIZATION", + } + hashes["decision_sha256"] = write_decision(outcome, hashes, None) + update_tracker(outcome, hashes) + print(json.dumps(outcome, sort_keys=True)) + return + raw_sha = write_jsonl(RAW_PATH, first) + hashes["raw_sha256"] = raw_sha + resolutions = resolution_rows(pool, first) + if any(row["prior_resolution_status"] != "AMBIGUOUS" for row in resolutions): + refuse("tier 3 row was not previously ambiguous") + resolution_sha = write_jsonl(RESOLUTION_PATH, resolutions) + hashes["resolution_sha256"] = resolution_sha + if sha256(SPEC_PATH) != spec_sha: + refuse("scoring mutated the spec") + projected, summary = project_rows(frozen, resolutions) + tier3_counts = Counter(row["model_resolution_status"] for row in resolutions) + confidence = summarize_confidence(resolutions) + projection = { + "combined_constituent_counts": summary["combined_constituent_counts"], + "confidence": confidence, + "determinism": IDENTICAL, + "json_schema_document": None, + "operator_labels_included": False, + "raw_output_sha256": raw_sha, + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "resolution_sha256": resolution_sha, + "row_counts": summary["row_counts"], + "rows": projected, + "schema": "hyperlex.model_based_wsd_candidate_v1_readiness_projection.v1", + "spec_sha256": spec_sha, + "tier3_attempts": len(resolutions), + "tier3_counts": { + "AMBIGUOUS": tier3_counts["AMBIGUOUS"], + "ERROR": tier3_counts["ERROR"], + "INVALID": tier3_counts["INVALID"], + "RESOLVED": tier3_counts["RESOLVED"], + }, + } + projection_text = json.dumps(projection, sort_keys=True) + if "operator_bucket" in projection_text or "HIGH" in projection["row_counts"]: + refuse("projection carries an operator field") + projection_sha = write_json(PROJECTION_PATH, projection) + hashes["projection_sha256"] = projection_sha + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {row["row_id"]: row["operator_bucket"] for row in manifest["rows"]} + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + ready_by = {name: {"RESIDUAL_READY": 0, "UNKNOWN": 0} for name in OPERATORS} + for row in projected: + ready_by[buckets[row["parent_row_id"]]][row["projected_status"]] += 1 + total_ready = summary["row_counts"]["RESIDUAL_READY"] + outcome = coverage_gate( + high_ready=ready_by["HIGH"]["RESIDUAL_READY"], + secondary_ready=ready_by["SECONDARY"]["RESIDUAL_READY"], + total_ready=total_ready, + invalid_output_count=tier3_counts["INVALID"], + error_count=tier3_counts["ERROR"], + determinism=IDENTICAL, + ) + outcome["determinism"] = IDENTICAL + outcome["high_ready"] = ready_by["HIGH"]["RESIDUAL_READY"] + outcome["secondary_ready"] = ready_by["SECONDARY"]["RESIDUAL_READY"] + outcome["total_ready"] = total_ready + analysis = { + "baseline": { + "high_ready": BASELINE_HIGH_READY, + "secondary_ready": BASELINE_SECONDARY_READY, + "total_ready": BASELINE_TOTAL_READY, + }, + "candidate_status": outcome["candidate_status"], + "change_versus_lexical_baseline": { + "high_ready_delta": outcome["high_ready"] - BASELINE_HIGH_READY, + "secondary_ready_delta": outcome["secondary_ready"] - BASELINE_SECONDARY_READY, + "total_ready_delta": total_ready - BASELINE_TOTAL_READY, + }, + "combined_constituent_counts": summary["combined_constituent_counts"], + "confidence": confidence, + "determinism": IDENTICAL, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "operator_labels_joined_after_projection_was_hashed": True, + "operator_labels_used_during_inference": False, + "projection_sha256": projection_sha, + "ready_by_operator": ready_by, + "residual_embeddings_used": False, + "residual_replay_performed": False, + "residual_scores_recomputed": False, + "row_counts": summary["row_counts"], + "rule": RULE_VERSION, + "schema": "hyperlex.model_based_wsd_candidate_v1_development_analysis.v1", + "selected_source": "none", + "spec_sha256": spec_sha, + "tier3_attempts": len(resolutions), + "tier3_counts": projection["tier3_counts"], + "yes_no_emitted": False, + } + for banned in ("accuracy", "precision", "recall", "f1"): + if banned in analysis: + refuse("analysis reports an accuracy metric") + analysis_sha = write_json(ANALYSIS_PATH, analysis) + hashes["analysis_sha256"] = analysis_sha + hashes["decision_sha256"] = write_decision(outcome, hashes, analysis_sha) + tracker_sha = update_tracker(outcome, hashes) + if sha256(SPEC_PATH) != spec_sha or sha256(PROVENANCE_PATH) != provenance_sha: + refuse("late write mutated the frozen spec") + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": outcome["candidate_status"], + "combined_constituent_counts": summary["combined_constituent_counts"], + "decision_sha256": hashes["decision_sha256"], + "determinism": IDENTICAL, + "high_ready": outcome["high_ready"], + "license_sha256": license_sha, + "projection_sha256": projection_sha, + "provenance_sha256": provenance_sha, + "raw_sha256": raw_sha, + "reject_ready": ready_by["REJECT"]["RESIDUAL_READY"], + "resolution_sha256": resolution_sha, + "secondary_ready": outcome["secondary_ready"], + "spec_sha256": spec_sha, + "tier3_counts": projection["tier3_counts"], + "total_ready": total_ready, + "tracker_sha256": tracker_sha, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py new file mode 100644 index 00000000..ae1ff1a7 --- /dev/null +++ b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1.py @@ -0,0 +1,578 @@ +"""Development comparison for one frozen residual replay. + +The functions do not encode text, do not read operator labels, and do not +choose a threshold. Scores enter only after the replay has hashed them. +""" + +from __future__ import annotations + +import random +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.model_based_wsd_candidate_v1 import project_row_status +from hyperlexical.semantic_compositionality_residual import ( + RESIDUAL_QUANTUM, + decimal_mean, + percentile, +) + +EXACT = "EXACT" +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +TIER1_STRUCTURAL = "TIER1_STRUCTURAL" +TIER1_UNIQUE = "TIER1_UNIQUE_LEMMA" +TIER2_LESK = "TIER2_EXTENDED_LESK" +TIER3_GLOSSBERT = "TIER3_GLOSSBERT" +TIER_UNRESOLVED = "UNRESOLVED" +READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +SUPPORTED = "SUPPORTED_DIRECTION" +NO_SEPARATION = "NO_DIRECTIONAL_SEPARATION" +INVERTED = "INVERTED_DIRECTION" +PROMISING = "CANDIDATE_PROMISING" +INSUFFICIENT = "CANDIDATE_INSUFFICIENT" +NOT_COMPUTABLE = "NOT_COMPUTABLE" +NOT_DETERMINISTIC = "NOT_DETERMINISTIC" +BOOTSTRAP_SEED = 0 +BOOTSTRAP_RESAMPLES = 10000 +TIER3_MIN_CELL = 3 +EFFECT_QUANTUM = Decimal("0.000001") +INTERVAL_LOW = Decimal("2.5") +INTERVAL_HIGH = Decimal("97.5") + +_DIRECTION_RULE = ( + "The hypothesized direction is a larger HIGH residual than a SECONDARY residual. " + "Supported requires a higher HIGH median, a positive rank-biserial, and an AUC above one half." +) +_AUC_RULE = ( + "AUC is the share of HIGH-versus-SECONDARY pairs in which the HIGH residual is larger. " + "Ties count one half. HIGH is the positive class. SECONDARY is the negative class. " + "REJECT is excluded." +) +_EFFECT_RULE = ( + "The rank-biserial is favorable pairs minus unfavorable pairs, divided by the " + "number of HIGH-SECONDARY pairs. A favorable pair has the larger HIGH residual." +) +_BOOTSTRAP_RULE = ( + "Resample each class with replacement. The interval is the interpolated 2.5 and 97.5 " + "percentiles. The seed and the resample count are fixed before the scores are joined to labels." +) +_EXTREME_RULE = ( + "Drop the largest HIGH residual and the smallest SECONDARY residual. " + "The result is extreme-driven when the HIGH median is no longer larger." +) +_TIER3_RULE = ( + "Tier 3 concentration requires both the no-tier-3 group and the tier-3 group " + "to have at least three HIGH rows and three SECONDARY rows, the no-tier-3 group " + "to lack the hypothesized direction, and the tier-3 group to show it." +) +_NO_THRESHOLD = "No residual threshold is selected. No row receives a semantic yes or no." + + +def analysis_plan() -> dict: + """Return the preregistered comparison. It contains no residual scores.""" + return { + "auc": _AUC_RULE, + "bootstrap_resamples": BOOTSTRAP_RESAMPLES, + "bootstrap_rule": _BOOTSTRAP_RULE, + "bootstrap_seed": BOOTSTRAP_SEED, + "direction": _DIRECTION_RULE, + "effect": _EFFECT_RULE, + "emits_yes_no": False, + "extreme_rule": _EXTREME_RULE, + "high_positive_class": "HIGH", + "json_schema_document": None, + "reject_excluded_from_auc": True, + "secondary_negative_class": "SECONDARY", + "semantic_noncompositionality_threshold": None, + "threshold_eligible": False, + "threshold_rule": _NO_THRESHOLD, + "tier3_min_cell": TIER3_MIN_CELL, + "tier3_rule": _TIER3_RULE, + } + + +def integrate_constituent(resolver_row: dict, model_row: dict | None) -> dict: + """Copy one frozen constituent decision. The model row is consulted only for a Lesk tie.""" + method = resolver_row["resolution_method"] + status = resolver_row["resolution_status"] + if method == "STRUCTURAL_EXACT" and status == EXACT: + _reject_model(model_row) + return _copy(resolver_row, TIER1_STRUCTURAL, EXACT, resolver_row["selected_synset"], None) + if method == "UNIQUE_LEMMA" and status == EXACT: + _reject_model(model_row) + return _copy(resolver_row, TIER1_UNIQUE, EXACT, resolver_row["selected_synset"], None) + if method == "EXTENDED_LESK_V1" and status == RESOLVED: + _reject_model(model_row) + return _copy(resolver_row, TIER2_LESK, RESOLVED, resolver_row["selected_synset"], None) + if method == "NONE" and status == UNRESOLVED: + _reject_model(model_row) + return _copy(resolver_row, TIER_UNRESOLVED, UNRESOLVED, None, None) + if method == "STRUCTURAL_EXACT" and status == AMBIGUOUS: + _reject_model(model_row) + return _copy(resolver_row, TIER1_STRUCTURAL, AMBIGUOUS, None, None) + if method != "EXTENDED_LESK_V1" or status != AMBIGUOUS: + raise RuntimeError("constituent decision is outside the frozen stack") + if model_row is None: + raise RuntimeError("lesk tie has no frozen model row") + if model_row["prior_resolution_status"] != AMBIGUOUS: + raise RuntimeError("model row is not a lesk tie") + if model_row["model_resolution_status"] == RESOLVED: + selected = model_row["selected_synset"] + if selected not in resolver_row["candidate_synsets"]: + raise RuntimeError("model synset is outside the frozen candidates") + return _copy( + resolver_row, + TIER3_GLOSSBERT, + RESOLVED, + selected, + model_row["selected_sense_key"], + model_row["primary_evidence_code"], + ) + if model_row["model_resolution_status"] == AMBIGUOUS: + if model_row["selected_synset"] is not None: + raise RuntimeError("abstaining model row names a synset") + return _copy(resolver_row, TIER3_GLOSSBERT, AMBIGUOUS, None, None, model_row["primary_evidence_code"]) + raise RuntimeError("model row is not a frozen resolved or abstaining decision") + + +def _reject_model(model_row: dict | None) -> None: + if model_row is not None: + raise RuntimeError("model row overrides a frozen non-tie") + + +def _copy(resolver_row, tier, status, synset, sense_key, provenance=None) -> dict: + return { + "constituent_index": resolver_row["constituent_index"], + "constituent_pos": resolver_row["constituent_pos"], + "constituent_surface": resolver_row["constituent_surface"], + "parent_row_id": resolver_row["parent_row_id"], + "parent_surface": resolver_row["parent_surface"], + "parent_synset": resolver_row["parent_synset"], + "resolution_provenance": provenance or resolver_row["primary_evidence_code"], + "resolution_status": status, + "resolution_tier": tier, + "selected_pwn30_synset": synset, + "selected_sense_key_if_available": sense_key, + } + + +def projection_token(tier: str, status: str) -> str: + """Map an integrated constituent onto the frozen projection vocabulary.""" + if status == EXACT: + return EXACT + if tier == TIER2_LESK and status == RESOLVED: + return "LESK_RESOLVED" + if tier == TIER3_GLOSSBERT and status == RESOLVED: + return "MODEL_RESOLVED" + if status == UNRESOLVED: + return UNRESOLVED + if status == AMBIGUOUS: + return AMBIGUOUS + raise RuntimeError("integrated constituent has no projection token") + + +def row_projection(statuses: list[str]) -> str: + return project_row_status(statuses) + + +def abstention_reason(statuses: list[str]) -> str | None: + """Return the leftmost frozen abstention. A ready row returns None.""" + if len(statuses) < 2: + return "fewer_than_two_content_constituents" + for status in statuses: + if status == AMBIGUOUS: + return "ambiguous_content_constituent" + if status == UNRESOLVED: + return "unresolved_content_constituent" + if status not in {EXACT, RESOLVED}: + raise RuntimeError("unknown integrated status") + return None + + +def _ordered(scores: list[str]) -> list[str]: + return sorted(scores, key=Decimal) + + +def sample_std(scores: list[str]) -> str | None: + """Sample standard deviation at the residual quantum. One row has no dispersion.""" + if len(scores) < 2: + return None + mean = sum((Decimal(score) for score in scores), start=Decimal(0)) / Decimal(len(scores)) + dispersion = sum((Decimal(score) - mean) ** 2 for score in scores) / Decimal(len(scores) - 1) + return str(dispersion.sqrt().quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def full_distribution(scores: list[str]) -> dict: + if not scores: + return { + "count": 0, + "max": None, + "mean": None, + "median": None, + "min": None, + "p10": None, + "p25": None, + "p75": None, + "p90": None, + "status": NOT_COMPUTABLE, + "std": None, + } + ordered = _ordered(scores) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "p10": percentile(ordered, 10), + "p25": percentile(ordered, 25), + "p75": percentile(ordered, 75), + "p90": percentile(ordered, 90), + "status": "DESCRIPTIVE", + "std": sample_std(ordered), + } + + +def _quantize_effect(value: Decimal) -> str: + return str(value.quantize(EFFECT_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def _quantize_residual(value: Decimal) -> str: + return str(value.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def pair_comparison(high_scores: list[str], secondary_scores: list[str]) -> dict: + """Compare HIGH with SECONDARY. REJECT is not an argument.""" + if not high_scores or not secondary_scores: + return {"status": NOT_COMPUTABLE} + favorable = 0 + unfavorable = 0 + ties = 0 + for high in high_scores: + high_value = Decimal(high) + for secondary in secondary_scores: + secondary_value = Decimal(secondary) + if high_value > secondary_value: + favorable += 1 + elif high_value < secondary_value: + unfavorable += 1 + else: + ties += 1 + pairs = Decimal(len(high_scores) * len(secondary_scores)) + u_high = Decimal(favorable) + (Decimal(ties) / Decimal(2)) + u_secondary = Decimal(unfavorable) + (Decimal(ties) / Decimal(2)) + high_dist = full_distribution(high_scores) + secondary_dist = full_distribution(secondary_scores) + mean_gap = Decimal(high_dist["mean"]) - Decimal(secondary_dist["mean"]) + median_gap = Decimal(high_dist["median"]) - Decimal(secondary_dist["median"]) + auc = u_high / pairs + effect = (Decimal(favorable) - Decimal(unfavorable)) / pairs + return { + "auc": _quantize_effect(auc), + "favorable_pairs": favorable, + "high": high_dist, + "mean_difference": _quantize_residual(mean_gap), + "median_difference": _quantize_residual(median_gap), + "rank_biserial": _quantize_effect(effect), + "secondary": secondary_dist, + "status": "DESCRIPTIVE", + "ties": ties, + "u_high": _quantize_effect(u_high), + "u_secondary": _quantize_effect(u_secondary), + "unfavorable_pairs": unfavorable, + } + + +def direction_result(comparison: dict) -> str: + if comparison.get("status") == NOT_COMPUTABLE: + return NOT_COMPUTABLE + median_gap = Decimal(comparison["median_difference"]) + effect = Decimal(comparison["rank_biserial"]) + auc = Decimal(comparison["auc"]) + if median_gap > 0 and effect > 0 and auc > Decimal("0.5"): + return SUPPORTED + if median_gap < 0 and effect < 0 and auc < Decimal("0.5"): + return INVERTED + return NO_SEPARATION + + +def _supports(scores_high: list[str], scores_secondary: list[str]) -> bool: + if not scores_high or not scores_secondary: + return False + return direction_result(pair_comparison(scores_high, scores_secondary)) == SUPPORTED + + +def _interpolated(sorted_values: list[Decimal], percent: Decimal) -> Decimal: + count = len(sorted_values) + if count == 1: + return sorted_values[0] + rank = Decimal(count - 1) * (percent / Decimal(100)) + low = int(rank) + high = min(low + 1, count - 1) + weight = rank - Decimal(low) + return sorted_values[low] + (sorted_values[high] - sorted_values[low]) * weight + + +def bootstrap_intervals( + high_scores: list[str], + secondary_scores: list[str], + *, + seed: int = BOOTSTRAP_SEED, + resamples: int = BOOTSTRAP_RESAMPLES, +) -> dict: + """Deterministic percentile intervals. The seed and count are arguments, not a search.""" + if seed != BOOTSTRAP_SEED or resamples != BOOTSTRAP_RESAMPLES: + if resamples < 1 or seed < 0: + raise RuntimeError("bootstrap settings are invalid") + if not high_scores or not secondary_scores: + return {"status": NOT_COMPUTABLE} + generator = random.Random(seed) + high_values = [Decimal(score) for score in high_scores] + secondary_values = [Decimal(score) for score in secondary_scores] + mean_gaps = [] + median_gaps = [] + aucs = [] + for _ in range(resamples): + high_draw = [high_values[generator.randrange(len(high_values))] for _ in high_values] + secondary_draw = [secondary_values[generator.randrange(len(secondary_values))] for _ in secondary_values] + high_text = [format(value, "f") for value in high_draw] + secondary_text = [format(value, "f") for value in secondary_draw] + comparison = pair_comparison(high_text, secondary_text) + mean_gaps.append(Decimal(comparison["mean_difference"])) + median_gaps.append(Decimal(comparison["median_difference"])) + aucs.append(Decimal(comparison["auc"])) + return { + "auc": _interval(aucs, EFFECT_QUANTUM), + "mean_difference": _interval(mean_gaps, RESIDUAL_QUANTUM), + "median_difference": _interval(median_gaps, RESIDUAL_QUANTUM), + "resamples": resamples, + "seed": seed, + "status": "DESCRIPTIVE", + } + + +def _interval(values: list[Decimal], quantum: Decimal) -> dict: + ordered = sorted(values) + low = _interpolated(ordered, INTERVAL_LOW).quantize(quantum, rounding=ROUND_HALF_EVEN) + high = _interpolated(ordered, INTERVAL_HIGH).quantize(quantum, rounding=ROUND_HALF_EVEN) + return {"high": str(high), "low": str(low)} + + +def _cell(rows: list[dict], bucket: str) -> list[str]: + return [row["residual_score"] for row in rows if row["bucket"] == bucket] + + +def tier3_concentration(rows: list[dict]) -> dict: + """Return whether the direction exists only inside the Tier 3 subgroup.""" + groups = {} + concentrated = False + flags = {} + for name, uses_tier3 in (("no_tier3", False), ("with_tier3", True)): + chosen = [row for row in rows if row["uses_tier3"] is uses_tier3 and row["bucket"] in {"HIGH", "SECONDARY"}] + high = _cell(chosen, "HIGH") + secondary = _cell(chosen, "SECONDARY") + computable = len(high) >= TIER3_MIN_CELL and len(secondary) >= TIER3_MIN_CELL + if not computable: + groups[name] = { + "high_count": len(high), + "secondary_count": len(secondary), + "status": NOT_COMPUTABLE, + } + flags[name] = None + continue + comparison = pair_comparison(high, secondary) + groups[name] = { + "comparison": comparison, + "direction": direction_result(comparison), + "high_count": len(high), + "secondary_count": len(secondary), + "status": "DESCRIPTIVE", + } + flags[name] = groups[name]["direction"] == SUPPORTED + if flags["no_tier3"] is False and flags["with_tier3"] is True: + concentrated = True + return {"concentrated": concentrated, "groups": groups} + + +def extreme_driven(high_scores: list[str], secondary_scores: list[str]) -> bool: + """True when dropping the most favorable extreme on each side removes the median gap.""" + if len(high_scores) < TIER3_MIN_CELL or len(secondary_scores) < TIER3_MIN_CELL: + return False + if not _supports(high_scores, secondary_scores): + return False + kept_high = list(high_scores) + kept_secondary = list(secondary_scores) + kept_high.remove(max(kept_high, key=Decimal)) + kept_secondary.remove(min(kept_secondary, key=Decimal)) + return not _supports(kept_high, kept_secondary) + + +def pos_partitioned(rows: list[dict]) -> bool: + high = {row["pos"] for row in rows if row["bucket"] == "HIGH"} + secondary = {row["pos"] for row in rows if row["bucket"] == "SECONDARY"} + return bool(high) and bool(secondary) and high.isdisjoint(secondary) + + +def replay_decision( + *, + readiness_reproduced: bool, + determinism: str, + direction: str, + tier3_concentrated: bool, + extremes: bool, + pos_split: bool, +) -> dict: + """Apply the preregistered status rule. The flags are not a threshold search.""" + if not readiness_reproduced: + status = NOT_COMPUTABLE + transition = "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION" + elif determinism != "IDENTICAL": + status = NOT_DETERMINISTIC + transition = "RESIDUAL_REPLAY_DETERMINISM_REVIEW_AUTHORIZATION" + elif direction == NOT_COMPUTABLE: + status = NOT_COMPUTABLE + transition = "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION" + elif direction == SUPPORTED and not tier3_concentrated and not extremes and not pos_split: + status = PROMISING + transition = "RESIDUAL_THRESHOLD_PREREGISTRATION_AUTHORIZATION" + else: + status = INSUFFICIENT + transition = "RESIDUAL_V2_DESIGN_AUTHORIZATION" + return { + "candidate_status": status, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "RESIDUAL_DEVELOPMENT_ANALYZED_V2", + "threshold_eligible": False, + } + + +def _ranks(values: list[Decimal]) -> list[Decimal]: + order = sorted(range(len(values)), key=lambda index: values[index]) + ranks = [Decimal(0)] * len(values) + start = 0 + while start < len(values): + stop = start + 1 + while stop < len(values) and values[order[stop]] == values[order[start]]: + stop += 1 + rank = Decimal(start + 1 + stop) / Decimal(2) + for index in order[start:stop]: + ranks[index] = rank + start = stop + return ranks + + +def spearman(xs: list[str], ys: list[str]) -> str | None: + if len(xs) != len(ys) or len(xs) < 3: + return None + x_values = [Decimal(value) for value in xs] + y_values = [Decimal(value) for value in ys] + x_ranks = _ranks(x_values) + y_ranks = _ranks(y_values) + x_mean = sum(x_ranks, start=Decimal(0)) / Decimal(len(xs)) + y_mean = sum(y_ranks, start=Decimal(0)) / Decimal(len(ys)) + covariance = sum((x - x_mean) * (y - y_mean) for x, y in zip(x_ranks, y_ranks)) + x_scale = sum((x - x_mean) ** 2 for x in x_ranks) + y_scale = sum((y - y_mean) ** 2 for y in y_ranks) + if x_scale == 0 or y_scale == 0: + return None + correlation = covariance / (x_scale.sqrt() * y_scale.sqrt()) + return _quantize_effect(correlation) + + +def numeric_summary(values: list[str]) -> dict: + if not values: + return {"count": 0, "status": NOT_COMPUTABLE} + ordered = _ordered(values) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "status": "DESCRIPTIVE", + } + + +def confound_report(rows: list[dict]) -> dict: + """Describe nuisance structure. It does not change a residual.""" + target = [row for row in rows if row["bucket"] in {"HIGH", "SECONDARY"}] + tier3 = tier3_concentration(target) + high_scores = _cell(target, "HIGH") + secondary_scores = _cell(target, "SECONDARY") + nuisance = {} + for field in ( + "token_count", + "character_length", + "content_count", + "max_candidate_senses", + "min_glossbert_confidence", + "min_glossbert_margin", + ): + nuisance[field] = _nuisance(target, field) + poses = {} + for pos in sorted({row["pos"] for row in target}): + chosen = [row for row in target if row["pos"] == pos] + high = _cell(chosen, "HIGH") + secondary = _cell(chosen, "SECONDARY") + if len(high) < TIER3_MIN_CELL or len(secondary) < TIER3_MIN_CELL: + poses[pos] = {"high_count": len(high), "secondary_count": len(secondary), "status": NOT_COMPUTABLE} + else: + poses[pos] = {"comparison": pair_comparison(high, secondary), "status": "DESCRIPTIVE"} + return { + "extreme_driven": extreme_driven(high_scores, secondary_scores), + "nuisance": nuisance, + "pos": poses, + "pos_partitioned": pos_partitioned(target), + "tier3": tier3, + "tier3_counts": { + "high_with_tier3": _tier_count(rows, "HIGH", True), + "high_without_tier3": _tier_count(rows, "HIGH", False), + "secondary_with_tier3": _tier_count(rows, "SECONDARY", True), + "secondary_without_tier3": _tier_count(rows, "SECONDARY", False), + }, + } + + +def _tier_count(rows: list[dict], bucket: str, uses_tier3: bool) -> int: + return sum(row["bucket"] == bucket and row["uses_tier3"] is uses_tier3 for row in rows) + + +def _nuisance(rows: list[dict], field: str) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY"): + paired = [ + (str(row[field]), row["residual_score"]) + for row in rows + if row["bucket"] == bucket and row[field] is not None + ] + if not paired: + report[bucket] = {"status": NOT_COMPUTABLE} + continue + values = [item[0] for item in paired] + report[bucket] = { + "spearman_with_residual": spearman(values, [item[1] for item in paired]), + "summary": numeric_summary(values), + } + return report + + +def outlier_pair(rows: list[dict], bucket: str) -> dict: + chosen = [row for row in rows if row["bucket"] == bucket] + if not chosen: + return {"status": NOT_COMPUTABLE} + low = min(chosen, key=lambda row: Decimal(row["residual_score"])) + high = max(chosen, key=lambda row: Decimal(row["residual_score"])) + return {"largest": _outlier(high), "smallest": _outlier(low), "status": "DESCRIPTIVE"} + + +def _outlier(row: dict) -> dict: + return { + "residual_score": row["residual_score"], + "resolution_tiers": list(row["tiers"]), + "row_id": row["row_id"], + "surface": row["surface"], + "synset": row["synset"], + } diff --git a/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py new file mode 100644 index 00000000..83f4b467 --- /dev/null +++ b/scripts/shadow/hyperlexical/residual_model_resolved_replay_v1_replay.py @@ -0,0 +1,766 @@ +"""Replay the frozen residual on the frozen constituent-sense stack. + +Operator labels are read only after the score artifact and its receipt have +been hashed. The residual model, the three resolution tiers, and the residual +formula are not changed. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import sys +from collections import Counter, defaultdict +from decimal import Decimal +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["NUMEXPR_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["OPENBLAS_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.constituent_sense_resolution_v1_replay import ( + EVENTS, + EVIDENCE_MANIFEST, + HYPERLEX, + LEDGER_FILE, + OPERATOR_COUNTS, + OPERATORS, + REPLAY_PATH as RESOLVER_REPLAY_PATH, + SOURCE, + TRACKER, + WORDNET, + load_sealed_rows, + refuse, + sha256, + write_json, + write_jsonl, +) +from hyperlexical.model_based_wsd_candidate_v1_replay import ( + ANALYSIS_PATH as WSD_ANALYSIS_PATH, + DECISION_PATH as WSD_DECISION_PATH, + EXPECTED as WSD_EXPECTED, + LICENSE_PATH as WSD_LICENSE_PATH, + PROJECTION_PATH, + PROVENANCE_PATH as WSD_PROVENANCE_PATH, + RAW_PATH as WSD_RAW_PATH, + RESOLUTION_PATH as WSD_RESOLUTION_PATH, + SPEC_PATH as WSD_SPEC_PATH, + load_jsonl, + load_sense_index, + sense_keys_for, +) +from hyperlexical.residual_model_resolved_replay_v1 import ( + AMBIGUOUS, + BOOTSTRAP_RESAMPLES, + BOOTSTRAP_SEED, + EXACT, + RESOLVED, + TIER1_STRUCTURAL, + TIER3_GLOSSBERT, + UNRESOLVED, + abstention_reason, + analysis_plan, + bootstrap_intervals, + confound_report, + direction_result, + full_distribution, + integrate_constituent, + outlier_pair, + pair_comparison, + projection_token, + replay_decision, + row_projection, +) +from hyperlexical.semantic_compositionality_residual import ( + COMPOSITION_OPERATOR, + DISTANCE_METRIC, + MODEL_NAME, + MODEL_REVISION, + representation_text, + residual_score, + select_lemma, + vector_hash, +) +from hyperlexical.semantic_compositionality_residual_replay import ( + MAX_SEQUENCE_LENGTH, + OUTPUT_DIMENSION, + build_indexes, + encode_texts, + load_encoder, + pointer_records, + token_length, +) +from hyperlexical.unbind_sense_screen_v1 import load_exceptions + +RESIDUAL_SPEC = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +RESIDUAL_SCORES = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +INTEGRATED_PATH = SOURCE / "INTEGRATED_CONSTITUENT_RESOLUTION_V1.jsonl" +SCORES_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_REPLAY_SCORES.jsonl" +RECEIPT_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_REPLAY_RECEIPT.json" +ANALYSIS_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_DEVELOPMENT_ANALYSIS.json" +CONFOUND_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_CONFOUND_ANALYSIS.json" +DECISION_PATH = SOURCE / "RESIDUAL_V1_MODEL_RESOLVED_CANDIDATE_DECISION.json" + +CURRENT_TRACKER = "6da1e9730c785d2784d23433455515989d312ed8c76f4763b3a2dd6a1f2c4b2d" +RESIDUAL_SPEC_SHA = "39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d" +RESIDUAL_SCORES_SHA = "cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7" +AUTHORIZATION = "RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION" +EXPECTED_READY = 73 +EXPECTED_UNKNOWN = 152 + +EXPECTED = dict(WSD_EXPECTED) +EXPECTED[TRACKER] = CURRENT_TRACKER +EXPECTED[WSD_SPEC_PATH] = "c861ff7fff11ae6a790531267229c18d6e6e0a171a9bf6c34cfb6f7e7b14498c" +EXPECTED[WSD_PROVENANCE_PATH] = "6d91283244a88f44477c836e6549118d5d96a36bb878bf448fcadb13ff765e11" +EXPECTED[WSD_LICENSE_PATH] = "67d45243959e3e76643a675d49b508a73d0f7f0bf8fbf060ce7b83c9200c621b" +EXPECTED[WSD_RAW_PATH] = "a0e508c225e6db4cdbcae701682202b1d854f3762546dbfe10b84a15e0e9e17c" +EXPECTED[WSD_RESOLUTION_PATH] = "ed945989cf4947ac84633ba2c4aa10c1ba381d2396da0b573a844f83ec367a18" +EXPECTED[PROJECTION_PATH] = "c75834faf4a84d36e83246244e0aa7c6c7788c3a57cfdb7f77c7628a52023328" +EXPECTED[WSD_ANALYSIS_PATH] = "8ca0d8c8dd7e510a04daef6b3fbd78ace6d20b173e88d2a3d0c2e10fdaafe30b" +EXPECTED[WSD_DECISION_PATH] = "c8ed0b8acaaa415277c5f9bdbf982d075136b795b5d95fd8813d5a3010072dbb" +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/model_based_wsd_candidate_v1.py"] = ( + "7f489772162dd0be249df3fe8c0a5476e69e5a4ca73fba7468f61c60cda84bad" +) +EXPECTED[HYPERLEX / "scripts/shadow/hyperlexical/model_based_wsd_candidate_v1_replay.py"] = ( + "5981978209da5884fe6cd6efbaf7593a37e1a40138a75aa70906694cee8b8a09" +) + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if not path.is_file() or sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def _digest(payload) -> str: + text = json.dumps(payload, sort_keys=True, ensure_ascii=True) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def attach_sense_key(item: dict, exceptions: dict, sense_index: dict) -> None: + """Fill a unique PWN 3.0 sense key. The selected synset stays as frozen.""" + synset = item["selected_pwn30_synset"] + if synset is None: + return + keys = sense_keys_for(synset, item["constituent_surface"], exceptions, sense_index) + current = item["selected_sense_key_if_available"] + if current is not None: + if current not in keys: + refuse(f"frozen sense key is not in the PWN 3.0 index for {synset}") + return + if len(keys) == 1: + item["selected_sense_key_if_available"] = keys[0] + + +def build_integrated(resolver_rows: list[dict], model_rows: list[dict], exceptions: dict, sense_index: dict): + model = {(row["parent_row_id"], row["constituent_index"]): row for row in model_rows} + if len(model) != len(model_rows): + refuse("duplicate model constituent keys") + consumed = set() + integrated = [] + by_parent = defaultdict(list) + for row in resolver_rows: + if row["constituent_index"] is None: + continue + key = (row["parent_row_id"], row["constituent_index"]) + model_row = model.get(key) + if model_row is not None: + consumed.add(key) + item = integrate_constituent(row, model_row) + attach_sense_key(item, exceptions, sense_index) + item["candidate_count"] = len(row["candidate_synsets"]) + integrated.append(item) + by_parent[row["parent_row_id"]].append(item) + if consumed != set(model): + refuse("a model row does not match a frozen lesk tie") + for items in by_parent.values(): + items.sort(key=lambda item: item["constituent_index"]) + return integrated, by_parent + + +def project_manifest(sealed: list[dict], by_parent: dict) -> list[dict]: + projected = [] + for row in sealed: + items = by_parent.get(row["row_id"], []) + tokens = [projection_token(item["resolution_tier"], item["resolution_status"]) for item in items] + projected.append( + { + "constituent_statuses": tokens, + "content_count": len(tokens), + "parent_row_id": row["row_id"], + "projected_status": row_projection(tokens), + } + ) + return projected + + +def projection_failure(projected: list[dict]) -> str | None: + frozen = json.loads(PROJECTION_PATH.read_text(encoding="utf-8")) + if frozen.get("operator_labels_included") is not False: + return "READINESS_REPRODUCTION_FAILURE projection includes operator labels" + frozen_rows = {row["parent_row_id"]: row for row in frozen["rows"]} + if len(frozen_rows) != 225 or len(projected) != 225: + return "READINESS_REPRODUCTION_FAILURE row count" + for row in projected: + prior = frozen_rows.get(row["parent_row_id"]) + if prior is None: + return f"READINESS_REPRODUCTION_FAILURE missing {row['parent_row_id']}" + if prior["constituent_statuses"] != row["constituent_statuses"] or prior["content_count"] != row["content_count"] or prior["projected_status"] != row["projected_status"]: + return ( + f"READINESS_REPRODUCTION_FAILURE {row['parent_row_id']} " + f"got {row['constituent_statuses']} {row['projected_status']} " + f"expected {prior['constituent_statuses']} {prior['projected_status']}" + ) + counts = Counter(row["projected_status"] for row in projected) + if counts["RESIDUAL_READY"] != EXPECTED_READY or counts["UNKNOWN"] != EXPECTED_UNKNOWN: + return "READINESS_REPRODUCTION_FAILURE totals" + return None + + +def public_integrated(item: dict) -> dict: + return { + "constituent_index": item["constituent_index"], + "constituent_pos": item["constituent_pos"], + "constituent_surface": item["constituent_surface"], + "parent_row_id": item["parent_row_id"], + "parent_surface": item["parent_surface"], + "parent_synset": item["parent_synset"], + "resolution_provenance": item["resolution_provenance"], + "resolution_status": item["resolution_status"], + "resolution_tier": item["resolution_tier"], + "selected_pwn30_synset": item["selected_pwn30_synset"], + "selected_sense_key_if_available": item["selected_sense_key_if_available"], + } + + +def constituent_representation(item: dict, by_id: dict, pointers: list, exceptions: dict) -> str: + synset = item["selected_pwn30_synset"] + record = by_id.get(synset) + if record is None: + refuse(f"selected synset is not in PWN 3.0: {synset}") + if item["resolution_tier"] == TIER1_STRUCTURAL: + pool = [ + lemma + for symbol, target_word, target_id, lemma in pointers + if target_id == synset and symbol in {"+", "\\"} and target_word > 0 and lemma + ] + else: + pool = list(record["lemmas"]) + lemma = select_lemma(pool, item["constituent_surface"], exceptions) + return representation_text(lemma, record["pos"], record["gloss"]) + + +def prepare_jobs(sealed: list[dict], by_parent: dict, by_id: dict, exceptions: dict, model_index: dict): + jobs = [] + unknown = [] + for row in sealed: + items = by_parent.get(row["row_id"], []) + tokens = [projection_token(item["resolution_tier"], item["resolution_status"]) for item in items] + readiness = row_projection(tokens) + if readiness != "RESIDUAL_READY": + statuses = [] + for item in items: + if item["resolution_status"] in {EXACT, RESOLVED}: + statuses.append(RESOLVED if item["resolution_status"] == RESOLVED else EXACT) + else: + statuses.append(item["resolution_status"]) + unknown.append( + { + "abstention_reason": abstention_reason(statuses), + "readiness": readiness, + "row_id": row["row_id"], + } + ) + continue + parent = f"{row['synset_pos']}:{row['synset_offset']}" + pointers = pointer_records(parent, by_id) + parts = [constituent_representation(item, by_id, pointers, exceptions) for item in items] + whole = representation_text(row["surface"], row["pos"], row["gloss"]) + confidences = [] + margins = [] + for item in items: + if item["resolution_tier"] != TIER3_GLOSSBERT: + continue + model_row = model_index[(row["row_id"], item["constituent_index"])] + confidences.append(Decimal(model_row["model_confidence"])) + margins.append(Decimal(model_row["model_margin"])) + jobs.append( + { + "constituent_representation_texts": parts, + "content_constituents": [item["constituent_surface"] for item in items], + "max_candidate_senses": max(item["candidate_count"] for item in items), + "min_glossbert_confidence": format(min(confidences), "f") if confidences else None, + "min_glossbert_margin": format(min(margins), "f") if margins else None, + "pos": row["pos"], + "readiness": readiness, + "resolved_constituent_synsets": [item["selected_pwn30_synset"] for item in items], + "row_id": row["row_id"], + "surface": row["surface"], + "synset": parent, + "tiers": [item["resolution_tier"] for item in items], + "whole_representation_text": whole, + } + ) + return jobs, unknown + + +def score_jobs(jobs: list[dict], vectors: dict[str, list[float]], integrated_sha: str) -> list[dict]: + scored = [] + for job in jobs: + whole = vectors[job["whole_representation_text"]] + parts = [vectors[text] for text in job["constituent_representation_texts"]] + result = residual_score(whole, parts) + if result is None: + refuse(f"residual formula returned no score for {job['row_id']}") + residual, composed = result + if len(whole) != OUTPUT_DIMENSION: + refuse("encoder width drifted") + scored.append( + { + "composed_vector_hash": vector_hash(composed), + "constituent_representation_texts": list(job["constituent_representation_texts"]), + "constituent_resolution_tiers": list(job["tiers"]), + "constituent_vector_hashes": [vector_hash(part) for part in parts], + "content_constituents": list(job["content_constituents"]), + "integrated_resolution_sha256": integrated_sha, + "pos": job["pos"], + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_score": residual, + "resolved_constituent_synsets": list(job["resolved_constituent_synsets"]), + "row_id": job["row_id"], + "score_status": "SCORED", + "surface": job["surface"], + "synset": job["synset"], + "whole_representation_text": job["whole_representation_text"], + "whole_vector_hash": vector_hash(whole), + } + ) + return scored + + +def identity_rows(jobs: list[dict], scored: list[dict], unknown: list[dict]) -> list[dict]: + scored_by = {row["row_id"]: row for row in scored} + rows = [] + for job in jobs: + row = scored_by[job["row_id"]] + rows.append( + { + "composed_vector_hash": row["composed_vector_hash"], + "constituent_representation_texts": row["constituent_representation_texts"], + "constituent_vector_hashes": row["constituent_vector_hashes"], + "readiness": job["readiness"], + "residual_score": row["residual_score"], + "row_id": job["row_id"], + "whole_representation_text": row["whole_representation_text"], + "whole_vector_hash": row["whole_vector_hash"], + } + ) + for row in unknown: + rows.append( + { + "composed_vector_hash": None, + "constituent_representation_texts": None, + "constituent_vector_hashes": None, + "readiness": row["readiness"], + "residual_score": None, + "row_id": row["row_id"], + "whole_representation_text": None, + "whole_vector_hash": None, + } + ) + return sorted(rows, key=lambda row: row["row_id"]) + + +def encode_needed(model, jobs: list[dict]) -> dict[str, list[float]]: + needed = sorted({text for job in jobs for text in [job["whole_representation_text"], *job["constituent_representation_texts"]]}) + return encode_texts(model, needed) + + +def apply_overflow(model, jobs: list[dict]) -> tuple[list[dict], list[dict]]: + kept = [] + overflow = [] + for job in jobs: + texts = [job["whole_representation_text"], *job["constituent_representation_texts"]] + if any(token_length(model, text) > MAX_SEQUENCE_LENGTH for text in texts): + overflow.append({"abstention_reason": "representation_exceeds_max_sequence_length", "row_id": job["row_id"]}) + continue + kept.append(job) + return kept, overflow + + +def assemble_scores(sealed: list[dict], scored: list[dict], unknown: list[dict], overflow: list[dict]) -> list[dict]: + scored_by = {row["row_id"]: row for row in scored} + unknown_by = {row["row_id"]: row["abstention_reason"] for row in unknown} + overflow_by = {row["row_id"]: row["abstention_reason"] for row in overflow} + rows = [] + for row in sealed: + if row["row_id"] in scored_by: + rows.append(scored_by[row["row_id"]]) + continue + reason = unknown_by.get(row["row_id"], overflow_by.get(row["row_id"])) + if reason is None: + refuse(f"row has no score and no abstention: {row['row_id']}") + rows.append({"abstention_reason": reason, "row_id": row["row_id"], "score_status": "UNKNOWN"}) + return rows + + +def nuisance_for(job: dict) -> dict: + return { + "character_length": str(len(job["surface"])), + "content_count": str(len(job["content_constituents"])), + "max_candidate_senses": str(job["max_candidate_senses"]), + "min_glossbert_confidence": job["min_glossbert_confidence"], + "min_glossbert_margin": job["min_glossbert_margin"], + "pos": job["pos"], + "surface": job["surface"], + "synset": job["synset"], + "tiers": list(job["tiers"]), + "token_count": str(len(job["surface"].split())), + "uses_tier3": TIER3_GLOSSBERT in job["tiers"], + } + + +def load_operator_buckets() -> dict[str, str]: + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {} + for row in manifest["rows"]: + buckets[row["row_id"]] = row["operator_bucket"] + if row.get("sense_class") is not None: + refuse("manifest sense class changed") + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + return buckets + + +def tier_counts(rows: list[dict]) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY"): + chosen = [row for row in rows if row["bucket"] == bucket] + report[f"{bucket.lower()}_with_tier3"] = sum(row["uses_tier3"] for row in chosen) + report[f"{bucket.lower()}_without_tier3"] = sum(not row["uses_tier3"] for row in chosen) + return report + + +def composition_counts(rows: list[dict]) -> dict: + report = {} + for bucket in ("HIGH", "SECONDARY", "REJECT"): + counter = Counter(tuple(row["tiers"]) for row in rows if row["bucket"] == bucket) + report[bucket] = {",".join(key): value for key, value in sorted(counter.items())} + return report + + +def write_failure(status: str, transition: str, detail: str, hashes: dict | None = None) -> None: + payload = { + "candidate_status": status, + "detail": detail, + "emits_yes_no": False, + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": transition, + "next_transition_authorized": False, + "residual_formula_count": 1, + "runtime_integration": False, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "state": "RESIDUAL_DEVELOPMENT_ANALYZED_V2", + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, payload) + recorded = dict(hashes or {}) + recorded["decision_sha256"] = decision_sha + update_tracker(payload, recorded) + refuse(detail) + + +def update_tracker(decision: dict, hashes: dict) -> str: + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if sha256(TRACKER) != CURRENT_TRACKER: + refuse("tracker hash drifted before update") + if tracker.get("model_based_wsd_candidate_status") != "CANDIDATE_PROMISING": + refuse("model WSD status drifted") + if tracker.get("constituent_sense_resolution_candidate_status") != "COVERAGE_INSUFFICIENT": + refuse("constituent resolver status drifted") + if tracker.get("residual_evaluation_status") != "CANDIDATE_DISTRIBUTION_FROZEN": + refuse("residual freeze status drifted") + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = CURRENT_TRACKER + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["residual_evaluation_status"] = "CANDIDATE_DISTRIBUTION_FROZEN" + tracker["residual_model_resolved_state"] = decision["state"] + tracker["residual_model_resolved_candidate_status"] = decision["candidate_status"] + tracker["residual_model_resolved_threshold_eligible"] = False + tracker["residual_model_resolved_integrated_sha256"] = hashes.get("integrated_sha256") + tracker["residual_model_resolved_scores_sha256"] = hashes.get("scores_sha256") + tracker["residual_model_resolved_receipt_sha256"] = hashes.get("receipt_sha256") + tracker["residual_model_resolved_analysis_sha256"] = hashes.get("analysis_sha256") + tracker["residual_model_resolved_confound_sha256"] = hashes.get("confound_sha256") + tracker["residual_model_resolved_decision_sha256"] = hashes.get("decision_sha256") + tracker["residual_threshold_eligible"] = False + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = decision["next_legal_transition"] + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + if sha256(RESIDUAL_SPEC) != RESIDUAL_SPEC_SHA or sha256(RESIDUAL_SCORES) != RESIDUAL_SCORES_SHA: + refuse("frozen residual artifact changed") + return tracker_sha + + +def main() -> None: + check_sealed() + if sha256(RESIDUAL_SPEC) != RESIDUAL_SPEC_SHA or sha256(RESIDUAL_SCORES) != RESIDUAL_SCORES_SHA: + refuse("frozen residual artifact changed") + if MODEL_NAME != "sentence-transformers/all-MiniLM-L6-v2": + refuse("residual model name drifted") + if MODEL_REVISION != "1110a243fdf4706b3f48f1d95db1a4f5529b4d41": + refuse("residual model revision drifted") + if COMPOSITION_OPERATOR != "normalized_mean_v1" or DISTANCE_METRIC != "one_minus_cosine_v1": + refuse("residual formula drifted") + sealed = load_sealed_rows() + resolver_rows = load_jsonl(RESOLVER_REPLAY_PATH) + model_rows = load_jsonl(WSD_RESOLUTION_PATH) + if any("operator_bucket" in row for row in resolver_rows) or any("operator_bucket" in row for row in model_rows): + refuse("upstream resolution artifact contains an operator bucket") + exceptions = load_exceptions(WORDNET) + sense_index = load_sense_index(WORDNET / "index.sense") + integrated, by_parent = build_integrated(resolver_rows, model_rows, exceptions, sense_index) + if len(integrated) != 504: + refuse(f"integrated constituent count is {len(integrated)}") + projected = project_manifest(sealed, by_parent) + failure = projection_failure(projected) + if failure: + refuse(failure) + public_rows = [public_integrated(item) for item in integrated] + integrated_sha = write_jsonl(INTEGRATED_PATH, public_rows) + if sha256(INTEGRATED_PATH) != integrated_sha: + refuse("integrated hash drifted at freeze") + print(f"INTEGRATED_FROZEN {integrated_sha}", file=sys.stderr, flush=True) + by_id, _index = build_indexes() + model_index = {(row["parent_row_id"], row["constituent_index"]): row for row in model_rows} + jobs, unknown = prepare_jobs(sealed, by_parent, by_id, exceptions, model_index) + if len(jobs) + len(unknown) != 225: + refuse("prepared row count drifted") + if len(jobs) != EXPECTED_READY or len(unknown) != EXPECTED_UNKNOWN: + refuse("READINESS_REPRODUCTION_FAILURE") + encoder = load_encoder() + kept, overflow = apply_overflow(encoder, jobs) + first_vectors = encode_needed(encoder, kept) + second_vectors = encode_needed(encoder, kept) + first_scored = score_jobs(kept, first_vectors, integrated_sha) + second_scored = score_jobs(kept, second_vectors, integrated_sha) + if identity_rows(kept, first_scored, unknown) != identity_rows(kept, second_scored, unknown): + write_failure( + "NOT_DETERMINISTIC", + "RESIDUAL_REPLAY_DETERMINISM_REVIEW_AUTHORIZATION", + "NOT_DETERMINISTIC", + {"integrated_sha256": integrated_sha}, + ) + scores = assemble_scores(sealed, first_scored, unknown, overflow) + if len(scores) != 225: + refuse("score row count drifted") + if any("operator_bucket" in row or row.get("score_status") not in {"SCORED", "UNKNOWN"} for row in scores): + refuse("score row is malformed") + if any(row.get("semantic_noncompositional") is not None for row in scores): + refuse("score row emits a semantic decision") + score_sha = write_jsonl(SCORES_PATH, scores) + if sha256(SCORES_PATH) != score_sha: + refuse("score hash drifted at freeze") + print(f"SCORES_FROZEN {score_sha}", file=sys.stderr, flush=True) + receipt = { + "analysis_plan": analysis_plan(), + "authorization": AUTHORIZATION, + "composition_operator": COMPOSITION_OPERATOR, + "determinism": "IDENTICAL", + "distance_metric": DISTANCE_METRIC, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "model_settings": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "max_sequence_length": MAX_SEQUENCE_LENGTH, + "normalize_embeddings": True, + "seed": 0, + "threads": 1, + }, + "operator_labels_joined": False, + "original_residual_score_sha256": RESIDUAL_SCORES_SHA, + "overflow_count": len(overflow), + "readiness_unknown_count": EXPECTED_UNKNOWN, + "ready_count": EXPECTED_READY, + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_formula": "1 - cosine_similarity(whole_sense_vector, normalized_mean(constituent_sense_vectors))", + "residual_formula_count": 1, + "scored_count": len(first_scored), + "scores_sha256": score_sha, + "sequence": [ + "integrated_resolution_hashed", + "residual_scores_hashed", + "operator_labels_not_yet_joined", + ], + "stored_precision": "10 decimal places", + "yes_no_emitted": False, + } + receipt_sha = write_json(RECEIPT_PATH, receipt) + if receipt["operator_labels_joined"] is not False: + refuse("receipt joined labels early") + buckets = load_operator_buckets() + joined = [] + job_by = {job["row_id"]: job for job in jobs} + score_by = {row["row_id"]: row for row in first_scored} + for row_id, score in score_by.items(): + job = job_by[row_id] + meta = nuisance_for(job) + joined.append( + { + "bucket": buckets[row_id], + "residual_score": score["residual_score"], + "row_id": row_id, + **meta, + } + ) + ready_joined = [] + for job in jobs: + meta = nuisance_for(job) + ready_joined.append({"bucket": buckets[job["row_id"]], "row_id": job["row_id"], **meta}) + ready_by = Counter(row["bucket"] for row in ready_joined) + if ready_by["HIGH"] != 28 or ready_by["SECONDARY"] != 11 or ready_by["REJECT"] != 34 or ready_by["QUARANTINE"] != 0: + write_failure( + "NOT_COMPUTABLE", + "RESIDUAL_REPLAY_READINESS_REVIEW_AUTHORIZATION", + "READINESS_REPRODUCTION_FAILURE", + ) + high = [row["residual_score"] for row in joined if row["bucket"] == "HIGH"] + secondary = [row["residual_score"] for row in joined if row["bucket"] == "SECONDARY"] + reject = [row["residual_score"] for row in joined if row["bucket"] == "REJECT"] + comparison = pair_comparison(high, secondary) + direction = direction_result(comparison) + intervals = bootstrap_intervals(high, secondary) + confounds = confound_report(joined) + decision = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=direction, + tier3_concentrated=confounds["tier3"]["concentrated"], + extremes=confounds["extreme_driven"], + pos_split=confounds["pos_partitioned"], + ) + analysis = { + "bootstrap": intervals, + "comparison": comparison, + "direction": direction, + "high": full_distribution(high), + "hypothesis": "HIGH residual > SECONDARY residual", + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "outliers": { + "HIGH": outlier_pair(joined, "HIGH"), + "REJECT": outlier_pair(joined, "REJECT"), + "SECONDARY": outlier_pair(joined, "SECONDARY"), + }, + "ready_by_operator": {name: ready_by[name] for name in OPERATORS}, + "receipt_sha256": receipt_sha, + "reject": full_distribution(reject), + "scores_sha256": score_sha, + "secondary": full_distribution(secondary), + "threshold_eligible": False, + "yes_no_emitted": False, + } + analysis_sha = write_json(ANALYSIS_PATH, analysis) + confound_payload = { + "composition": composition_counts(joined), + "confounds": confounds, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "ready_tier3_counts": tier_counts(ready_joined), + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + "token_definition": "whitespace_separated_surface_tokens", + } + confound_sha = write_json(CONFOUND_PATH, confound_payload) + decision_payload = { + "analysis_sha256": analysis_sha, + "candidate_status": decision["candidate_status"], + "confound_sha256": confound_sha, + "determinism": "IDENTICAL", + "direction": direction, + "emits_yes_no": False, + "integrated_resolution_sha256": integrated_sha, + "json_schema_document": None, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": decision["next_legal_transition"], + "next_transition_authorized": False, + "original_residual_score_sha256": RESIDUAL_SCORES_SHA, + "receipt_sha256": receipt_sha, + "residual_candidate_spec_sha256": RESIDUAL_SPEC_SHA, + "residual_formula_count": 1, + "runtime_integration": False, + "scores_sha256": score_sha, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "state": decision["state"], + "threshold_eligible": False, + } + decision_sha = write_json(DECISION_PATH, decision_payload) + hashes = { + "analysis_sha256": analysis_sha, + "confound_sha256": confound_sha, + "decision_sha256": decision_sha, + "integrated_sha256": integrated_sha, + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + } + tracker_sha = update_tracker(decision_payload, hashes) + print( + json.dumps( + { + "analysis_sha256": analysis_sha, + "candidate_status": decision["candidate_status"], + "confound_sha256": confound_sha, + "decision_sha256": decision_sha, + "direction": direction, + "integrated_sha256": integrated_sha, + "receipt_sha256": receipt_sha, + "scores_sha256": score_sha, + "tracker_sha256": tracker_sha, + }, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/screen_eval.py b/scripts/shadow/hyperlexical/screen_eval.py new file mode 100644 index 00000000..a14224a4 --- /dev/null +++ b/scripts/shadow/hyperlexical/screen_eval.py @@ -0,0 +1,488 @@ +"""Evaluation lane for a frozen unbind screen. + +Prediction, operator judgment, gold, admission, and settlement stay separate. +This module does not settle a target, admit an identity, or authorize SELECT-005. +""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import json +from datetime import datetime +from pathlib import Path +from typing import Any, Iterable, Mapping, Sequence + +from hyperlexical.holdout_guard import normalized_text_sha256 + +SAMPLE_SCHEMA = "hyperlex.unbind_screen_sample_row.v1" +PREDICTION_SCHEMA = "hyperlex.unbind_screen_prediction.v1" +REVIEW_SCHEMA = "hyperlex.unbind_screen_review_row.v1" +LABEL_SCHEMA = "hyperlex.unbind_screen_operator_label.v1" +REPORT_SCHEMA = "hyperlex.unbind_screen_report.v1" +RECEIPT_SCHEMA = "hyperlex.unbind_screen_evaluation_receipt.v1" + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v3" +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) +RAW_BUCKETS = {"HIGH_VALUE": "HIGH", "HIGH": "HIGH", "SECONDARY": "SECONDARY", "REJECT": "REJECT", "QUARANTINE": "QUARANTINE"} +REASONS = { + "HIGH": frozenset({"STRONG_IDIOM", "PHRASAL_BINDING", "FIXED_NONLITERAL", "VARIABLE_SLOT", "CONVENTIONALIZED_SHIFT"}), + "SECONDARY": frozenset({"LEXICALIZED_TRANSPARENT", "FIXED_COMPOSITIONAL", "DOMAIN_LEXICALIZED"}), + "REJECT": frozenset({ + "PERSON_NAME", "ORGANIZATION", "TITLE_OR_DESIGNATION", "TAXONOMY", + "SPECIES_COMMON_NAME", "TECHNICAL_PROCEDURE", "TECHNICAL_MEASUREMENT", + "PRODUCTIVE_NUMBER", "FREE_COMPOSITION", "REFERENTIAL_DOMINANCE", + }), + "QUARANTINE": frozenset({"DIALECTAL", "ARCHAIC", "AMBIGUOUS_SENSE", "PROVENANCE_UNCLEAR"}), + "UNRESOLVED": frozenset({"INSUFFICIENT_SIGNAL"}), +} +STATES = ( + "DRAFT", + "SAMPLE_FROZEN", + "PREDICTIONS_FROZEN", + "OPERATOR_LABELING", + "LABELS_FROZEN", + "SCORED", + "ERROR_ANALYZED", + "REVISION_ELIGIBLE", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", +}) + + +class ScreenEvalError(ValueError): + pass + + +def refuse(message: str) -> None: + raise ScreenEvalError(message) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def canonical_bucket(raw: str) -> str: + bucket = RAW_BUCKETS.get(str(raw or "")) + if bucket is None: + refuse(f"prediction bucket {raw!r} is not a screen bucket") + return bucket + + +def _load_jsonl(path: Path) -> list[dict[str, Any]]: + rows = [] + for line in path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + rows.append(json.loads(line)) + return rows + + +def _dump_jsonl(path: Path, rows: Sequence[Mapping[str, Any]]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _parse_time(value: str) -> datetime: + text = str(value or "").strip() + if text.endswith("Z"): + text = text[:-1] + "+00:00" + try: + parsed = datetime.fromisoformat(text) + except ValueError as exc: + refuse(f"timestamp {value!r} is not ISO-8601") + raise exc + if parsed.tzinfo is None: + refuse(f"timestamp {value!r} needs a timezone") + return parsed + + +def _surfaces(path: Path) -> set[str]: + return {line.strip() for line in path.read_text(encoding="utf-8").splitlines() if line.strip()} + + +def _train_ids(path: Path) -> set[str]: + found = set() + for row in _load_jsonl(path): + text = str(row.get("text") or row.get("surface") or "") + if text: + found.add(normalized_text_sha256(text)) + return found + + +def _identity_rows(frozen: Sequence[Mapping[str, Any]], *, evaluation_id: str, sample_id: str, sample_sha: str) -> tuple[list[dict[str, Any]], list[dict[str, Any]], list[dict[str, Any]]]: + samples, predictions, reviews = [], [], [] + seen: set[str] = set() + for raw in frozen: + surface = str(raw.get("text") or "") + tokens = list(raw.get("tokens") or []) + if " ".join(str(tok) for tok in tokens) != surface: + refuse(f"tokens do not reconstruct {surface!r}") + row_id = normalized_text_sha256(surface) + if row_id in seen: + refuse(f"duplicate row identity {row_id}") + seen.add(row_id) + provenance = {"source": "wordnet-3.0", "sample_sha256": sample_sha} + samples.append({ + "schema": SAMPLE_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "surface": surface, + "pos": raw.get("source_pos"), + "token_count": len(tokens), + "provenance": provenance, + }) + predictions.append({ + "schema": PREDICTION_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "bucket": canonical_bucket(str(raw.get("predicted") or "")), + "bucket_raw": raw.get("predicted"), + "relation": raw.get("rule"), + "rule_version": RULE_VERSION, + "provenance": provenance, + }) + reviews.append({ + "schema": REVIEW_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "row_id": row_id, + "surface": surface, + "pos": raw.get("source_pos"), + "token_count": len(tokens), + "gloss": raw.get("gloss") or "", + "provenance": provenance, + }) + samples.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + reviews.sort(key=lambda row: row["row_id"]) + return samples, predictions, reviews + + +def _assert_blind(rows: Sequence[Mapping[str, Any]]) -> None: + for row in rows: + leaked = LEAK_KEYS.intersection(row) + if leaked: + refuse(f"blind review carries {sorted(leaked)}") + + +def _assert_disjoint(heldout: Iterable[str], blocked: set[str], name: str) -> None: + overlap = sorted(set(heldout) & blocked) + if overlap: + refuse(f"held-out row is also in {name}") + + +def materialize( + frozen_sample: Path, + out_dir: Path, + *, + expected_sha256: str, + development: Path, + validation_development: Path, + train_jsonl: Path | None = None, + evaluation_id: str = "HLX-EVAL-UNBIND-SCREEN-V3-001", + sample_id: str = "heldout-001", + frozen_at: str = "2026-09-27T22:57:17Z", +) -> dict[str, Any]: + """Copy a frozen prediction file into separate sample, prediction, and blind review artifacts.""" + digest = file_sha256(frozen_sample) + if digest != expected_sha256: + refuse("frozen sample hash does not match the expected seal") + frozen = _load_jsonl(frozen_sample) + samples, predictions, reviews = _identity_rows( + frozen, evaluation_id=evaluation_id, sample_id=sample_id, sample_sha=digest, + ) + _assert_blind(reviews) + held_ids = {row["row_id"] for row in samples} + dev_ids = {normalized_text_sha256(text) for text in _surfaces(development)} + val_ids = {normalized_text_sha256(text) for text in _surfaces(validation_development)} + _assert_disjoint(held_ids, dev_ids, "development") + _assert_disjoint(held_ids, val_ids, "validation_development") + train_checked = train_jsonl is not None + if train_jsonl is not None: + _assert_disjoint(held_ids, _train_ids(train_jsonl), "training_gold") + out = Path(out_dir) + sample_sha = _dump_jsonl(out / "samples" / f"{sample_id}.jsonl", samples) + prediction_sha = _dump_jsonl(out / "predictions" / f"{sample_id}.predictions.jsonl", predictions) + review_sha = _dump_jsonl(out / "operator" / f"{sample_id}.review.jsonl", reviews) + if file_sha256(frozen_sample) != digest: + refuse("frozen sample changed during materialize") + receipt = { + "schema": RECEIPT_SCHEMA, + "evaluation_id": evaluation_id, + "sample_id": sample_id, + "state": "PREDICTIONS_FROZEN", + "rule_version": RULE_VERSION, + "sample_sha256": digest, + "sample_artifact_sha256": sample_sha, + "prediction_artifact_sha256": prediction_sha, + "blind_review_sha256": review_sha, + "operator_label_sha256": None, + "sample_frozen_at": frozen_at, + "development_rows": len(dev_ids), + "validation_development_rows": len(val_ids), + "held_out_rows": len(samples), + "v3_application_count": 1, + "hand_corrections": 0, + "operator_labels": "pending", + "held_out_precision": "NOT_COMPUTABLE", + "confusion_matrix": "NOT_COMPUTABLE", + "train_gold_checked": train_checked, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + } + (out / "reports").mkdir(parents=True, exist_ok=True) + _write_json(out / "reports" / f"{sample_id}.receipt.json", receipt) + return receipt + + +def _write_json(path: Path, body: Mapping[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(body, indent=2, sort_keys=True) + "\n", encoding="utf-8") + + +def _receipt(out_dir: Path, sample_id: str) -> dict[str, Any]: + path = out_dir / "reports" / f"{sample_id}.receipt.json" + return json.loads(path.read_text(encoding="utf-8")) + + +def _check_hashes(out_dir: Path, receipt: Mapping[str, Any]) -> None: + sample_id = str(receipt["sample_id"]) + pairs = { + "sample_artifact_sha256": out_dir / "samples" / f"{sample_id}.jsonl", + "prediction_artifact_sha256": out_dir / "predictions" / f"{sample_id}.predictions.jsonl", + "blind_review_sha256": out_dir / "operator" / f"{sample_id}.review.jsonl", + } + for field, path in pairs.items(): + if file_sha256(path) != receipt[field]: + refuse(f"{field} changed after freeze") + _assert_blind(_load_jsonl(pairs["blind_review_sha256"])) + + +def validate_label(row: Mapping[str, Any], *, frozen_at: str) -> dict[str, Any]: + bucket = str(row.get("operator_bucket") or "") + if bucket not in OPERATOR_BUCKETS: + refuse(f"operator bucket {bucket!r} is not allowed") + reason = str(row.get("reason_code") or "") + if reason not in REASONS[bucket]: + refuse(f"reason {reason!r} is not valid for {bucket}") + row_id = str(row.get("row_id") or "") + if len(row_id) != 64: + refuse("operator label row_id must be the immutable identity") + labeled_at = str(row.get("labeled_at") or "") + if _parse_time(labeled_at) <= _parse_time(frozen_at): + refuse("operator label timestamp is not after the sample freeze") + return { + "schema": LABEL_SCHEMA, + "evaluation_id": row.get("evaluation_id"), + "row_id": row_id, + "operator_bucket": bucket, + "reason_code": reason, + "note": row.get("note"), + "labeled_at": labeled_at, + } + + +def freeze_labels(out_dir: Path, labels_path: Path, *, sample_id: str = "heldout-001") -> dict[str, Any]: + """Store operator labels beside the frozen predictions. Does not score or settle.""" + out_dir = Path(out_dir) + receipt = _receipt(out_dir, sample_id) + if receipt.get("state") not in {"PREDICTIONS_FROZEN", "OPERATOR_LABELING", "LABELS_FROZEN"}: + refuse(f"cannot freeze labels from state {receipt.get('state')}") + _check_hashes(out_dir, receipt) + predictions = _load_jsonl(out_dir / "predictions" / f"{sample_id}.predictions.jsonl") + expected = {row["row_id"] for row in predictions} + cleaned = [validate_label(row, frozen_at=str(receipt["sample_frozen_at"])) for row in _load_jsonl(labels_path)] + got = [row["row_id"] for row in cleaned] + if len(got) != len(set(got)): + refuse("operator labels repeat a row identity") + if set(got) != expected: + refuse("operator labels do not cover the frozen rows exactly once") + cleaned.sort(key=lambda row: row["row_id"]) + label_sha = _dump_jsonl(out_dir / "operator" / f"{sample_id}.labels.jsonl", cleaned) + receipt["operator_label_sha256"] = label_sha + receipt["operator_labels"] = "frozen" + receipt["state"] = "LABELS_FROZEN" + receipt["held_out_precision"] = "NOT_COMPUTABLE" + receipt["confusion_matrix"] = "NOT_COMPUTABLE" + receipt["select_authorized"] = False + _write_json(out_dir / "reports" / f"{sample_id}.receipt.json", receipt) + return receipt + + +def _metrics(pairs: Sequence[tuple[str, str, str, str]]) -> dict[str, Any]: + """pairs are (row_id, predicted, operator, reason).""" + resolved = [item for item in pairs if item[2] != "UNRESOLVED"] + unresolved = [item for item in pairs if item[2] == "UNRESOLVED"] + matrix = {op: {pred: 0 for pred in BUCKETS} for op in OPERATOR_BUCKETS} + for _row_id, pred, op, _reason in pairs: + matrix[op][pred] += 1 + per_bucket = {} + for bucket in BUCKETS: + tp = sum(1 for _i, pred, op, _r in resolved if pred == bucket and op == bucket) + fp = sum(1 for _i, pred, op, _r in resolved if pred == bucket and op != bucket) + fn = sum(1 for _i, pred, op, _r in resolved if op == bucket and pred != bucket) + precision = None if tp + fp == 0 else tp / (tp + fp) + recall = None if tp + fn == 0 else tp / (tp + fn) + f1 = None + if precision is not None and recall is not None and precision + recall: + f1 = 2 * precision * recall / (precision + recall) + per_bucket[bucket] = { + "precision": precision, + "recall": recall, + "f1": f1, + "support": tp + fn, + } + accuracy = None if not resolved else sum(1 for _i, pred, op, _r in resolved if pred == op) / len(resolved) + reasons: dict[str, int] = {} + for _i, _p, _o, reason in pairs: + reasons[reason] = reasons.get(reason, 0) + 1 + return { + "overall_accuracy": accuracy, + "per_bucket": per_bucket, + "confusion_matrix": matrix, + "reason_code_distribution": reasons, + "unresolved_count": len(unresolved), + "resolved_count": len(resolved), + } + + +def _error_class(predicted: str, operator: str) -> str | None: + if operator == "UNRESOLVED" or predicted == operator: + return None + return { + "HIGH": "false_high", + "SECONDARY": "false_secondary", + "REJECT": "false_reject", + "QUARANTINE": "false_quarantine", + }[predicted] + + +def score(out_dir: Path, *, sample_id: str = "heldout-001") -> dict[str, Any]: + """Join labels to predictions on row_id. Pending labels stay NOT_COMPUTABLE.""" + out_dir = Path(out_dir) + receipt = _receipt(out_dir, sample_id) + _check_hashes(out_dir, receipt) + label_path = out_dir / "operator" / f"{sample_id}.labels.jsonl" + report: dict[str, Any] = { + "schema": REPORT_SCHEMA, + "evaluation_id": receipt["evaluation_id"], + "sample_id": sample_id, + "rule_version": RULE_VERSION, + "select_authorized": False, + "revision_eligible": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + if receipt.get("operator_labels") != "frozen" or not label_path.is_file(): + report["status"] = "NOT_COMPUTABLE" + report["held_out_precision"] = "NOT_COMPUTABLE" + report["confusion_matrix"] = "NOT_COMPUTABLE" + report["overall_accuracy"] = "NOT_COMPUTABLE" + _write_json(out_dir / "reports" / f"{sample_id}.metrics.json", report) + return report + if file_sha256(label_path) != receipt.get("operator_label_sha256"): + refuse("operator artifact hash changed after label freeze") + predictions = {row["row_id"]: row for row in _load_jsonl(out_dir / "predictions" / f"{sample_id}.predictions.jsonl")} + samples = {row["row_id"]: row for row in _load_jsonl(out_dir / "samples" / f"{sample_id}.jsonl")} + labels = {row["row_id"]: row for row in _load_jsonl(label_path)} + if set(labels) != set(predictions): + refuse("join row identities do not match") + pairs = [] + errors = [] + classes = {name: 0 for name in ("false_high", "false_secondary", "false_reject", "false_quarantine")} + for row_id in sorted(predictions): + pred = predictions[row_id] + label = labels[row_id] + kind = _error_class(pred["bucket"], label["operator_bucket"]) + pairs.append((row_id, pred["bucket"], label["operator_bucket"], label["reason_code"])) + if kind: + classes[kind] += 1 + errors.append({ + "row_id": row_id, + "surface": samples[row_id]["surface"], + "predicted_bucket": pred["bucket"], + "operator_bucket": label["operator_bucket"], + "prediction_relation": pred["relation"], + "operator_reason": label["reason_code"], + "error_class": kind, + }) + metrics = _metrics(pairs) + report.update(metrics) + report["status"] = "SCORED" + report["error_classes"] = classes + report["select_authorized"] = False + _write_json(out_dir / "reports" / f"{sample_id}.metrics.json", report) + _dump_jsonl(out_dir / "reports" / f"{sample_id}.errors.jsonl", errors) + with (out_dir / "reports" / f"{sample_id}.confusion.csv").open("w", encoding="utf-8", newline="") as handle: + writer = csv.writer(handle) + writer.writerow(["operator_bucket", "predicted_bucket", "count"]) + for operator, cols in metrics["confusion_matrix"].items(): + for predicted, count in cols.items(): + writer.writerow([operator, predicted, count]) + receipt["state"] = "SCORED" + receipt["held_out_precision"] = "SCORED" + receipt["confusion_matrix"] = "SCORED" + receipt["select_authorized"] = False + receipt["revision_eligible"] = False + _write_json(out_dir / "reports" / f"{sample_id}.receipt.json", receipt) + return report + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Score a frozen unbind screen without settling it.") + sub = parser.add_subparsers(dest="command", required=True) + mat = sub.add_parser("materialize") + mat.add_argument("--frozen-sample", type=Path, required=True) + mat.add_argument("--expected-sha", required=True) + mat.add_argument("--development", type=Path, required=True) + mat.add_argument("--validation-development", type=Path, required=True) + mat.add_argument("--train-jsonl", type=Path) + mat.add_argument("--out", type=Path, required=True) + mat.add_argument("--evaluation-id", default="HLX-EVAL-UNBIND-SCREEN-V3-001") + mat.add_argument("--frozen-at", default="2026-09-27T22:57:17Z") + sc = sub.add_parser("score") + sc.add_argument("--evaluation", type=Path, required=True) + fr = sub.add_parser("freeze-labels") + fr.add_argument("--evaluation", type=Path, required=True) + fr.add_argument("--labels", type=Path, required=True) + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + args = build_parser().parse_args(argv) + try: + if args.command == "materialize": + receipt = materialize( + args.frozen_sample, + args.out, + expected_sha256=args.expected_sha, + development=args.development, + validation_development=args.validation_development, + train_jsonl=args.train_jsonl, + evaluation_id=args.evaluation_id, + frozen_at=args.frozen_at, + ) + elif args.command == "freeze-labels": + receipt = freeze_labels(args.evaluation, args.labels) + else: + receipt = score(args.evaluation) + except ScreenEvalError as exc: + raise SystemExit(f"REFUSE: {exc}") from exc + print(json.dumps(receipt, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/shadow/hyperlexical/semantic_compositionality_residual.py b/scripts/shadow/hyperlexical/semantic_compositionality_residual.py new file mode 100644 index 00000000..6287e027 --- /dev/null +++ b/scripts/shadow/hyperlexical/semantic_compositionality_residual.py @@ -0,0 +1,564 @@ +"""Continuous semantic residual for one bound PWN 3.0 sense. + +The score compares the supplied synset with a normalized mean of resolved +constituent synsets. It is not a yes/no label, and it does not select a source. +""" + +from __future__ import annotations + +import hashlib +import math +import struct +from decimal import Decimal, ROUND_HALF_EVEN + +from hyperlexical.km_candidate_evaluation import lookup_key + +CANDIDATE = "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1" +COMPOSITION_OPERATOR = "normalized_mean_v1" +DISTANCE_METRIC = "one_minus_cosine_v1" +MODEL_NAME = "sentence-transformers/all-MiniLM-L6-v2" +MODEL_REVISION = "1110a243fdf4706b3f48f1d95db1a4f5529b4d41" +RESIDUAL_QUANTUM = Decimal("0.0000000001") +RELATION_SYMBOLS = frozenset({"+", "\\"}) +MIN_CONTENT_CONSTITUENTS = 2 +EXTRACTED = "EXTRACTED" +UNKNOWN = "UNKNOWN" +EXACT = "EXACT" +UNIQUE = "UNIQUE" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +SCORED = "SCORED" +RESOLVED = frozenset({EXACT, UNIQUE}) + +STRUCTURAL_TOKENS = frozenset( + { + "a", + "about", + "across", + "after", + "against", + "all", + "amid", + "among", + "amongst", + "an", + "and", + "any", + "are", + "as", + "at", + "be", + "been", + "before", + "behind", + "being", + "below", + "beside", + "between", + "beyond", + "both", + "but", + "by", + "can", + "could", + "did", + "do", + "does", + "down", + "during", + "each", + "every", + "except", + "for", + "from", + "had", + "has", + "have", + "her", + "his", + "if", + "in", + "into", + "is", + "it", + "its", + "least", + "less", + "may", + "might", + "more", + "most", + "must", + "my", + "no", + "none", + "nor", + "not", + "of", + "off", + "on", + "one's", + "onto", + "or", + "our", + "out", + "over", + "per", + "shall", + "should", + "some", + "than", + "that", + "the", + "their", + "them", + "then", + "these", + "this", + "those", + "through", + "to", + "toward", + "towards", + "under", + "up", + "upon", + "via", + "versus", + "was", + "were", + "will", + "with", + "within", + "without", + "would", + "your", + } +) + + +_RULE_AMBIGUOUS = ( + "A content constituent with two or more exact targets, or with no exact " + "target and two or more lexical synsets, abstains the row." +) +_RULE_SHORT = ( + "Whitespace tokenization must leave at least two tokens outside the frozen structural class." +) +_RULE_OVERFLOW = ( + "A whole-sense or constituent text longer than the pinned max sequence length " + "abstains the row. The encoder must not truncate it." +) +_RULE_UNRESOLVED = "A content constituent with no exact target and no lexical synset abstains the row." +_RULE_ZERO = "A non-finite or zero encoder vector, or a zero composed vector, abstains the row." +_HIGH_PROXY = "a proxy for expected noncompositionality" +_SECONDARY_PROXY = "a proxy for expected compositionality" +_REJECT_AXIS = "a referential axis, not semantic no" +_QUARANTINE_AXIS = "not semantic evidence" +_PERCENTILE_RULE = ( + "linear interpolation at rank (n-1)*(p/100), then round-half-even to 10 decimal places" +) +_COMPOSITION_PROCEDURE = ( + "L2-normalize each constituent vector in binary64, take the arithmetic mean, " + "then L2-normalize that mean" +) +_DUPLICATE_SYNSETS = "kept once per content token" +_HYPHEN_RULE = "kept as one token" +_MEMBERSHIP_RULE = "lookup_key of the whitespace token is in the structural class" +_EXACT_RULE = ( + "one synset reached by a derivation or pertainym pointer whose target word " + "number is nonzero and whose target lemma matches the constituent or a one-hop " + "exception neighbor" +) +_AMBIGUOUS_RULE = ( + "more than one exact target synset, or no exact target and more than one lexical synset" +) +_LEXICAL_SCOPE = ( + "one lookup key plus one-hop bidirectional exception neighbors, across noun, verb, adj, and adv" +) +_SOURCE_WORD = "not a filter" +_TARGET_ZERO = "does not name a lemma" +_UNIQUE_RULE = "no exact target, and the lexical keys name one synset" +_UNRESOLVED_RULE = "no exact target and no lexical synset" +_DISTANCE_FORMULA = "1 - binary64_dot(l2_normalize(whole), l2_normalize(composed))" +_DISTANCE_RECORD = "format(value, '.10f'), round-half-even, no clamp" +_TEMPLATE = "{surface} ({pos}): {gloss}" +_TEXT_NORMALIZATION = "underscores become spaces; whitespace collapses; gloss is the supplied first clause" +_VECTOR_HASH_RULE = "sha256 of little-endian binary32 bytes in dimension order" +_WHOLE_SENSE_RULE = "the supplied Hyperlex surface, POS, and frozen gloss; no substitute synset" + + +def candidate_policy() -> dict: + """Return the preregistered design. The dict has no row scores.""" + return { + "abstention_priority": [ + "fewer_than_two_content_constituents", + "leftmost_unresolved_or_ambiguous_content_constituent", + "representation_exceeds_max_sequence_length", + "zero_vector", + ], + "abstention_rules": { + "ambiguous_content_constituent": _RULE_AMBIGUOUS, + "fewer_than_two_content_constituents": _RULE_SHORT, + "representation_exceeds_max_sequence_length": _RULE_OVERFLOW, + "unresolved_content_constituent": _RULE_UNRESOLVED, + "zero_vector": _RULE_ZERO, + }, + "analysis_plan": { + "high_comparison_if_no_high_row_is_scored": "NOT_COMPUTABLE", + "high_is": _HIGH_PROXY, + "join_operator_labels_only_after_score_artifact_is_hashed": True, + "operator_labels_are_scoring_inputs": False, + "percentile": _PERCENTILE_RULE, + "primary_bucket": "HIGH", + "quarantine_is": _QUARANTINE_AXIS, + "reject_is": _REJECT_AXIS, + "secondary_is": _SECONDARY_PROXY, + "semantic_noncompositionality_threshold": None, + }, + "candidate": CANDIDATE, + "composition_operator": { + "duplicate_constituent_synsets": _DUPLICATE_SYNSETS, + "name": COMPOSITION_OPERATOR, + "procedure": _COMPOSITION_PROCEDURE, + "weights": None, + }, + "constituent_extraction": { + "content_minimum": MIN_CONTENT_CONSTITUENTS, + "hyphenated_token": _HYPHEN_RULE, + "membership": _MEMBERSHIP_RULE, + "structural_tokens": sorted(STRUCTURAL_TOKENS), + "tokenizer": "str.split", + }, + "constituent_sense_resolution": { + "ambiguous": _AMBIGUOUS_RULE, + "exact": _EXACT_RULE, + "forbidden": [ + "arbitrary_first_sense", + "embedding_similarity", + "gloss_similarity", + "language_model_judge", + "manual_selection", + "operator_labels", + ], + "lexical_scope": _LEXICAL_SCOPE, + "source_word_number": _SOURCE_WORD, + "target_word_number_zero": _TARGET_ZERO, + "unique": _UNIQUE_RULE, + "unresolved": _UNRESOLVED_RULE, + }, + "distance_metric": { + "formula": _DISTANCE_FORMULA, + "name": DISTANCE_METRIC, + "record": _DISTANCE_RECORD, + }, + "emits_yes_no": False, + "model_identity": { + "name": MODEL_NAME, + "revision": MODEL_REVISION, + }, + "representation_template": _TEMPLATE, + "representation_text_normalization": _TEXT_NORMALIZATION, + "row_unknown_if_any_required_content_constituent_is_not_exact_or_unique": True, + "semantic_noncompositionality_threshold": None, + "status_rule": { + "scored_count_eq_0": { + "next_legal_transition": "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + "status": "CANDIDATE_INSUFFICIENT", + }, + "scored_count_gt_0": { + "next_legal_transition": "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + "status": "CANDIDATE_DISTRIBUTION_FROZEN", + }, + }, + "vector_hash": _VECTOR_HASH_RULE, + "whole_sense": _WHOLE_SENSE_RULE, + } + + +def extract_constituents(surface: str) -> dict: + """Split a surface on whitespace and apply the frozen structural class.""" + tokens = surface.split() + content = [] + structural = [] + for token in tokens: + if lookup_key(token) in STRUCTURAL_TOKENS: + structural.append(token) + else: + content.append(token) + status = EXTRACTED if len(content) >= MIN_CONTENT_CONSTITUENTS else UNKNOWN + return { + "constituent_extraction_status": status, + "content_constituents": content, + "ignored_structural_tokens": structural, + "surface_tokens": tokens, + } + + +def neighbor_keys(text: str, exceptions: dict[str, set[str]]) -> set[str]: + """Return the lookup key plus one-hop exception neighbors.""" + own = lookup_key(text) + keys = {own} + raw = text.casefold().replace("\u2019", "'").replace("\u2018", "'").replace("`", "'") + related: set[str] = set() + for form in (raw, own, own.replace("_", " ")): + related.update(exceptions.get(form, ())) + for item in related: + keys.add(lookup_key(item)) + return keys + + +def exact_synset_ids( + pointers: list[tuple[str, int, str, str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect derivation and pertainym targets that name this constituent.""" + keys = neighbor_keys(constituent, exceptions) + found = [] + for symbol, target_word, synset_id, lemma in pointers: + if symbol not in RELATION_SYMBOLS or target_word <= 0 or not lemma: + continue + if lookup_key(lemma) not in keys: + continue + found.append(synset_id) + return found + + +def lexical_synset_ids( + index: dict[str, list[str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect synsets named by the constituent key or one exception hop.""" + found = [] + for key in sorted(neighbor_keys(constituent, exceptions)): + found.extend(index.get(key, ())) + return found + + +def resolve_constituent(exact_synset_ids_found: list[str], lexical_synset_ids_found: list[str]) -> str: + """Resolve one constituent. Exact evidence outranks lemma polysemy.""" + exact = list(dict.fromkeys(exact_synset_ids_found)) + if len(exact) == 1: + return EXACT + if len(exact) > 1: + return AMBIGUOUS + lexical = list(dict.fromkeys(lexical_synset_ids_found)) + if len(lexical) == 1: + return UNIQUE + if len(lexical) > 1: + return AMBIGUOUS + return UNRESOLVED + + +def resolved_synset(exact_synset_ids_found: list[str], lexical_synset_ids_found: list[str]) -> str | None: + status = resolve_constituent(exact_synset_ids_found, lexical_synset_ids_found) + if status == EXACT: + return list(dict.fromkeys(exact_synset_ids_found))[0] + if status == UNIQUE: + return list(dict.fromkeys(lexical_synset_ids_found))[0] + return None + + +def select_lemma(lemmas: list[str], constituent: str, exceptions: dict[str, set[str]]) -> str: + """Pick one lemma. Matching keys win, and lookup order breaks remaining ties.""" + keys = neighbor_keys(constituent, exceptions) + matches = [lemma for lemma in lemmas if lookup_key(lemma) in keys] + pool = matches or list(lemmas) + if not pool: + raise RuntimeError("resolved synset has no lemma") + return min(pool, key=lookup_key) + + +def representation_text(surface: str, pos: str, gloss: str) -> str: + shown = " ".join(surface.replace("_", " ").split()) + gloss_text = " ".join(gloss.split()) + return f"{shown} ({pos}): {gloss_text}" + + +def l2_normalize(values: list[float]) -> list[float] | None: + if not values or any(not math.isfinite(value) for value in values): + return None + norm = math.sqrt(sum(value * value for value in values)) + if not math.isfinite(norm) or norm == 0.0: + return None + return [value / norm for value in values] + + +def composed_vector(parts: list[list[float]]) -> list[float] | None: + """Normalized mean. Each input vector is normalized again in binary64.""" + if len(parts) < MIN_CONTENT_CONSTITUENTS: + return None + width = len(parts[0]) + if any(len(part) != width for part in parts): + raise RuntimeError("constituent vectors differ in width") + normalized = [] + for part in parts: + unit = l2_normalize(part) + if unit is None: + return None + normalized.append(unit) + count = float(len(normalized)) + mean = [sum(part[index] for part in normalized) / count for index in range(width)] + return l2_normalize(mean) + + +def format_residual(dot: float) -> str: + value = 1.0 - dot + if not math.isfinite(value): + raise RuntimeError("non-finite residual") + if value == 0.0: + value = 0.0 + return format(value, ".10f") + + +def residual_score(whole: list[float], constituents: list[list[float]]) -> tuple[str, list[float]] | None: + """Return the 10-decimal residual and the composed vector.""" + whole_unit = l2_normalize(whole) + composed = composed_vector(constituents) + if whole_unit is None or composed is None: + return None + dot = sum(left * right for left, right in zip(whole_unit, composed)) + if not math.isfinite(dot): + return None + return format_residual(dot), composed + + +def vector_hash(values: list[float]) -> str: + blob = b"".join(struct.pack(" str | None: + if extraction_status != EXTRACTED or len(resolutions) < MIN_CONTENT_CONSTITUENTS: + return "fewer_than_two_content_constituents" + for status in resolutions: + if status == AMBIGUOUS: + return "ambiguous_content_constituent" + if status == UNRESOLVED: + return "unresolved_content_constituent" + if status not in RESOLVED: + raise RuntimeError(f"unknown resolution status {status}") + return None + + +def score_record( + *, + row_id: str, + surface: str, + pos: str, + synset: str, + extraction: dict, + resolutions: list[str], + resolved_synsets: list[str | None], + resolved_lemmas: list[str | None], + whole_representation: str | None, + constituent_representations: list[str] | None, + whole_vector: list[float] | None, + constituent_vectors: list[list[float]] | None, + candidate_spec_sha256: str, + sequence_overflow: bool = False, +) -> dict: + """Score one row. Operator labels are not parameters.""" + reason = primary_abstention(extraction["constituent_extraction_status"], resolutions) + if reason is None and sequence_overflow: + reason = "representation_exceeds_max_sequence_length" + residual = None + whole_hash = None + constituent_hashes = None + composed_hash = None + if reason is None: + if whole_vector is None or constituent_vectors is None: + raise RuntimeError("a resolvable row has no vectors") + if len(constituent_vectors) != len(resolutions): + raise RuntimeError("constituent vector count does not match resolutions") + scored = residual_score(whole_vector, constituent_vectors) + whole_hash = vector_hash(whole_vector) + constituent_hashes = [vector_hash(vector) for vector in constituent_vectors] + if scored is None: + reason = "zero_vector" + else: + residual, composed = scored + composed_hash = vector_hash(composed) + elif whole_vector is not None or constituent_vectors is not None: + raise RuntimeError("an abstaining row was encoded") + extracted = extraction["constituent_extraction_status"] == EXTRACTED + return { + "candidate_spec_sha256": candidate_spec_sha256, + "composition_operator": COMPOSITION_OPERATOR, + "composed_vector_hash": composed_hash, + "constituent_extraction_status": extraction["constituent_extraction_status"], + "constituent_representations": constituent_representations if residual is not None else None, + "constituent_resolution_status": list(resolutions), + "constituent_vector_hashes": constituent_hashes, + "content_constituents": list(extraction["content_constituents"]), + "distance_metric": DISTANCE_METRIC, + "ignored_structural_tokens": list(extraction["ignored_structural_tokens"]), + "pos": pos, + "primary_abstention_reason": reason, + "residual_score": residual, + "resolved_constituent_lemmas": list(resolved_lemmas) if extracted else [], + "resolved_constituent_synsets": list(resolved_synsets) if extracted else [], + "row_id": row_id, + "score_status": SCORED if residual is not None else UNKNOWN, + "surface": surface, + "surface_tokens": list(extraction["surface_tokens"]), + "synset": synset, + "whole_representation": whole_representation if residual is not None else None, + "whole_vector_hash": whole_hash, + } + + +def evaluation_status(scored_count: int) -> tuple[str, str]: + """Map the scored count to a status. The count is not a label agreement.""" + if scored_count < 0: + raise RuntimeError("negative scored count") + if scored_count == 0: + return "CANDIDATE_INSUFFICIENT", "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION" + return "CANDIDATE_DISTRIBUTION_FROZEN", "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION" + + +def percentile(sorted_values: list[str], percent: int) -> str: + """Linear interpolation on already-sorted 10-decimal residual strings.""" + count = len(sorted_values) + if count == 0: + raise RuntimeError("percentile of an empty sample") + if count == 1: + return sorted_values[0] + rank = Decimal(count - 1) * (Decimal(percent) / Decimal(100)) + low = int(rank) + high = min(low + 1, count - 1) + weight = rank - Decimal(low) + blended = Decimal(sorted_values[low]) + (Decimal(sorted_values[high]) - Decimal(sorted_values[low])) * weight + return str(blended.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def decimal_mean(values: list[str]) -> str: + total = sum((Decimal(value) for value in values), start=Decimal(0)) + mean = total / Decimal(len(values)) + return str(mean.quantize(RESIDUAL_QUANTUM, rounding=ROUND_HALF_EVEN)) + + +def distribution(scores: list[str]) -> dict: + if not scores: + return { + "count": 0, + "max": None, + "mean": None, + "median": None, + "min": None, + "p25": None, + "p75": None, + "status": "NOT_COMPUTABLE", + } + ordered = sorted(scores, key=Decimal) + return { + "count": len(ordered), + "max": ordered[-1], + "mean": decimal_mean(ordered), + "median": percentile(ordered, 50), + "min": ordered[0], + "p25": percentile(ordered, 25), + "p75": percentile(ordered, 75), + "status": "DESCRIPTIVE", + } diff --git a/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py b/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py new file mode 100644 index 00000000..95346449 --- /dev/null +++ b/scripts/shadow/hyperlexical/semantic_compositionality_residual_replay.py @@ -0,0 +1,752 @@ +"""Score the semantic-compositionality residual on the 225 development rows. + +The candidate specification is written before any row is encoded. Operator +labels are read only after the score artifact has been hashed. The pass does +not select the source, integrate it, or draw a measurement sample. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from collections import Counter, defaultdict +from datetime import datetime, timezone +from pathlib import Path + +os.environ["MKL_NUM_THREADS"] = "1" +os.environ["OMP_NUM_THREADS"] = "1" +os.environ["TOKENIZERS_PARALLELISM"] = "false" + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + COMPOSITION_OPERATOR, + DISTANCE_METRIC, + MODEL_NAME, + MODEL_REVISION, + candidate_policy, + distribution, + evaluation_status, + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + primary_abstention, + representation_text, + resolve_constituent, + resolved_synset, + score_record, + select_lemma, + vector_hash, +) +from hyperlexical.unbind_sense_screen_v1 import load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +SENSE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +SOURCE = LEDGER / "operator-review/HLX-EVAL-UNBIND-SEMANTIC-EVIDENCE-SOURCE-V1-001" +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +MODEL_DIR = ( + LEDGER + / "acquisition/sources/all-MiniLM-L6-v2" + / MODEL_REVISION +) +TRACKER = SENSE / "HYPOTHESIS.json" +EVIDENCE_MANIFEST = SENSE / "DEVELOPMENT_EVIDENCE.json" +EVENTS = LEDGER / "events.jsonl" +LEDGER_FILE = LEDGER / "ledger.json" + +PROVENANCE_PATH = SOURCE / "RESIDUAL_SOURCE_PROVENANCE.json" +LICENSE_PATH = SOURCE / "RESIDUAL_LICENSE_RECEIPT.json" +SPEC_PATH = SOURCE / "RESIDUAL_CANDIDATE_SPEC.json" +SCORES_PATH = SOURCE / "RESIDUAL_DEVELOPMENT_SCORES.jsonl" +EVALUATION_PATH = SOURCE / "RESIDUAL_DEVELOPMENT_EVALUATION.json" +DECISION_PATH = SOURCE / "RESIDUAL_CANDIDATE_DECISION.json" + +OPERATORS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE", "UNRESOLVED") +OPERATOR_COUNTS = {"HIGH": 101, "SECONDARY": 46, "REJECT": 77, "QUARANTINE": 1, "UNRESOLVED": 0} +POS_NAME = {"n": "noun", "v": "verb", "a": "adj", "r": "adv", "s": "adj"} +FILE_POS = frozenset({"noun", "verb", "adj", "adv"}) +MAX_SEQUENCE_LENGTH = 256 +OUTPUT_DIMENSION = 384 + +EXPECTED = { + SENSE / "CLASSIFICATION_PROCEDURE.json": "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + SENSE / "CLASSIFICATION_PROCEDURE.v2.json": "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + SENSE / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + SENSE / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE_MANIFEST: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + SENSE / "development_replay_predictions.jsonl": "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + SENSE / "development_replay_report.json": "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + SENSE / "development_replay_v2_predictions.jsonl": "1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a", + SENSE / "development_replay_v2_report.json": "93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5", + SENSE / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + SENSE / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + SENSE / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + SENSE / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + SENSE / "V2_DEVELOPMENT_RESULT_REVIEW.json": "77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d", + SENSE / "WORDNET_STRUCTURAL_SOURCE_LIMITATION.json": "3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a", + SOURCE / "HYPOTHESIS.json": "39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787", + SOURCE / "ACCEPTANCE.json": "1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94", + SOURCE / "CANDIDATE_SOURCE_EVALUATION_PLAN.json": "472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a", + SOURCE / "SEMANTIC_COMPOSITIONALITY.architecture.json": "180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36", + SOURCE / "MAGPIE_SOURCE_PROVENANCE.json": "bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd", + SOURCE / "MAGPIE_LICENSE_RECEIPT.json": "8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972", + SOURCE / "MAGPIE_DEVELOPMENT_MATCHES.jsonl": "84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83", + SOURCE / "MAGPIE_SENSE_ALIGNMENT.jsonl": "037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66", + SOURCE / "MAGPIE_SEMANTIC_EVIDENCE.jsonl": "89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577", + SOURCE / "MAGPIE_DEVELOPMENT_EVALUATION.json": "74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba", + SOURCE / "MAGPIE_CANDIDATE_DECISION.json": "6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d", + SOURCE / "KM_SOURCE_PROVENANCE.json": "88fbfa077d2394b8ce631ec700c482ad98a0f62f0ab06964f4f43a77d906f9aa", + SOURCE / "KM_LICENSE_RECEIPT.json": "eb4c9406aab7f9021d346ebd24634cad1dcf0e6076f069e73b1d4b2702dbe2f8", + SOURCE / "KM_RAW_EVALUATION_INVENTORY.jsonl": "3a35ddc0b2c04a5386c6112a2bb3cdf22735edbe2fd791f0c2ec542fe9184d81", + SOURCE / "KM_PWN30_ALIGNMENT.jsonl": "531d13cf1bdbc939fc11d9ef5864c6878a9f58210368c596c0aad08ff75e61c9", + SOURCE / "KM_HYPERLEX_SEMANTIC_EVIDENCE.jsonl": "4e98f8f06713ffcf02549305aef140e92b9b8b2790f471f3ce1f41fe208b2f7a", + SOURCE / "KM_DEVELOPMENT_EVALUATION.json": "00a1d1f667635354e20e5002c4ece846fe3a8125a7ca12ebe09bb7e28dedd1a7", + SOURCE / "KM_MAGPIE_COMPARISON.json": "240b3ea468d22a80ac5e5765521cc691b80f914b7baff6bb1ae4819475f54965", + SOURCE / "KM_CANDIDATE_DECISION.json": "93a07e3c78b53a69965497410c34ddb52c2a5d3add3fb2f3cd3fd9ca84eb3d9f", + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7/measurement_error_analysis.json": "ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8", + EVENTS: "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + LEDGER_FILE: "77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0", + TRACKER: "b3546102410058d3596c4563604998685753a698b2bc533b353e1f039440f704", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v2.py": "4b6f125da435b365af143c187902093bee5c9502db5b813a3d2bba11f889fa2b", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", + HYPERLEX / "scripts/shadow/hyperlexical/km_candidate_evaluation.py": "c6e6cb69a215de395023fa44237fc4a9f1a02199fd4b92627b7d9345b2b5fbeb", + HYPERLEX / "scripts/shadow/hyperlexical/km_candidate_evaluation_replay.py": "274da840325e0782c938a42a7e5f8ca7a3945982391cd144a50e674059bb333f", + HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation.py": "3cf19b6e3468e55d0636d4df0d2882bc886e9c7024548653a08c26f7ab43c7a9", + HYPERLEX / "scripts/shadow/hyperlexical/magpie_candidate_evaluation_replay.py": "f01c3361956ae81df772c7b78448d7a58dad342a1f4fe7d084ef00a889587710", + WORDNET / "data.noun": "489f145e0f68877c0be5bd0eb4117adaaac52f38f6204eb8d85dbe2158b614cc", + WORDNET / "data.verb": "29cc96ed80c9f47d94fe75e332a9df80f4b1c737205f92d2f433d63c6da2ab51", + WORDNET / "data.adj": "f24b635368be441501c9b8001e9271fd3b30b203f00d91e332979e6f8fe35646", + WORDNET / "data.adv": "e66dbbda0e0359e41b7f225bff71dd0c263dc7c66c1b61abc9ba334973d92979", + WORDNET / "README": "adad8d28ddea1db05b67ba1ac23506b025d29e0bcbf23bb35dde346089d8808d", + WORDNET / "LICENSE": "7731175a77952e259390b496fab905e57118b8d19ad3a8383c67eee724ff443f", + MODEL_DIR / "model.safetensors": "53aa51172d142c89d9012cce15ae4d6cc0ca6895895114379cacb4fab128d9db", + MODEL_DIR / "tokenizer.json": "be50c3628f2bf5bb5e3a7f17b1f74611b2561a3a27eeab05e5aa30f411572037", + MODEL_DIR / "vocab.txt": "07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3", +} + +MODEL_FILES = ( + "1_Pooling/config.json", + "README.md", + "config.json", + "config_sentence_transformers.json", + "model.safetensors", + "modules.json", + "sentence_bert_config.json", + "special_tokens_map.json", + "tokenizer.json", + "tokenizer_config.json", + "vocab.txt", +) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def check_sealed(skip: set[Path] | None = None) -> None: + skipped = skip or set() + for path, expected in EXPECTED.items(): + if path in skipped: + continue + if sha256(path) != expected: + refuse(f"sealed file changed: {path}") + + +def package_license(name: str) -> str: + root = Path("/home/morpheus/hlx-private/venv-residual/lib") + matches = sorted(root.glob(f"python*/site-packages/{name}-*.dist-info/METADATA")) + if not matches: + refuse(f"missing package metadata for {name}") + header = matches[-1].read_text(encoding="utf-8", errors="replace").split("\n\n", 1)[0] + expression = None + generic = None + classified = None + for line in header.splitlines(): + if line.startswith("License-Expression:"): + expression = line.split(":", 1)[1].strip() + elif line.startswith("License:"): + generic = line.split(":", 1)[1].strip() + elif line.startswith("Classifier: License"): + classified = line.split("::")[-1].strip() + found = expression or generic or classified + if not found: + refuse(f"no license line for {name}") + return found + + +def build_indexes() -> tuple[dict, dict]: + synsets, glosses = load_wordnet(WORDNET) + by_id = {} + index = defaultdict(list) + for (pos, offset), synset in synsets.items(): + if pos not in FILE_POS: + continue + synset_id = f"{pos}:{offset}" + if synset_id in by_id: + continue + by_id[synset_id] = { + "gloss": " ".join(glosses[(pos, offset)].split()), + "lemmas": list(synset.lemmas), + "pos": pos, + "synset": synset, + } + for lemma in synset.lemmas: + index[lookup_key(lemma)].append(synset_id) + for key, identifiers in index.items(): + index[key] = sorted(set(identifiers)) + return by_id, dict(index) + + +def pointer_records(synset_id: str, by_id: dict) -> list[tuple[str, int, str, str]]: + record = by_id.get(synset_id) + if record is None: + return [] + rows = [] + for pointer in record["synset"].pointers: + pos = POS_NAME.get(pointer.pos) + if pos is None: + continue + target_id = f"{pos}:{pointer.offset}" + target = by_id.get(target_id) + if target is None or pointer.target < 1 or pointer.target > len(target["lemmas"]): + lemma = "" + else: + lemma = target["lemmas"][pointer.target - 1] + rows.append((pointer.symbol, pointer.target, target_id, lemma)) + return rows + + +def prepare_rows(manifest_rows: list[dict], by_id: dict, index: dict, exceptions: dict[str, set[str]]) -> list[dict]: + prepared = [] + for row in manifest_rows: + if row.get("sense_class") is not None: + refuse("development row has a sense class") + if row.get("pos") != row.get("synset_pos"): + refuse("row POS and synset POS differ") + if not row.get("synset_offset") or not row.get("gloss") or not row.get("surface"): + refuse("development row is missing surface, gloss, or synset") + synset = f"{row['synset_pos']}:{row['synset_offset']}" + extraction = extract_constituents(row["surface"]) + pointers = pointer_records(synset, by_id) + resolutions = [] + synsets = [] + lemmas = [] + representations = [] + for constituent in extraction["content_constituents"]: + exact = exact_synset_ids(pointers, constituent, exceptions) + lexical = lexical_synset_ids(index, constituent, exceptions) + status = resolve_constituent(exact, lexical) + chosen = resolved_synset(exact, lexical) + resolutions.append(status) + synsets.append(chosen) + if chosen is None: + lemmas.append(None) + continue + record = by_id.get(chosen) + if record is None: + refuse(f"resolved synset is not in PWN 3.0: {chosen}") + if status == "EXACT": + pool = [ + lemma + for symbol, target_word, target_id, lemma in pointers + if target_id == chosen and symbol in {"+", "\\"} and target_word > 0 and lemma + ] + else: + pool = list(record["lemmas"]) + lemma = select_lemma(pool, constituent, exceptions) + lemmas.append(lemma) + representations.append(representation_text(lemma, record["pos"], record["gloss"])) + reason = primary_abstention(extraction["constituent_extraction_status"], resolutions) + whole = None + constituent_texts = None + if reason is None: + whole = representation_text(row["surface"], row["pos"], row["gloss"]) + constituent_texts = representations + prepared.append( + { + "constituent_representations": constituent_texts, + "extraction": extraction, + "gloss": row["gloss"], + "lemmas": lemmas, + "pos": row["pos"], + "resolutions": resolutions, + "row_id": row["row_id"], + "surface": row["surface"], + "synset": synset, + "synsets": synsets, + "whole_representation": whole, + } + ) + return prepared + + +def load_encoder(): + import torch + from sentence_transformers import SentenceTransformer + + torch.manual_seed(0) + torch.set_num_threads(1) + try: + torch.set_num_interop_threads(1) + except RuntimeError: + pass + model = SentenceTransformer( + str(MODEL_DIR), + device="cpu", + local_files_only=True, + backend="torch", + model_kwargs={"torch_dtype": torch.float32}, + ) + model.eval() + if int(model.max_seq_length) != MAX_SEQUENCE_LENGTH: + refuse(f"max sequence length is {model.max_seq_length}") + return model + + +def token_length(model, text: str) -> int: + encoded = model.tokenizer(text, add_special_tokens=True, truncation=False) + return len(encoded["input_ids"]) + + +def encode_texts(model, texts: list[str]) -> dict[str, list[float]]: + import torch + + encoded = {} + with torch.inference_mode(): + for text in texts: + torch.manual_seed(0) + vector = model.encode( + [text], + batch_size=1, + convert_to_numpy=True, + device="cpu", + normalize_embeddings=True, + precision="float32", + show_progress_bar=False, + )[0] + values = [float(item) for item in vector.tolist()] + if len(values) != OUTPUT_DIMENSION: + refuse(f"encoder width is {len(values)}") + encoded[text] = values + return encoded + + +def runtime_versions() -> dict: + import numpy + import tokenizers + import torch + import transformers + import sentence_transformers + + return { + "numpy": numpy.__version__, + "python": ".".join(map(str, __import__("sys").version_info[:3])), + "sentence_transformers": sentence_transformers.__version__, + "tokenizers": tokenizers.__version__, + "torch": torch.__version__, + "transformers": transformers.__version__, + } + + +def file_hashes() -> dict[str, str]: + return {name: sha256(MODEL_DIR / name) for name in MODEL_FILES} + + +def iso_mtime(path: Path) -> str: + stamp = datetime.fromtimestamp(path.stat().st_mtime, timezone.utc) + return stamp.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def main() -> None: + check_sealed() + manifest = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + rows = manifest["rows"] + if len(rows) != 225: + refuse(f"manifest row count is {len(rows)}") + for row in rows: + row.pop("operator_bucket", None) + row.pop("operator_reason", None) + row.pop("historical_unbind_screen", None) + exceptions = load_exceptions(WORDNET) + by_id, index = build_indexes() + prepared = prepare_rows(rows, by_id, index, exceptions) + if len(prepared) != 225: + refuse("prepared row count drifted") + + import torch + + torch.manual_seed(0) + torch.set_num_threads(1) + versions = runtime_versions() + hashes = file_hashes() + pooling = json.loads((MODEL_DIR / "1_Pooling/config.json").read_text(encoding="utf-8")) + tokenizer_config = json.loads((MODEL_DIR / "tokenizer_config.json").read_text(encoding="utf-8")) + modules = json.loads((MODEL_DIR / "modules.json").read_text(encoding="utf-8")) + provenance = { + "acquired_at": iso_mtime(MODEL_DIR / "model.safetensors"), + "api_embedding_service_used": False, + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "device": "cpu", + "dtype": "float32", + "file_sha256": hashes, + "json_schema_document": None, + "local_dir": str(MODEL_DIR), + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "model_source": "https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2", + "modules": modules, + "not_acquired": ["onnx", "openvino", "pytorch_model.bin", "tf_model.h5"], + "output_dimension": OUTPUT_DIMENSION, + "pooling": pooling, + "runtime_versions": versions, + "schema": "hyperlex.residual_source_provenance.v1", + "tokenizer_do_lower_case": tokenizer_config["do_lower_case"], + "tokenizer_model_max_length": tokenizer_config["model_max_length"], + "weights_sha256": hashes["model.safetensors"], + "wordnet_license_sha256": EXPECTED[WORDNET / "LICENSE"], + "wordnet_readme_sha256": EXPECTED[WORDNET / "README"], + "wordnet_root": str(WORDNET), + } + provenance_sha = write_json(PROVENANCE_PATH, provenance) + license_receipt = { + "api_embedding_service_used": False, + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "json_schema_document": None, + "model_license": "apache-2.0", + "model_license_source": "README.md front matter of the pinned revision", + "runtime_licenses": { + "numpy": package_license("numpy"), + "sentence_transformers": package_license("sentence_transformers"), + "tokenizers": package_license("tokenizers"), + "torch": package_license("torch"), + "transformers": package_license("transformers"), + }, + "schema": "hyperlex.residual_license_receipt.v1", + "wordnet_license": "WordNet Release 3.0, Copyright 2006 Princeton University", + "wordnet_license_sha256": EXPECTED[WORDNET / "LICENSE"], + } + receipt_sha = write_json(LICENSE_PATH, license_receipt) + spec = { + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "cuda_available_at_spec_freeze": bool(torch.cuda.is_available()), + "determinism": { + "batch_size": 1, + "device": "cpu", + "dtype": "float32", + "eval_mode": True, + "inference_mode": True, + "interop_threads": 1, + "mkl_num_threads": "1", + "normalize_embeddings": True, + "num_threads": 1, + "omp_num_threads": "1", + "precision": "float32", + "seed": 0, + "tokenizers_parallelism": "false", + "use_deterministic_algorithms": False, + }, + "effective_max_sequence_length": MAX_SEQUENCE_LENGTH, + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "model_name": MODEL_NAME, + "model_revision": MODEL_REVISION, + "output_dimension": OUTPUT_DIMENSION, + "policy": candidate_policy(), + "provenance_sha256": provenance_sha, + "row_scores_included": False, + "runtime_versions": versions, + "schema": "hyperlex.residual_candidate_spec.v1", + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + "tokenizer_json_sha256": hashes["tokenizer.json"], + "tokenizer_revision": MODEL_REVISION, + "vocab_sha256": hashes["vocab.txt"], + "weights_sha256": hashes["model.safetensors"], + } + spec_text = json.dumps(spec, indent=2, sort_keys=True, ensure_ascii=True) + if "residual_score" in spec_text or "operator_bucket" in spec_text: + refuse("candidate spec contains a score or an operator label") + spec_sha = write_json(SPEC_PATH, spec) + if sha256(SPEC_PATH) != spec_sha: + refuse("spec hash mismatch") + + model = load_encoder() + for item in prepared: + item["sequence_overflow"] = False + if item["whole_representation"] is None: + continue + texts = [item["whole_representation"], *item["constituent_representations"]] + if any(token_length(model, text) > MAX_SEQUENCE_LENGTH for text in texts): + item["sequence_overflow"] = True + item["whole_representation"] = None + item["constituent_representations"] = None + needed = [] + seen = set() + for item in prepared: + if item["whole_representation"] is None: + continue + for text in [item["whole_representation"], *item["constituent_representations"]]: + if text not in seen: + seen.add(text) + needed.append(text) + needed.sort() + first = encode_texts(model, needed) + second = encode_texts(model, needed) + for text in needed: + if vector_hash(first[text]) != vector_hash(second[text]): + refuse("encoder replay did not match") + vectors = first + + scores = [] + for item in prepared: + whole_vector = None + constituent_vectors = None + if item["whole_representation"] is not None: + whole_vector = vectors[item["whole_representation"]] + constituent_vectors = [vectors[text] for text in item["constituent_representations"]] + record = score_record( + row_id=item["row_id"], + surface=item["surface"], + pos=item["pos"], + synset=item["synset"], + extraction=item["extraction"], + resolutions=item["resolutions"], + resolved_synsets=item["synsets"], + resolved_lemmas=item["lemmas"], + whole_representation=item["whole_representation"], + constituent_representations=item["constituent_representations"], + whole_vector=whole_vector, + constituent_vectors=constituent_vectors, + candidate_spec_sha256=spec_sha, + sequence_overflow=item["sequence_overflow"], + ) + if "operator_bucket" in record: + refuse("score row carries an operator bucket") + scores.append(record) + if [row["row_id"] for row in scores] != [row["row_id"] for row in rows]: + refuse("score order drifted from the manifest") + score_sha = write_jsonl(SCORES_PATH, scores) + if sha256(SCORES_PATH) != score_sha: + refuse("score hash mismatch") + if sha256(SPEC_PATH) != spec_sha: + refuse("scoring mutated the spec") + + rejoined = json.loads(EVIDENCE_MANIFEST.read_text(encoding="utf-8")) + if sha256(EVIDENCE_MANIFEST) != EXPECTED[EVIDENCE_MANIFEST]: + refuse("manifest changed during scoring") + buckets = {} + for row in rejoined["rows"]: + buckets[row["row_id"]] = row["operator_bucket"] + counted = Counter(buckets.values()) + for name, expected_count in OPERATOR_COUNTS.items(): + if counted[name] != expected_count: + refuse(f"operator count {name} is {counted[name]}") + if any(row.get("sense_class") is not None for row in rejoined["rows"]): + refuse("manifest sense class changed") + + by_operator = {name: [] for name in OPERATORS} + unknown_by_operator = Counter() + for record in scores: + bucket = buckets[record["row_id"]] + if bucket not in by_operator: + refuse(f"unexpected operator bucket {bucket}") + if record["score_status"] == "SCORED": + by_operator[bucket].append(record["residual_score"]) + else: + unknown_by_operator[bucket] += 1 + distributions = {name: distribution(by_operator[name]) for name in OPERATORS} + high = distributions["HIGH"] + high_comparison = "NOT_COMPUTABLE" if high["count"] == 0 else "DESCRIPTIVE_ONLY" + scored_count = sum(item["count"] for item in distributions.values()) + status, next_transition = evaluation_status(scored_count) + extraction_counts = Counter(record["constituent_extraction_status"] for record in scores) + resolution_counts = Counter( + status_name + for record in scores + for status_name in record["constituent_resolution_status"] + ) + abstention_counts = Counter( + record["primary_abstention_reason"] + for record in scores + if record["primary_abstention_reason"] + ) + odd_content_tokens = 0 + content_tokens = 0 + for record in scores: + for token in record["content_constituents"]: + content_tokens += 1 + folded = token.casefold() + if any(not (character.isalpha() or character.isdigit() or character in "-'") for character in folded): + odd_content_tokens += 1 + evaluation = { + "abstention_reason_counts": dict(sorted(abstention_counts.items())), + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "candidate_rule": "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1", + "candidate_spec_sha256": spec_sha, + "composition_operator": COMPOSITION_OPERATOR, + "content_tokens": content_tokens, + "content_tokens_outside_letter_digit_hyphen_apostrophe": odd_content_tokens, + "distance_metric": DISTANCE_METRIC, + "distributions_by_operator": distributions, + "emits_yes_no": False, + "encode_replay_matched": True, + "encoded_text_count": len(needed), + "extraction_counts": dict(sorted(extraction_counts.items())), + "high_comparison": high_comparison, + "high_distribution": high, + "high_proxy": "expected noncompositionality; not an identity", + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "operator_counts": {name: OPERATOR_COUNTS[name] for name in OPERATORS}, + "operator_labels_joined_after_score_artifact_was_hashed": True, + "operator_labels_used_as_scoring_inputs": False, + "operator_unknown_counts": {name: unknown_by_operator[name] for name in OPERATORS}, + "primary_analysis_bucket": "HIGH", + "quarantine_is_not_semantic_evidence": True, + "reject_is_not_semantic_no": True, + "resolution_counts": dict(sorted(resolution_counts.items())), + "row_count": 225, + "schema": "hyperlex.residual_development_evaluation.v1", + "score_artifact_sha256": score_sha, + "scored_count": scored_count, + "secondary_is_a_proxy_for_expected_compositionality": True, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "source_provenance_sha256": provenance_sha, + "unknown_count": 225 - scored_count, + } + evaluation_sha = write_json(EVALUATION_PATH, evaluation) + decision = { + "candidate": "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + "candidate_rule": "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1", + "candidate_spec_sha256": spec_sha, + "evaluation_sha256": evaluation_sha, + "evaluation_status": status, + "high_comparison": high_comparison, + "high_scored_count": high["count"], + "json_schema_document": None, + "license_receipt_sha256": receipt_sha, + "measurement_eligible": False, + "measurement_sample_drawn": False, + "next_legal_transition": next_transition, + "next_transition_authorized": False, + "runtime_integration": False, + "schema": "hyperlex.residual_candidate_decision.v1", + "score_artifact_sha256": score_sha, + "scored_count": scored_count, + "select_005_authorized": False, + "selected_source": "none", + "semantic_noncompositionality_threshold": None, + "source_provenance_sha256": provenance_sha, + "state": "CANDIDATE_SOURCE_EVALUATED", + "unknown_count": 225 - scored_count, + "why": ( + "The residual is a continuous score. No semantic-noncompositionality " + "threshold is chosen in this pass, and operator labels were joined " + "only after the score file was hashed." + ), + "yes_no_emitted": False, + } + decision_sha = write_json(DECISION_PATH, decision) + check_sealed(skip={TRACKER}) + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + tracker["previous_state"] = tracker.get("state") + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["semantic_evidence_source_state"] = "CANDIDATE_SOURCE_EVALUATED" + tracker["evaluated_candidates"] = [ + "MAGPIE", + "KORKONTZELOS_MANANDHAR", + "SEMANTIC_COMPOSITIONALITY_RESIDUAL", + ] + tracker["latest_evaluated_candidate"] = "SEMANTIC_COMPOSITIONALITY_RESIDUAL" + tracker["residual_candidate"] = "RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1" + tracker["residual_evaluation_status"] = status + tracker["residual_candidate_spec_sha256"] = spec_sha + tracker["residual_source_provenance_sha256"] = provenance_sha + tracker["residual_license_receipt_sha256"] = receipt_sha + tracker["residual_development_scores_sha256"] = score_sha + tracker["residual_development_evaluation_sha256"] = evaluation_sha + tracker["residual_candidate_decision_sha256"] = decision_sha + tracker["residual_model_name"] = MODEL_NAME + tracker["residual_model_revision"] = MODEL_REVISION + tracker["residual_scored_count"] = scored_count + tracker["residual_unknown_count"] = 225 - scored_count + tracker["residual_high_comparison"] = high_comparison + tracker["residual_threshold"] = None + tracker["residual_yes_no_emitted"] = False + tracker["selected_source"] = "none" + tracker["semantic_evidence_source_selected"] = "none" + tracker["semantic_evidence_source_runtime_integration"] = False + tracker["semantic_evidence_source_applied"] = False + tracker["semantic_evidence_source_encoded"] = False + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["admitted"] = 0 + tracker["settled"] = 0 + tracker["gold"] = 0 + tracker["procedure_v3_created"] = False + tracker["procedure_v2_retuned"] = False + tracker["next_legal_transition"] = next_transition + tracker["next_transition_authorized"] = False + tracker_sha = write_json(TRACKER, tracker) + check_sealed(skip={TRACKER}) + if sha256(EVENTS) != EXPECTED[EVENTS] or sha256(LEDGER_FILE) != EXPECTED[LEDGER_FILE]: + refuse("ledger or events changed") + print( + json.dumps( + { + "abstention_reason_counts": evaluation["abstention_reason_counts"], + "candidate_decision_sha256": decision_sha, + "candidate_spec_sha256": spec_sha, + "development_evaluation_sha256": evaluation_sha, + "development_scores_sha256": score_sha, + "distributions_by_operator": distributions, + "encoded_text_count": len(needed), + "evaluation_status": status, + "extraction_counts": evaluation["extraction_counts"], + "high_comparison": high_comparison, + "license_receipt_sha256": receipt_sha, + "next_legal_transition": next_transition, + "resolution_counts": evaluation["resolution_counts"], + "scored_count": scored_count, + "selected_source": "none", + "source_provenance_sha256": provenance_sha, + "tracker_sha256": tracker_sha, + "unknown_count": 225 - scored_count, + }, + indent=2, + sort_keys=True, + ) + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v3.py b/scripts/shadow/hyperlexical/unbind_screen_v3.py new file mode 100644 index 00000000..3c2ab94d --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v3.py @@ -0,0 +1,204 @@ +"""Frozen RUNE.UNBIND_SCREEN.v3. + +The bucket function is the scored screen. v4 may call it to obtain a bucket. +v4 must not rewrite this decision procedure. +""" + +from __future__ import annotations + +import re +from functools import lru_cache +from pathlib import Path + +FILES = { + "noun": ("index.noun", "data.noun"), + "verb": ("index.verb", "data.verb"), + "adj": ("index.adj", "data.adj"), + "adv": ("index.adv", "data.adv"), +} +STOP = { + "a", "an", "the", "of", "to", "in", "on", "for", "and", "or", "with", "from", + "by", "at", "as", "into", "over", "that", "this", "it", "be", "is", "are", + "was", "were", "been", "being", "not", "no", "than", "then", "if", "but", + "its", "who", "which", "when", "where", "what", "how", "about", "such", + "all", "every", "other", "one", +} +GEO_HEADS = frozenset({ + "gulf", "bay", "cape", "lake", "sea", "mount", "strait", "ocean", "island", + "river", "port", "peninsula", "isthmus", "archipelago", "republic", "kingdom", +}) +INST_HEADS = frozenset({ + "department", "ministry", "bureau", "agency", "university", "committee", + "commission", "court", "office", "board", "council", "senate", "congress", + "parliament", "institute", "administration", "authority", +}) +PLACES = frozenset({ + "alabama", "alaska", "arizona", "arkansas", "california", "colorado", + "connecticut", "delaware", "florida", "georgia", "hawaii", "idaho", + "illinois", "indiana", "iowa", "kansas", "kentucky", "louisiana", "maine", + "maryland", "massachusetts", "michigan", "minnesota", "mississippi", + "missouri", "montana", "nebraska", "nevada", "hampshire", "jersey", + "mexico", "york", "carolina", "dakota", "ohio", "oklahoma", "oregon", + "pennsylvania", "rhode", "tennessee", "texas", "utah", "vermont", + "virginia", "washington", "wisconsin", "wyoming", "america", "canada", + "brazil", "argentina", "chile", "peru", "france", "germany", "italy", + "spain", "portugal", "ireland", "scotland", "england", "britain", "europe", + "africa", "asia", "australia", "india", "china", "japan", "korea", "egypt", + "greece", "russia", "poland", +}) +PARTICLES = frozenset({ + "out", "up", "off", "away", "down", "back", "over", "through", "apart", + "aside", "along", "in", "on", +}) +POSSESSORS = frozenset({"my", "your", "his", "her", "our", "their", "one's"}) +DEGREE = frozenset({"too", "very", "more", "most", "so", "quite", "rather", "extremely"}) +NUMBERS = frozenset({ + "zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", + "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen", + "seventeen", "eighteen", "nineteen", "twenty", "thirty", "forty", "fifty", + "sixty", "seventy", "eighty", "ninety", "hundred", "thousand", "million", "billion", +}) +TITLE_NOUNS = frozenset({ + "law", "laws", "theory", "rules", "principle", "theorem", "equation", + "constant", "effect", "doctrine", +}) +NAME_PARTICLES = frozenset({"de", "von", "van", "di", "da", "del", "du", "des"}) +TITLE_HEADS = frozenset({"master", "bachelor", "doctor"}) +PROCEDURE_TAILS = frozenset({"surgery", "procedure", "operation"}) +TAXON_SUFFIX = ("idae", "aceae", "inae", "iformes", "oidea") +LATIN_SPECIES = re.compile(r"(ensis|oides|aceae|aris)$") +INITIALISM = re.compile(r"[a-z]\.$") +POSSESSIVE = re.compile(r"[a-z]+'s$") +NAME_TOKEN = re.compile(r"[a-z]+$") +YEAR = re.compile(r"\b(?:1[0-9]{3}|20[0-9]{2})\b") + +def stems(text): + out = [] + for word in re.findall(r"[a-z']+", text.lower()): + word = word.strip("'") + if len(word) < 3 or word in STOP: + continue + out.append(word[:4]) + return out + +def load_index(path): + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + parts = line.split() + if "_" not in parts[0]: + continue + synset_cnt = int(parts[2]) + p_cnt = int(parts[3]) + rest = parts[4 + p_cnt:] + offsets = rest[2:2 + synset_cnt] + if offsets: + found[parts[0]] = offsets[0] + return found + +def load_glosses(path, wanted): + glosses = {} + want = set(wanted) + if not want: + return glosses + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + offset = line.split(" ", 1)[0] + if offset not in want: + continue + raw = line.split("|", 1)[1].strip() + glosses[offset] = raw.split(";", 1)[0].strip() + if len(glosses) == len(want): + break + return glosses + +@lru_cache(maxsize=4) +def load_indexes(root: str): + base = Path(root) + return {pos: load_index(base / pair[0]) for pos, pair in FILES.items()} + + +def gloss_for(surface, pos, root): + """First-sense gloss clause. The gloss is screening evidence, not a target.""" + indexes = load_indexes(str(root)) + lemma = surface.replace(" ", "_") + table = indexes.get(pos) or {} + offset = table.get(lemma) + if not offset: + for alt, idx in indexes.items(): + if lemma in idx: + pos, offset = alt, idx[lemma] + break + if not offset: + return pos, "" + data = load_glosses(Path(root) / FILES[pos][1], {offset}) + return pos, data.get(offset, "") + +def screen(surface, source_pos, tokens, gloss): + gloss_l = gloss.lower() + if not (2 <= len(tokens) <= 6) or len(surface) > 80 or " ".join(tokens) != surface: + return "REJECT", "surface_constraint", "exclude" + if any(INITIALISM.fullmatch(t) for t in tokens) or YEAR.search(gloss): + return "REJECT", "proper_person_name", "exclude" + if "organization" in gloss_l: + return "REJECT", "named_organization", "exclude" + if len(tokens) >= 2 and tokens[1] == "of" and tokens[0] in GEO_HEADS: + return "REJECT", "primarily_referential_expression", "exclude" + if len(tokens) >= 2 and tokens[1] == "of" and tokens[0] in INST_HEADS: + return "REJECT", "institutional_title", "exclude" + if "saint" in tokens or "st." in tokens: + return "REJECT", "titled_work_or_designation", "exclude" + if any(POSSESSIVE.fullmatch(t) for t in tokens) and any(t in TITLE_NOUNS for t in tokens): + return "REJECT", "titled_work_or_designation", "exclude" + if tokens and tokens[0] in TITLE_HEADS and tokens[1:2] == ["of"]: + return "REJECT", "degree_or_credential_name", "exclude" + if ( + len(tokens) >= 5 + and any(t in NAME_PARTICLES for t in tokens) + and all(NAME_TOKEN.fullmatch(t) for t in tokens) + ): + return "REJECT", "proper_person_name", "exclude" + if any(t.endswith(TAXON_SUFFIX) for t in tokens) or ( + tokens and tokens[0] in {"family", "genus", "species", "order", "phylum", "tribe"} + ) or "family of" in gloss_l: + return "REJECT", "taxonomy_or_species_label", "exclude" + if ( + source_pos == "noun" + and len(tokens) == 2 + and all(re.fullmatch(r"[a-z]{4,}", t) for t in tokens) + and LATIN_SPECIES.search(tokens[1]) + ): + return "REJECT", "taxonomy_or_species_label", "exclude" + if any(t in PLACES for t in tokens) and len(tokens) <= 3: + return "REJECT", "primarily_referential_expression", "exclude" + if tokens and all(t in NUMBERS for t in tokens): + return "REJECT", "productive_numeric_expression", "exclude" + if "per" in tokens or "unit for measuring" in gloss_l: + return "REJECT", "technical_measurement_expression", "exclude" + if len(tokens) >= 3 and tokens[-1] in PROCEDURE_TAILS: + return "REJECT", "named_technical_procedure", "exclude" + if source_pos == "adj" and len(tokens) == 2 and tokens[0] in DEGREE: + return "REJECT", "unconstrained_free_composition", "exclude" + if len(tokens) == 2 and tokens[0] in {"much", "more", "less"} and tokens[1] == "as": + return "REJECT", "unconstrained_free_composition", "exclude" + gstem = set(stems(gloss)) + sstem = stems(surface) + overlap = 0 if not sstem else len([w for w in sstem if w in gstem]) / len(sstem) + if source_pos == "verb" and len(tokens) == 2 and tokens[1] in PARTICLES and tokens[0][:4] not in gstem: + return "HIGH_VALUE", "strong_phrasal_binding", "score" + for i, tok in enumerate(tokens): + if tok in POSSESSORS: + rest = [w for w in tokens[i + 1:] if len(w) >= 3 and w not in STOP] + if rest and rest[-1][:4] not in gstem: + return "HIGH_VALUE", "lexicalized_variable_slot", "score" + if len(tokens) == 3 and tokens[1] == "and": + sides = [t[:4] for t in (tokens[0], tokens[2]) if len(t) >= 3] + if sides and all(s not in gstem for s in sides): + return "HIGH_VALUE", "fixed_nonliteral_expression", "score" + if len(tokens) == 4 and tokens[1] == "as" and tokens[2] == "a" and tokens[3][:4] not in gstem: + return "HIGH_VALUE", "fixed_nonliteral_expression", "score" + if source_pos in {"verb", "adj", "adv"} and len(tokens) >= 4 and overlap == 0: + return "HIGH_VALUE", "conventionalized_semantic_shift", "score" + return "SECONDARY", "transparent_or_moderate", "score" diff --git a/scripts/shadow/hyperlexical/unbind_screen_v4.py b/scripts/shadow/hyperlexical/unbind_screen_v4.py new file mode 100644 index 00000000..c8de7e1a --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v4.py @@ -0,0 +1,560 @@ +"""Secondary-only coverage patch over a frozen v3 bucket. + +Input is the v3 bucket, the surface, and the first-sense gloss. +A row is inspected only when that bucket is SECONDARY. +Patch A may move SECONDARY to REJECT. Patch B may move SECONDARY to HIGH. +Existing HIGH and REJECT decisions are returned unchanged. +""" + +from __future__ import annotations + +import ast +import re +import unicodedata +from pathlib import Path + +from hyperlexical.unbind_screen_v3 import NUMBERS, PARTICLES, STOP, stems + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v4" +PATCH_A = ( + "multi_token_person_name", + "organization_from_gloss", + "species_or_common_name_referent", + "medical_technical_expression", + "productive_number", +) +PATCH_B = ( + "nonliteral_semantic_shift", + "conventionalized_idiom", + "noncompositional_phrasal_binding", + "fixed_lexicalized_expression", +) +_CANONICAL = { + "HIGH_VALUE": "HIGH", + "HIGH": "HIGH", + "SECONDARY": "SECONDARY", + "REJECT": "REJECT", + "QUARANTINE": "QUARANTINE", +} +_HYPHEN = re.compile(r"(?<=\w)[\u2010\u2011\u2012\u2013\u2014-](?=\w)") +_PUNCT = re.compile(r"[^\w\s']+", re.UNICODE) +_WORD = re.compile(r"[A-Za-z]+") +_DELIMITERS = frozenset({ + "with", "that", "which", "used", "yielding", "having", "resulting", "who", "whose", +}) +_MULTIPLIERS = frozenset({"times", "fold"}) +_CLINICAL = frozenset({ + "impairment", "disease", "disorder", "syndrome", "inflammation", "symptom", + "lesion", "pathology", "hemorrhage", "haemorrhage", "infection", "paralysis", + "fracture", "tumor", "tumour", "carcinoma", "edema", "oedema", "surgery", + "surgical", "clinical", +}) +_COMPARATIVES = frozenset({ + "better", "worse", "greater", "lesser", "more", "less", "higher", "lower", + "bigger", "smaller", "older", "younger", "sooner", "later", "richer", "poorer", + "well", +}) +_INTENSIFIERS = frozenset({ + "bone", "brand", "stone", "rock", "pitch", "crystal", "soaking", "dripping", + "stark", "dirt", "stock", "wide", +}) +_LIGHT_VERBS = frozenset({"give", "take", "have", "make", "get"}) +_DETERMINERS = frozenset({"a", "an", "the"}) +_LIFE = frozenset({"noun.animal", "noun.plant"}) + +SUCCESS_CRITERIA = { + "schema": "hyperlex.unbind_screen_v4_success_criteria.v1", + "high_precision_floor": 1.0, + "reject_precision_floor": 1.0, + "false_high_allowed": 0, + "false_reject_allowed": 0, + "false_secondary_rate_must_be_strictly_below": "13/29", + "gate_b_violations_allowed": 0, + "gate_e_violations_allowed": 0, + "hand_corrections_allowed": 0, + "sample_reuse_allowed": False, + "perfect_accuracy_required": False, + "question": "reduce_secondary_fallthrough_without_false_high_or_false_reject", +} + + +class ScreenV4Error(ValueError): + pass + + +class EmptyLexicon: + def noun_lex(self, lemma: str) -> str | None: + return None + + def has_adjective(self, lemma: str) -> bool: + return False + + +class WordNetLexicon: + """First-sense lex-file names. Used as gloss evidence, not as a phrase list.""" + + def __init__(self, root: str | Path): + base = Path(root) + self._lexnames = _lexnames(base / "lexnames") + noun_offset = _offset_lexnum(base / "data.noun") + self._noun = { + lemma: self._lexnames.get(noun_offset[offset]) + for lemma, offset in _first_offsets(base / "index.noun").items() + if offset in noun_offset + } + self._adj = set(_first_offsets(base / "index.adj")) + + def noun_lex(self, lemma: str) -> str | None: + return self._noun.get(lemma.casefold()) + + def has_adjective(self, lemma: str) -> bool: + return lemma.casefold() in self._adj + + +def normalize_lexical(surface: str) -> str: + """Fold hyphenation and punctuation. Apostrophes stay lexical.""" + text = unicodedata.normalize("NFKC", surface).casefold() + text = text.replace("\u2019", "'").replace("`", "'") + text = _HYPHEN.sub(" ", text) + text = _PUNCT.sub(" ", text) + return re.sub(r"\s+", " ", text).strip() + + +def canonical_bucket(raw: str) -> str: + bucket = _CANONICAL.get(str(raw or "")) + if bucket is None: + raise ScreenV4Error(f"bucket {raw!r} is not a screen bucket") + return bucket + + +def apply_v4(v3_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v4 bucket. HIGH and REJECT are not inspected.""" + bucket = canonical_bucket(v3_bucket) + normalized = normalize_lexical(surface) + if bucket != "SECONDARY": + return _decision(bucket, bucket, None, [], normalized, inspected=False) + tokens = normalized.split() if normalized else [] + gloss_text = gloss or "" + gstem = set(stems(gloss_text)) + patch_a = _patch_a(tokens, source_pos, gloss_text, lexicon) + if patch_a: + return _decision("SECONDARY", "REJECT", patch_a[0], patch_a[1:], normalized, inspected=True) + patch_b = _patch_b(tokens, source_pos, gloss_text, gstem) + if patch_b: + return _decision("SECONDARY", "HIGH", patch_b[0], patch_b[1:], normalized, inspected=True) + return _decision("SECONDARY", "SECONDARY", None, [], normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 113) -> dict: + """Mechanical GATE_A through GATE_E report. This is not a precision score.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = {"secondary_to_high": 0, "secondary_to_reject": 0, "secondary_unchanged": 0} + held = {"high": 0, "reject": 0} + conflicts = 0 + for row in rows: + v3 = canonical_bucket(str(row["v3_bucket"])) + v4 = canonical_bucket(str(row["v4_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if v3 == "HIGH": + held["high"] += 1 + if v4 != "HIGH": + failures.append("HIGH row changed bucket") + if operator == "HIGH" and v4 != "HIGH": + failures.append("v3-correct HIGH is no longer operator-correct") + elif v3 == "REJECT": + held["reject"] += 1 + if v4 != "REJECT": + failures.append("REJECT row changed bucket") + if operator == "REJECT" and v4 != "REJECT": + failures.append("v3-correct REJECT is no longer operator-correct") + elif v3 == "SECONDARY": + if v4 == "SECONDARY": + moves["secondary_unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged SECONDARY carries transition evidence") + elif v4 == "HIGH": + moves["secondary_to_high"] += 1 + _require_transition(primary, supporting, PATCH_B, failures) + elif v4 == "REJECT": + moves["secondary_to_reject"] += 1 + _require_transition(primary, supporting, PATCH_A, failures) + else: + failures.append("SECONDARY moved outside HIGH and REJECT") + else: + failures.append("v3 bucket is outside HIGH, REJECT, and SECONDARY") + if v3 != v4 and operator != v4: + conflicts += 1 + if v3 != v4 and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + verified = not failures + return { + "schema": "hyperlex.unbind_screen_v4_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "high_unchanged": held["high"], + "reject_unchanged": held["reject"], + "moves": moves, + "operator_conflict_on_move": conflicts, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "gate_b_violations": sum(1 for item in failures if "operator-correct" in item or "changed bucket" in item), + "gate_e_violations": sum( + 1 for item in failures + if "changed bucket" in item or "outside HIGH and REJECT" in item or "no primary" in item + or "unchanged SECONDARY" in item or "moved outside" in item + ), + "assertions": { + "A_historical_replay": len(rows) == expected_rows and len(set(surfaces)) == len(surfaces), + "B_outer_bucket_preservation": not any("operator-correct" in item or "changed bucket" in item for item in failures), + "C_patch_a_targeting": not any(item.startswith("SECONDARY to REJECT") for item in failures), + "D_patch_b_targeting": not any(item.startswith("SECONDARY to HIGH") for item in failures), + "E_no_other_movement": not any( + "changed bucket" in item or "outside HIGH" in item or "moved outside" in item + for item in failures + ), + "no_phrase_specific_rule": not phrase_specific_rule_fired, + }, + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + ) + + +def rule_surface_violations(source: str, forbidden: list[str] | tuple[str, ...]) -> list[str]: + """Return phrase-rule violations in scorer source. The probe list is not a rule.""" + found = [phrase for phrase in forbidden if phrase and phrase in source] + tree = ast.parse(source) + for node in ast.walk(tree): + if isinstance(node, ast.Compare): + if not any(isinstance(op, (ast.Eq, ast.NotEq)) for op in node.ops): + continue + candidates = [node.left, *node.comparators] + elif isinstance(node, (ast.List, ast.Tuple, ast.Set)): + candidates = list(node.elts) + elif isinstance(node, ast.Dict): + candidates = [item for item in (*node.keys, *node.values) if item is not None] + else: + continue + for item in candidates: + if ( + isinstance(item, ast.Constant) + and isinstance(item.value, str) + and len(item.value.split()) >= 2 + ): + found.append(item.value) + found.extend(re.findall(r"\b[0-9a-f]{64}\b", source)) + return found + + +def _decision(v3, v4, primary, supporting, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v3_bucket": v3, + "v4_bucket": v4, + "primary_evidence": primary, + "supporting_evidence": list(supporting), + "normalized": normalized, + "inspected": inspected, + } + + +def _require_transition(primary, supporting, allowed: tuple[str, ...], failures: list[str]) -> None: + label = "SECONDARY to HIGH" if allowed is PATCH_B else "SECONDARY to REJECT" + if primary not in allowed: + failures.append(f"{label} lacks patch evidence") + return + if supporting.count(primary) or primary in supporting: + failures.append(f"{label} repeats its primary evidence") + if len([primary]) != 1: + failures.append(f"{label} lacks one primary evidence") + extra = [item for item in supporting if item not in allowed] + if extra: + failures.append(f"{label} supporting evidence is outside the patch") + + +def _patch_a(tokens: list[str], pos: str, gloss: str, lexicon) -> list[str]: + nouns = _span_nouns(gloss, lexicon) + matched = [] + if _person(tokens, pos, nouns): + matched.append("multi_token_person_name") + if _organization(tokens, pos, nouns): + matched.append("organization_from_gloss") + if _species(tokens, pos, nouns): + matched.append("species_or_common_name_referent") + if _medical(tokens, pos, gloss, lexicon): + matched.append("medical_technical_expression") + if _productive_number(tokens): + matched.append("productive_number") + return matched + + +def _patch_b(tokens: list[str], pos: str, gloss: str, gstem: set[str]) -> list[str]: + matched = [] + if _nonliteral(tokens, pos, gstem): + matched.append("nonliteral_semantic_shift") + if _conventionalized(tokens, pos, gstem): + matched.append("conventionalized_idiom") + if _phrasal(tokens, pos, gloss, gstem): + matched.append("noncompositional_phrasal_binding") + if _fixed(tokens, pos, gstem): + matched.append("fixed_lexicalized_expression") + return matched + + +def _person(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun" or not _name_shape(tokens) or not nouns: + return False + head, lex = nouns[0] + return lex == "noun.person" and head not in tokens + + +def _organization(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun" or not nouns: + return False + head, lex = nouns[0] + return lex == "noun.group" and head not in tokens + + +def _species(tokens: list[str], pos: str, nouns: list[tuple[str, str]]) -> bool: + if pos != "noun": + return False + return any(lex in _LIFE and lemma not in tokens for lemma, lex in nouns) + + +def _medical(tokens: list[str], pos: str, gloss: str, lexicon) -> bool: + if pos != "noun": + return False + words = {tok.lower() for tok in _WORD.findall(gloss or "")} + if words.isdisjoint(_CLINICAL): + return False + for lemma, lex in _all_nouns(gloss, lexicon): + if lex == "noun.body" and lemma not in tokens: + return True + return False + + +def _productive_number(tokens: list[str]) -> bool: + if len(tokens) < 2: + return False + body = list(tokens) + if body[0] in {"a", "an"}: + body = body[1:] + if body and body[-1] in _MULTIPLIERS: + body = body[:-1] + if not body: + return False + return all(tok in NUMBERS or tok == "and" for tok in body) and any(tok in NUMBERS for tok in body) + + +def _name_shape(tokens: list[str]) -> bool: + if len(tokens) < 2: + return False + return all( + re.fullmatch(r"[a-z]+", tok) and tok not in STOP and tok not in PARTICLES + for tok in tokens + ) + + +def _span_nouns(gloss: str, lexicon) -> list[tuple[str, str]]: + found = [] + for low in _gloss_words(gloss): + if low in _DELIMITERS and found: + break + item = _noun(low, lexicon) + if item is not None: + found.append(item) + return found + + +def _all_nouns(gloss: str, lexicon) -> list[tuple[str, str]]: + return [item for low in _gloss_words(gloss) if (item := _noun(low, lexicon)) is not None] + + +def _gloss_words(gloss: str) -> list[str]: + return [tok.lower() for tok in _WORD.findall(gloss or "") if tok.lower() not in STOP] + + +def _noun(low: str, lexicon) -> tuple[str, str] | None: + if lexicon.has_adjective(low): + return None + lex = lexicon.noun_lex(low) + if not lex: + return None + return low, lex + + +def _content(tokens: list[str]) -> list[str]: + return [tok for tok in tokens if len(tok) >= 3 and tok not in STOP] + + +def _absent(token: str, gstem: set[str]) -> bool: + return len(token) >= 3 and token[:4] not in gstem + + +def _nonliteral(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return any(( + _verb_the(tokens, pos, gstem), + _verb_determiner(tokens, pos, gstem), + _verb_pivot_noun(tokens, pos, gstem), + _in_the_head(tokens, pos, gstem), + )) + + +def _conventionalized(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return any(( + _like_vehicle(tokens, pos, gstem), + _as_frame(tokens, pos, gstem), + _for_all(tokens, pos, gstem), + _light_verb(tokens, pos, gstem), + )) + + +def _verb_the(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) < 3 or tokens[1] != "the": + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _verb_determiner(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) != 3 or tokens[1] not in {"a", "an"}: + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _verb_pivot_noun(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or len(tokens) != 4: + return False + if tokens[1] not in {"on", "in"} or tokens[2] not in {"a", "an"}: + return False + return _absent(tokens[3], gstem) + + +def _in_the_head(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv", "noun"} or len(tokens) != 4: + return False + if tokens[0] != "in" or tokens[1] != "the": + return False + content = _content(tokens) + present = [tok for tok in content if tok[:4] in gstem] + return bool(present) and _absent(tokens[-1], gstem) + + +def _like_vehicle(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv"} or len(tokens) != 3: + return False + if tokens[0] != "like" or tokens[1] not in {"a", "an"}: + return False + return _absent(tokens[2], gstem) + + +def _as_frame(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "adv"} or len(tokens) != 3 or tokens[1] != "as": + return False + return _absent(tokens[2], gstem) + + +def _for_all(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "adv" or len(tokens) < 4 or tokens[:2] != ["for", "all"]: + return False + content = _content(tokens) + present = [tok for tok in content if tok[:4] in gstem] + return bool(present) and _absent(tokens[-1], gstem) + + +def _light_verb(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "verb" or not tokens or tokens[0] not in _LIGHT_VERBS: + return False + if not any(tok in _DETERMINERS for tok in tokens) or not gstem: + return False + surface = {tok[:4] for tok in _content(tokens)} + if not gstem <= surface: + return False + return _absent(tokens[0], gstem) or len(tokens[0]) < 3 + + +def _phrasal(tokens: list[str], pos: str, gloss: str, gstem: set[str]) -> bool: + if pos not in {"verb", "adj"} or not tokens or tokens[-1] not in PARTICLES: + return False + if tokens[0] in _COMPARATIVES: + return False + gloss_words = {tok.lower() for tok in _WORD.findall(gloss or "")} + if tokens[-1] in gloss_words: + return False + return _absent(tokens[0], gstem) or len(tokens[0]) < 3 + + +def _fixed(tokens: list[str], pos: str, gstem: set[str]) -> bool: + return _intensifier(tokens, pos, gstem) or _for_result(tokens, pos, gstem) + + +def _intensifier(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos != "adj" or len(tokens) != 2 or tokens[0] not in _INTENSIFIERS: + return False + return _absent(tokens[1], gstem) + + +def _for_result(tokens: list[str], pos: str, gstem: set[str]) -> bool: + if pos not in {"adj", "verb"} or not (2 <= len(tokens) <= 4) or "for" not in tokens: + return False + content = _content(tokens) + return bool(content) and all(_absent(tok, gstem) for tok in content) + + +def _lexnames(path: Path) -> dict[int, str]: + names = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line.strip(): + continue + number, name, *_rest = line.split() + names[int(number)] = name + return names + + +def _offset_lexnum(path: Path) -> dict[str, int]: + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + offset, lexnum, *_rest = line.split(" ", 2) + found[offset] = int(lexnum) + return found + + +def _first_offsets(path: Path) -> dict[str, str]: + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + parts = line.split() + synset_cnt = int(parts[2]) + pointer_cnt = int(parts[3]) + rest = parts[4 + pointer_cnt:] + offsets = rest[2:2 + synset_cnt] + if not offsets: + continue + found[parts[0].casefold()] = offsets[0] + return found diff --git a/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py new file mode 100644 index 00000000..e080b47a --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v4_measure.py @@ -0,0 +1,415 @@ +"""Replay and measurement driver for the v4 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import ( + RULE_VERSION, + SUCCESS_CRITERIA, + WordNetLexicon, + apply_v4, + assess, + canonical_bucket, + measurement_allowed, + normalize_lexical, + rule_surface_violations, +) + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +TRAIN = Path("/home/morpheus/hlx-private/d1-spark-tree-20260924T213846Z/morph78-train-export.jsonl") +FIT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/development_fit.json" +EVAL = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V3-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-001/unbind_screen_v4" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_SAMPLE = "8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e" +EXPECTED_LABELS = "4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5" +EXPECTED_ACCEPTANCE = "ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", +}) +SCORER_FILES = ( + Path(__file__).with_name("unbind_screen_v3.py"), + Path(__file__).with_name("unbind_screen_v4.py"), +) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + found = [] + for path in SCORER_FILES: + found.extend(rule_surface_violations(path.read_text(encoding="utf-8"), PROBE_SURFACES)) + return found + + +def load_replay_rows() -> list[dict]: + fit = json.loads(FIT.read_text(encoding="utf-8")) + rows = [] + for raw in fit["rows"]: + rows.append({ + "surface": raw["text"], + "pos": raw["source_pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["predicted"], + "operator_bucket": raw["operator"], + "split": raw.get("split") or "development", + }) + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "operator/heldout-001.labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "predictions/heldout-001.predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + reviews = { + json.loads(line)["row_id"]: json.loads(line) + for line in (EVAL / "operator/heldout-001.review.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(labels) != set(predictions) or set(labels) != set(reviews): + raise SystemExit("held-out label, prediction, and review identities differ") + for row_id, review in reviews.items(): + rows.append({ + "surface": review["surface"], + "pos": review["pos"], + "gloss": review.get("gloss") or "", + "v3_bucket": predictions[row_id]["bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "held_out_29", + }) + if len(rows) != 113: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 113") + if len({row["surface"] for row in rows}) != 113: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed before replay") + sample_sha = file_sha256( + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/held_out_sample.jsonl" + ) + label_sha = file_sha256(EVAL / "operator/heldout-001.labels.jsonl") + if sample_sha != EXPECTED_SAMPLE or label_sha != EXPECTED_LABELS: + raise SystemExit("v3 sample or label hash changed") + lexicon = WordNetLexicon(WORDNET) + predictions = [] + assessed_rows = [] + for raw in load_replay_rows(): + decision = apply_v4(raw["v3_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + row_id = normalized_text_sha256(raw["surface"]) + record = { + "schema": "hyperlex.unbind_screen_v4_replay_row.v1", + "row_id": row_id, + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + **decision, + } + predictions.append(record) + assessed_rows.append(record) + violations = scorer_violations() + report = assess(assessed_rows, phrase_specific_rule_fired=bool(violations), expected_rows=113) + report["phrase_violations"] = violations + report["acceptance_sha256"] = EXPECTED_ACCEPTANCE + report["events_sha256"] = events_sha256() + report["v3_sample_sha256"] = sample_sha + report["v3_label_sha256"] = label_sha + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v3_bucket"] != row["v4_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + prediction_sha = _dump_jsonl(out_dir / "replay_113_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "replay_113_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "replay_113_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "implementation_receipt.json", receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + receipt_path = out_dir / "implementation_receipt.json" + report = json.loads((out_dir / "replay_113_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report): + raise SystemExit("measurement draw refused: regression is not verified") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed before the draw") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", dict(SUCCESS_CRITERIA)) + reviewed = load_replay_rows() + blocked_hash = {normalized_text_sha256(row["surface"]) for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + lexicon = WordNetLexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + decision = apply_v4(v3_bucket, surface, gloss, pos, lexicon) + row_id = normalized_text_sha256(surface) + provenance = { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 113, + } + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V4-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": provenance, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v4_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V4-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v4_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 113, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v4_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads(receipt_path.read_text(encoding="utf-8")) + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v4_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(receipt_path, receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + return freeze + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + source_sha = { + path.name: file_sha256(path) + for path in SCORER_FILES + } + return { + "schema": "hyperlex.unbind_screen_v4_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "ENCODED", + "regression": report["regression"], + "measurement_eligible": report["measurement_eligible"], + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v3": "proposed_coverage_extension_only", + "acceptance_sha256": EXPECTED_ACCEPTANCE, + "source_sha256": source_sha, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "moves": report["moves"], + "operator_conflict_on_move": report["operator_conflict_on_move"], + "assertions": report["assertions"], + "v3_sample_sha256": report["v3_sample_sha256"], + "v3_label_sha256": report["v3_label_sha256"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + + +def _stratified_sample( + blocked_hash: set[str], blocked_norm: set[str], *, per_cell: int +) -> tuple[list[dict], list[dict]]: + from hyperlexical.clean_unbind import ( + WORDNET_LICENSE, + gate_rows, + load_jsonl, + read_wordnet_index, + rows_from_wordnet_atoms, + ) + from hyperlexical.identity_ledger import IdentityLedger + + atoms = read_wordnet_index(WORDNET) + rows = rows_from_wordnet_atoms(atoms, license=WORDNET_LICENSE) + train_rows = load_jsonl(TRAIN) + ledger = IdentityLedger.load(LEDGER) + admissible, _rejections, _account = gate_rows( + rows, + train_rows=train_rows, + ledger=ledger, + require_settlement=False, + ) + best: dict[str, dict] = {} + for row in admissible: + if row.get("role_scheme") != "positional": + continue + surface = str(row.get("text") or "") + identity = normalize_lexical(surface) + digest = normalized_text_sha256(surface) + if digest in blocked_hash or identity in blocked_norm: + continue + current = best.get(identity) + if current is None or digest < normalized_text_sha256(str(current.get("text") or "")): + best[identity] = row + cells: dict[tuple[str, int], list[dict]] = {} + for row in best.values(): + fillers = [str(tok) for tok in row.get("fillers") or []] + cells.setdefault((str(row.get("source_pos") or ""), len(fillers)), []).append(row) + picked = [] + strata = [] + for key in sorted(cells): + group = cells[key] + group.sort(key=lambda row: normalized_text_sha256(str(row.get("text") or ""))) + taken = group[:per_cell] + picked.extend(taken) + strata.append({ + "source_pos": key[0], + "token_count": key[1], + "available": len(group), + "taken": len(taken), + }) + if not picked: + raise SystemExit("measurement sample is empty") + return picked, strata + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v4 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw")) + args = parser.parse_args() + if args.command == "replay": + receipt = replay() + else: + receipt = draw_measurement() + summary = { + key: receipt[key] + for key in receipt + if key in { + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "operator_conflict_on_move", + "failures", "phrase_specific_rule_fired", "rows", "sample_sha256", + "precision", "v4_application_count", "events_sha256", "measurement_rows", + "measurement_sample_sha256", + } + } + print(json.dumps(summary, sort_keys=True, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v5.py b/scripts/shadow/hyperlexical/unbind_screen_v5.py new file mode 100644 index 00000000..06790b04 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v5.py @@ -0,0 +1,427 @@ +"""Secondary-only wrapper over a frozen v4 bucket. + +HIGH and REJECT pass through uninspected. A SECONDARY row is inspected once. +Referential or terminological dominance moves it to REJECT. A conventionalized +surface whose gloss is not recoverable from the ordinary first senses of its +constituents moves it to HIGH. A lexicalized but compositional surface stays +SECONDARY. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path + +from hyperlexical.unbind_screen_v3 import STOP +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v5" +REJECT_EVIDENCE = "referential_terminological_dominance" +HIGH_EVIDENCE = "lexicalized_noncompositional" +_FUNCTION = set(STOP) | {"one's", "jr", "jr's"} +_META = frozenset({"used", "introducing"}) +_LIFE = frozenset({"noun.plant", "noun.animal"}) +_WORD = re.compile(r"[A-Za-z']+") +_TOKEN = re.compile(r"[a-z0-9']+") +_POS = ("noun", "verb", "adj", "adv") + + +@dataclass(frozen=True) +class Sense: + token: str + pos: str + lex: str + gloss: str + + +@dataclass(frozen=True) +class Entry: + pos: str + lemma: str + lex: str + lemmas: tuple[str, ...] + hypernyms: tuple[tuple[str, ...], ...] + + +class EmptyLexicon: + def entry(self, surface: str) -> Entry | None: + return None + + def senses(self, token: str) -> tuple[Sense, ...]: + return () + + +class WordNetLexicon: + """First-sense synsets. Gloss evidence, not a phrase list.""" + + def __init__(self, root: str | Path): + base = Path(root) + lexnames = _lexnames(base / "lexnames") + self._data = {pos: _load_data(base / f"data.{pos}") for pos in _POS} + self._index = {pos: _load_index(base / f"index.{pos}") for pos in _POS} + self._lexnames = lexnames + self._entries: dict[str, tuple[str, str, str]] = {} + for pos in _POS: + for key, offsets in self._index[pos].items(): + if not offsets: + continue + folded = key.casefold().replace("-", "_") + self._entries.setdefault(folded, (pos, key, offsets[0])) + + def entry(self, surface: str) -> Entry | None: + folded = surface.casefold().replace(" ", "_").replace("-", "_") + found = self._entries.get(folded) + if found is None: + return None + pos, key, offset = found + built = self._entry(pos, offset) + return Entry(built.pos, key, built.lex, built.lemmas, built.hypernyms) + + def senses(self, token: str) -> tuple[Sense, ...]: + found = [] + for pos in _POS: + offsets = self._index[pos].get(token) + if not offsets: + continue + entry = self._entry(pos, offsets[0]) + gloss = self._gloss(pos, offsets[0]) + found.append(Sense(token, pos, entry.lex, gloss)) + return tuple(found) + + def _entry(self, pos: str, offset: str) -> Entry: + lemmas, lex, hypers = _parse(self._data[pos][offset], self._lexnames, self._data["noun"]) + return Entry(pos, lemmas[0] if lemmas else "", lex, tuple(lemmas), tuple(tuple(item) for item in hypers)) + + def _gloss(self, pos: str, offset: str) -> str: + line = self._data[pos][offset] + return line.split("|", 1)[1].split(";", 1)[0].strip() + + +def apply_v5(v4_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v5 bucket. HIGH and REJECT are not inspected.""" + del source_pos + bucket = canonical_bucket(v4_bucket) + normalized = normalize_lexical(surface) + if bucket != "SECONDARY": + return _decision(bucket, bucket, None, normalized, inspected=False) + evidence = _inspect(surface or "", gloss or "", lexicon) + if evidence == REJECT_EVIDENCE: + return _decision(bucket, "REJECT", evidence, normalized, inspected=True) + if evidence == HIGH_EVIDENCE: + return _decision(bucket, "HIGH", evidence, normalized, inspected=True) + return _decision(bucket, "SECONDARY", None, normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 141) -> dict: + """Replay gate. This is not a precision score.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = {"secondary_to_high": 0, "secondary_to_reject": 0, "secondary_unchanged": 0} + held = {"high": 0, "reject": 0} + conflicts = 0 + for row in rows: + prior = canonical_bucket(str(row["v4_bucket"])) + nxt = canonical_bucket(str(row["v5_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == "HIGH": + held["high"] += 1 + if nxt != "HIGH": + failures.append("HIGH row changed bucket") + if operator == "HIGH" and nxt != "HIGH": + failures.append("previously correct HIGH is no longer correct") + elif prior == "REJECT": + held["reject"] += 1 + if nxt != "REJECT": + failures.append("REJECT row changed bucket") + if operator == "REJECT" and nxt != "REJECT": + failures.append("previously correct REJECT is no longer correct") + elif prior == "SECONDARY": + if nxt == "SECONDARY": + moves["secondary_unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged SECONDARY carries transition evidence") + elif nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + else: + failures.append("SECONDARY moved outside HIGH and REJECT") + else: + failures.append("prior bucket is outside HIGH, REJECT, and SECONDARY") + if prior != nxt and operator != nxt: + conflicts += 1 + failures.append("operator conflict on a move") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + verified = not failures + return { + "schema": "hyperlex.unbind_screen_v5_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "high_unchanged": held["high"], + "reject_unchanged": held["reject"], + "moves": moves, + "operator_conflict_on_move": conflicts, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + and report.get("operator_conflict_on_move") == 0 + ) + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v4_bucket": prior, + "v5_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _inspect(surface: str, gloss: str, lexicon) -> str | None: + found = lexicon.entry(surface) + if found is None: + return None + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + ordinary, missing, senses = _ordinary(content, lexicon) + if _designates(found, stems, content): + return REJECT_EVIDENCE + if _noncompositional(found, gloss, content, stems, ordinary, missing, senses): + return HIGH_EVIDENCE + return None + + +def _designates(found: Entry, stems: set[str], content: list[str]) -> bool: + if any(char.isupper() for char in found.lemma): + return True + if found.lex == "adj.pert": + return True + if found.lex in _LIFE and _exocentric_life(found.hypernyms, stems): + return True + if _exocentric_category(found.hypernyms, stems) and not _ordinary_synonym(found.lemmas, content): + return True + return False + + +def _noncompositional(found: Entry, gloss, content, stems, ordinary, missing, senses) -> bool: + if _blocked(gloss, ordinary, missing, content): + return False + if _orthographic(content) or _body_clash(found, gloss, content, senses): + return True + if _verb_shift(found, content, senses): + return True + return ( + found.pos == "adv" + and _unrelated_paraphrase(found.lemmas, content, stems) + and not _morphological(found.lemmas, content, stems) + ) + + +def _blocked(gloss: str, ordinary: set[str], missing: list[str], content: list[str]) -> bool: + words = _WORD.findall(gloss or "") + if words and words[0].casefold() in _META: + return True + if missing or len(content) != len(set(content)): + return True + return bool(_stems(gloss) & ordinary) + + +def _ordinary(content: list[str], lexicon): + ordinary: set[str] = set() + missing: list[str] = [] + senses: list[Sense] = [] + for token in content: + if len(token) < 3: + continue + found = tuple(lexicon.senses(token)) + if not found: + missing.append(token) + continue + for sense in found: + ordinary |= _stems(sense.gloss) + stemmed = _stem(token) + if stemmed: + ordinary.add(stemmed) + senses.append(sense) + return ordinary, missing, senses + + +def _exocentric_life(hypernyms, stems: set[str]) -> bool: + if not hypernyms: + return False + for lemmas in hypernyms: + for lemma in lemmas: + if any(_stem(part) in stems for part in re.split(r"[_-]", lemma)): + return False + return True + + +def _exocentric_category(hypernyms, stems: set[str]) -> bool: + for lemmas in hypernyms: + for lemma in lemmas: + if "_" not in lemma: + continue + parts = [part for part in (_stem(item) for item in re.split(r"[_-]", lemma)) if part] + if parts and not any(part in stems for part in parts): + return True + return False + + +def _ordinary_synonym(lemmas, content: list[str]) -> bool: + owned = set(content) + for lemma in _single_words(lemmas): + if _initialism(lemma): + continue + low = re.sub(r"[^a-z]", "", lemma.casefold()) + if low and low not in owned: + return True + return False + + +def _body_clash(found: Entry, gloss: str, content: list[str], senses: list[Sense]) -> bool: + if found.lex == "noun.body": + return False + if not any(sense.lex == "noun.body" for sense in senses): + return False + gloss_l = (gloss or "").casefold() + return not any(token in gloss_l for token in content) + + +def _verb_shift(found: Entry, content: list[str], senses: list[Sense]) -> bool: + if found.pos != "verb" or not content or found.lex.endswith(".all"): + return False + head = content[0] + head_lex = next((sense.lex for sense in senses if sense.token == head and sense.pos == "verb"), None) + if not head_lex or head_lex.endswith(".all") or head_lex == found.lex: + return False + return True + + +def _unrelated_paraphrase(lemmas, content, stems) -> bool: + return any(not _related(lemma, content, stems) for lemma in _single_words(lemmas)) + + +def _morphological(lemmas, content, stems) -> bool: + return any(_related(lemma, content, stems) for lemma in _single_words(lemmas)) + + +def _related(lemma: str, content: list[str], stems: set[str]) -> bool: + low = lemma.casefold() + stemmed = _stem(low) + if stemmed and stemmed in stems: + return True + return any(len(token) >= 4 and (token in low or low in token) for token in content) + + +def _orthographic(content: list[str]) -> bool: + return any(len(token) == 1 and token.isalpha() for token in content) + + +def _single_words(lemmas) -> list[str]: + return [lemma for lemma in lemmas if "_" not in lemma and "-" not in lemma and "(" not in lemma] + + +def _initialism(lemma: str) -> bool: + letters = re.sub(r"[^A-Za-z]", "", lemma) + return bool(letters) and letters.isupper() + + +def _content(surface: str) -> list[str]: + text = surface.casefold().replace("-", " ") + return [token for token in _TOKEN.findall(text) if token not in _FUNCTION] + + +def _stem(word: str) -> str: + token = word.casefold().strip("'") + if len(token) < 3 or token in _FUNCTION: + return "" + return token[:4] + + +def _stems(text: str) -> set[str]: + return {item for item in (_stem(word) for word in _WORD.findall(text or "")) if item} + + +def _lexnames(path: Path) -> dict[int, str]: + names = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line.strip(): + continue + number, name, *_rest = line.split() + names[int(number)] = name + return names + + +def _load_index(path: Path) -> dict[str, list[str]]: + found = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line or line[0] == " ": + continue + parts = line.split() + synset_cnt = int(parts[2]) + pointer_cnt = int(parts[3]) + rest = parts[4 + pointer_cnt:] + found[parts[0]] = rest[2:2 + synset_cnt] + return found + + +def _load_data(path: Path) -> dict[str, str]: + found = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line or line[0] == " ": + continue + found[line.split(" ", 1)[0]] = line + return found + + +def _parse(line: str, lexnames: dict[int, str], noun_data: dict[str, str]): + parts = line.split() + lex = lexnames[int(parts[1])] + count = int(parts[3], 16) + lemmas = [] + index = 4 + for _ in range(count): + lemmas.append(parts[index]) + index += 2 + pointer_cnt = int(parts[index]) + index += 1 + hypers = [] + for _ in range(pointer_cnt): + symbol, target, target_pos = parts[index], parts[index + 1], parts[index + 2] + index += 4 + if symbol in {"@", "@i"} and target_pos == "n" and target in noun_data: + hyper, _lex, _nested = _parse(noun_data[target], lexnames, noun_data) + hypers.append(hyper) + return lemmas, lex, hypers diff --git a/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py new file mode 100644 index 00000000..625b6a52 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v5_measure.py @@ -0,0 +1,425 @@ +"""Replay and measurement driver for the v5 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import ( + RULE_VERSION, + WordNetLexicon, + apply_v5, + assess, + measurement_allowed, +) + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +EVAL3 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V3-001" +V4 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-001/unbind_screen_v4" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-001/unbind_screen_v5" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V4_SAMPLE = "dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b" +EXPECTED_V4_PREDICTIONS = "a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f" +EXPECTED_V4_LABELS = "023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463" +EXPECTED_V4_ACCEPTANCE = "ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b" +EXPECTED_V3_SAMPLE = "8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e" +EXPECTED_V3_LABELS = "4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", "v5", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", "v5_bucket", +}) +SCORER = Path(__file__).with_name("unbind_screen_v5.py") + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + return rule_surface_violations(SCORER.read_text(encoding="utf-8"), PROBE_SURFACES) + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V4 / "replay_113_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_113", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V4 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v4 measurement identities differ") + for row_id, blind in sample.items(): + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": predictions[row_id]["v3_bucket"], + "v4_bucket": predictions[row_id]["v4_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v4_measurement_28", + "row_id": row_id, + }) + if len(rows) != 141 or len({row["surface"] for row in rows}) != 141: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 141 unique") + if len({row["row_id"] for row in rows}) != 141: + raise SystemExit("reviewed row ids are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("replay refused: the v5 measurement sample is already frozen") + _require_frozen_parents() + lexicon = WordNetLexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v5(raw["v4_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + record = { + "schema": "hyperlex.unbind_screen_v5_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + **decision, + } + predictions.append(record) + violations = scorer_violations() + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=141) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["events_sha256"] = events_sha256() + report["v4_sample_sha256"] = EXPECTED_V4_SAMPLE + report["v4_prediction_sha256"] = EXPECTED_V4_PREDICTIONS + report["v4_label_sha256"] = EXPECTED_V4_LABELS + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v4_bucket"] != row["v5_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v5_replay_141_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v5_replay_141_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v5_replay_141_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v5_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + report = json.loads((out_dir / "v5_replay_141_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report): + raise SystemExit("measurement draw refused: regression is not verified") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("false_secondary_rate_must_be_strictly_below") != "12/28": + raise SystemExit("acceptance bar is not the frozen false-secondary rate") + if criteria.get("accuracy_gain_required") is not False or criteria.get("high_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + reviewed = load_replay_rows() + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + lexicon = WordNetLexicon(WORDNET) + v4_lexicon = V4Lexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, lexicon) + row_id = normalized_text_sha256(surface) + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V5-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 141, + }, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v5_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V5-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + if {row["row_id"] for row in blind} & blocked_hash: + raise SystemExit("measurement sample reuses a reviewed row id") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v5_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 141, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v5_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v5_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v5_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v5_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted") + return freeze + + +def _require_frozen_parents() -> None: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed") + checks = { + V4 / "measurement_sample.jsonl": EXPECTED_V4_SAMPLE, + V4 / "measurement_predictions.jsonl": EXPECTED_V4_PREDICTIONS, + V4 / "measurement_labels.jsonl": EXPECTED_V4_LABELS, + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V4-HYPOTHESIS-001/ACCEPTANCE.json": EXPECTED_V4_ACCEPTANCE, + LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-2026-09-27-004/held_out_sample.jsonl": EXPECTED_V3_SAMPLE, + EVAL3 / "operator/heldout-001.labels.jsonl": EXPECTED_V3_LABELS, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path.name}") + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + return { + "schema": "hyperlex.unbind_screen_v5_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if report["measurement_eligible"] else "ENCODED", + "regression": report["regression"], + "measurement_eligible": report["measurement_eligible"], + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v4": "pure_wrapper_over_a_frozen_v4_bucket", + "acceptance_sha256": report["acceptance_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "moves": report["moves"], + "operator_conflict_on_move": report["operator_conflict_on_move"], + "high_unchanged": report["high_unchanged"], + "reject_unchanged": report["reject_unchanged"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + draft = file_sha256(HYP / "HYPOTHESIS.draft.json") + acceptance = file_sha256(HYP / "ACCEPTANCE.json") + if hypothesis.get("draft_hypothesis_sha256") != draft or hypothesis.get("acceptance_sha256") != acceptance: + raise SystemExit("v5 hypothesis no longer points at the frozen draft and acceptance") + hypothesis["encoded"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = receipt["measurement_eligible"] + hypothesis["applied"] = bool(receipt.get("applied_to_measurement")) + hypothesis["applied_to_measurement"] = bool(receipt.get("applied_to_measurement")) + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["operator_conflict_on_move"] = receipt["operator_conflict_on_move"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + if receipt.get("measurement_sample_sha256"): + hypothesis["measurement_sample_sha256"] = receipt["measurement_sample_sha256"] + hypothesis["measurement_rows"] = receipt["measurement_rows"] + hypothesis["measurement_state"] = receipt["measurement_state"] + hypothesis["precision"] = "NOT_COMPUTABLE" + text = json.dumps(hypothesis, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v5 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw")) + args = parser.parse_args() + receipt = replay() if args.command == "replay" else draw_measurement() + summary = { + key: receipt[key] + for key in ( + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "operator_conflict_on_move", + "failures", "phrase_specific_rule_fired", "rows", "sample_sha256", + "precision", "v5_application_count", "events_sha256", "measurement_rows", + "measurement_sample_sha256", "high_unchanged", "reject_unchanged", + ) + if key in receipt + } + print(json.dumps(summary, sort_keys=True, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v6.py b/scripts/shadow/hyperlexical/unbind_screen_v6.py new file mode 100644 index 00000000..5d908e91 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v6.py @@ -0,0 +1,262 @@ +"""Challengeable outer buckets over a frozen v5 decision. + +A frozen v5 bucket is provisional. High may fall to secondary when the gloss +is recoverable from ordinary constituent senses and ordinary syntax. Reject +may fall to secondary when the surface is a lexical state or relation rather +than a designation. A demotion stops for that application. A provisional +secondary row may still move by the two v5 evidences. High and reject do not +swap. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical +from hyperlexical.unbind_screen_v5 import ( + HIGH_EVIDENCE, + REJECT_EVIDENCE, + Entry, + _body_clash, + _content, + _exocentric_category, + _exocentric_life, + _initialism, + _ordinary, + _ordinary_synonym, + _orthographic, + _related, + _single_words, + _stem, + _stems, + _verb_shift, + _LIFE, +) + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v6" +COMPOSITIONAL_EVIDENCE = "compositional_recoverability" +NONREFERENTIAL_EVIDENCE = "nonreferential_lexical_use" +_RELATIONAL = re.compile(r"^(of or relating to|relating to|related to)\b", re.IGNORECASE) + + +def apply_v6(v5_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v6 bucket. A demoted row is not reconsidered.""" + del source_pos + bucket = canonical_bucket(v5_bucket) + normalized = normalize_lexical(surface) + if bucket == "HIGH": + if _compositional(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", COMPOSITIONAL_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "HIGH", None, normalized, inspected=True) + if bucket == "REJECT": + if _nonreferential(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", NONREFERENTIAL_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "REJECT", None, normalized, inspected=True) + if bucket != "SECONDARY": + raise ValueError(f"provisional bucket {bucket} is outside HIGH, REJECT, and SECONDARY") + evidence = _secondary_evidence(surface or "", gloss or "", lexicon) + if evidence == REJECT_EVIDENCE: + return _decision(bucket, "REJECT", evidence, normalized, inspected=True) + if evidence == HIGH_EVIDENCE: + return _decision(bucket, "HIGH", evidence, normalized, inspected=True) + return _decision(bucket, "SECONDARY", None, normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 169) -> dict: + """Replay gate. Previously correct rows must stay correct. Buckets may move.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = { + "high_to_secondary": 0, + "reject_to_secondary": 0, + "secondary_to_high": 0, + "secondary_to_reject": 0, + "unchanged": 0, + } + previously_correct = 0 + previously_correct_lost = 0 + direct_swaps = 0 + for row in rows: + prior = canonical_bucket(str(row["v5_bucket"])) + nxt = canonical_bucket(str(row["v6_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == operator: + previously_correct += 1 + if nxt != operator: + previously_correct_lost += 1 + failures.append("previously correct row is no longer correct") + if prior == nxt: + moves["unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged row carries transition evidence") + elif prior == "HIGH" and nxt == "SECONDARY": + moves["high_to_secondary"] += 1 + if primary != COMPOSITIONAL_EVIDENCE or supporting: + failures.append("HIGH to SECONDARY lacks compositional_recoverability") + if nxt != operator: + failures.append("outer reversal does not land on the operator bucket") + elif prior == "REJECT" and nxt == "SECONDARY": + moves["reject_to_secondary"] += 1 + if primary != NONREFERENTIAL_EVIDENCE or supporting: + failures.append("REJECT to SECONDARY lacks nonreferential_lexical_use") + if nxt != operator: + failures.append("outer reversal does not land on the operator bucket") + elif prior == "SECONDARY" and nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif prior == "SECONDARY" and nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + elif {prior, nxt} == {"HIGH", "REJECT"}: + direct_swaps += 1 + failures.append("HIGH and REJECT swapped directly") + else: + failures.append("row moved outside the four legal transitions") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + assertions = { + "replay_count": len(rows) == expected_rows, + "previously_correct_remain_correct": previously_correct_lost == 0, + "high_to_secondary_evidence": "HIGH to SECONDARY lacks compositional_recoverability" not in failures, + "reject_to_secondary_evidence": "REJECT to SECONDARY lacks nonreferential_lexical_use" not in failures, + "secondary_to_high_evidence": "SECONDARY to HIGH lacks lexicalized_noncompositional" not in failures, + "secondary_to_reject_evidence": "SECONDARY to REJECT lacks referential_terminological_dominance" not in failures, + "no_direct_outer_swap": direct_swaps == 0, + "no_phrase_specific_rules": not phrase_specific_rule_fired, + } + verified = not failures and all(assertions.values()) + return { + "schema": "hyperlex.unbind_screen_v6_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "assertions": assertions, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "previously_correct": previously_correct, + "previously_correct_lost": previously_correct_lost, + "direct_swaps": direct_swaps, + "moves": moves, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + and report.get("previously_correct_lost") == 0 + and report.get("direct_swaps") == 0 + and all((report.get("assertions") or {}).values()) + ) + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v5_bucket": prior, + "v6_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _secondary_evidence(surface: str, gloss: str, lexicon) -> str | None: + """The frozen v5 secondary tests. Not applied to a row demoted in this pass.""" + from hyperlexical.unbind_screen_v5 import _inspect + + return _inspect(surface, gloss, lexicon) + + +def _compositional(surface: str, gloss: str, lexicon) -> bool: + """Recoverable from ordinary senses and ordinary syntax, without an idiomatic mapping.""" + found = lexicon.entry(surface) + if found is None: + return False + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + if _idiomatic_mapping(found, content, stems): + return False + ordinary, missing, senses = _ordinary(content, lexicon) + if missing or _orthographic(content) or _body_clash(found, gloss, content, senses): + return False + if _verb_shift(found, content, senses): + return False + gloss_stems = _stems(gloss) + if not gloss_stems or not gloss_stems <= ordinary: + return False + return _contributors(content, gloss_stems, lexicon) >= 2 + + +def _nonreferential(surface: str, gloss: str, lexicon) -> bool: + """A pertainym used as a state or relation, not as a designation.""" + found = lexicon.entry(surface) + if found is None or found.lex != "adj.pert": + return False + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + if _hard_designation(found, stems, content): + return False + text = (gloss or "").strip() + if not text or _RELATIONAL.match(text): + return False + ordinary, missing, _senses = _ordinary(content, lexicon) + if missing: + return False + gloss_stems = _stems(text) + sense_stems = ordinary - stems + return bool(gloss_stems & sense_stems) + + +def _hard_designation(found: Entry, stems: set[str], content: list[str]) -> bool: + if any(char.isupper() for char in found.lemma): + return True + if found.lex in _LIFE and _exocentric_life(found.hypernyms, stems): + return True + if _exocentric_category(found.hypernyms, stems) and not _ordinary_synonym(found.lemmas, content): + return True + return False + + +def _idiomatic_mapping(found: Entry, content: list[str], stems: set[str]) -> bool: + for lemma in _single_words(found.lemmas): + if _initialism(lemma): + continue + if not _related(lemma, content, stems): + return True + return False + + +def _contributors(content: list[str], gloss_stems: set[str], lexicon) -> int: + count = 0 + for token in content: + if len(token) < 3: + continue + token_stems = set() + for sense in lexicon.senses(token): + token_stems |= _stems(sense.gloss) + stemmed = _stem(token) + if stemmed: + token_stems.add(stemmed) + if gloss_stems & token_stems: + count += 1 + return count diff --git a/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py new file mode 100644 index 00000000..dee8d2f6 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v6_measure.py @@ -0,0 +1,503 @@ +"""Replay and measurement driver for the v6 unbind screen. + +This module does not admit, settle, or append the ledger. +The forbidden-surface tuple is a regression probe. The scorer does not import it. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import WordNetLexicon as V5Lexicon +from hyperlexical.unbind_screen_v5 import apply_v5 +from hyperlexical.unbind_screen_v6 import RULE_VERSION, apply_v6, assess, measurement_allowed + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +V5 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-001/unbind_screen_v5" +V5H = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V5-HYPOTHESIS-001" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-001/unbind_screen_v6" +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V5_SAMPLE = "a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359" +EXPECTED_V5_PREDICTIONS = "51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f" +EXPECTED_V5_LABELS = "b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f" +EXPECTED_V5_ACCEPTANCE = "752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a" +EXPECTED_V6_ACCEPTANCE = "68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f" +EXPECTED_V6_DRAFT = "942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5" +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", + "keep out", + "on the job", + "take orders", + "from nowhere", + "all told", + "rolled into one", + "handle with kid gloves", + "fleet ballistic missile submarine", + "rapid eye movement sleep", + "law of conservation of energy", + "coronoid process of the mandible", +) +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "forecast", "diagnostic", "v4", "v3", "v5", "v6", + "primary_evidence", "supporting_evidence", "evidence", "v3_bucket", "v4_bucket", + "v5_bucket", "v6_bucket", +}) +SCORER = Path(__file__).with_name("unbind_screen_v6.py") + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def scorer_violations() -> list[str]: + return rule_surface_violations(SCORER.read_text(encoding="utf-8"), PROBE_SURFACES) + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V5 / "v5_replay_141_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "v5_bucket": raw["v5_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_141", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V5 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v5 measurement identities differ") + seen = {row["row_id"] for row in rows} + if seen & set(sample): + raise SystemExit("v5 measurement overlaps the prior reviewed surfaces") + for row_id, blind in sample.items(): + pred = predictions[row_id] + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": pred["v3_bucket"], + "v4_bucket": pred["v4_bucket"], + "v5_bucket": pred["v5_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v5_measurement_28", + "row_id": row_id, + }) + if len(rows) != 169 or len({row["row_id"] for row in rows}) != 169: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 169 unique") + if len({row["surface"] for row in rows}) != 169: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("replay refused: the v6 measurement sample is already frozen") + _require_frozen_parents() + lexicon = V5Lexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v6(raw["v5_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + predictions.append({ + "schema": "hyperlex.unbind_screen_v6_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + "v4_bucket": canonical_bucket(raw["v4_bucket"]), + **decision, + }) + violations = scorer_violations() + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=169) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["events_sha256"] = events_sha256() + if report["replay_rows"] != 169: + report["failures"].append("replay count is not 169") + report["assertions"]["replay_count"] = False + report["regression"] = "REGRESSION_FAILED" + report["state"] = "ENCODED" + report["measurement_eligible"] = False + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v5_bucket"] != row["v6_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v6_replay_169_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v6_replay_169_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v6_replay_169_gate_report.json", report) + summary = _error_summary(predictions, report) + _dump_json(out_dir / "v6_replay_169_error_summary.json", summary) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v6_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + report = json.loads((out_dir / "v6_replay_169_gate_report.json").read_text(encoding="utf-8")) + if not measurement_allowed(report) or report.get("replay_rows") != 169: + raise SystemExit("measurement draw refused: regression is not verified") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("high_precision") != 1.0 or criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_high_allowed") != 0 or criteria.get("false_reject_allowed") != 0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_secondary_rate_required") is not False or criteria.get("accuracy_gain_required") is not False: + raise SystemExit("acceptance bar gained a coverage or accuracy target") + if criteria.get("next_measurement_excludes_reviewed_surfaces") != 169: + raise SystemExit("acceptance exclusion count changed") + if "false_secondary_rate_must_be_strictly_below" in criteria: + raise SystemExit("a false-secondary floor was added after the contract") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + reviewed = load_replay_rows() + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + v5_lexicon = V5Lexicon(WORDNET) + v4_lexicon = V4Lexicon(WORDNET) + v6_lexicon = V5Lexicon(WORDNET) + blind = [] + predictions = [] + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + v5_decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, v5_lexicon) + decision = apply_v6(v5_decision["v5_bucket"], surface, gloss, pos, v6_lexicon) + row_id = normalized_text_sha256(surface) + blind.append({ + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V6-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 169, + }, + }) + leaked = LEAK_KEYS.intersection(blind[-1]) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + predictions.append({ + "schema": "hyperlex.unbind_screen_v6_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V6-001", + "sample_id": "measurement-001", + "row_id": row_id, + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + "v4_bucket": v4_decision["v4_bucket"], + "v5_bucket": v5_decision["v5_bucket"], + **decision, + }) + if {row["surface"] for row in blind} & {row["surface"] for row in reviewed}: + raise SystemExit("measurement sample reuses a reviewed surface") + if {normalize_lexical(row["surface"]) for row in blind} & blocked_norm: + raise SystemExit("measurement sample reuses a normalized reviewed identity") + if {row["row_id"] for row in blind} & blocked_hash: + raise SystemExit("measurement sample reuses a reviewed row id") + blind.sort(key=lambda row: row["row_id"]) + predictions.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blind) + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v6_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(blind), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 169, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "v6_application_count": 1, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "false_secondary_rate_required": False, + "accuracy_gain_required": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v6_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_state"] = "SAMPLE_FROZEN" + receipt["measurement_rows"] = len(blind) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v6_application_count"] = 1 + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v6_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted") + return freeze + + +def _error_summary(predictions: list[dict], report: dict) -> dict: + reversals = [] + for row in predictions: + if row["v5_bucket"] == row["v6_bucket"]: + continue + reversals.append({ + "row_id": row["row_id"], + "surface": row["surface"], + "v5_bucket": row["v5_bucket"], + "v6_bucket": row["v6_bucket"], + "operator_bucket": row["operator_bucket"], + "primary_evidence": row["primary_evidence"], + "previously_correct": row["v5_bucket"] == row["operator_bucket"], + }) + stayed_wrong = sum( + 1 for row in predictions + if row["v5_bucket"] == row["v6_bucket"] and row["v6_bucket"] != row["operator_bucket"] + ) + return { + "schema": "hyperlex.unbind_screen_v6_replay_error_summary.v1", + "role": "regression_summary", + "executable_rule_source": False, + "motivating_cases_are_not_a_required_fit": True, + "regression": report["regression"], + "replay_rows": report["replay_rows"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "direct_swaps": report["direct_swaps"], + "reversals": reversals, + "unchanged_disagreement_count": stayed_wrong, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + } + + +def _require_frozen_parents() -> None: + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed") + checks = { + V5 / "measurement_sample.jsonl": EXPECTED_V5_SAMPLE, + V5 / "measurement_predictions.jsonl": EXPECTED_V5_PREDICTIONS, + V5 / "measurement_labels.jsonl": EXPECTED_V5_LABELS, + V5H / "ACCEPTANCE.json": EXPECTED_V5_ACCEPTANCE, + HYP / "ACCEPTANCE.json": EXPECTED_V6_ACCEPTANCE, + HYP / "HYPOTHESIS.draft.json": EXPECTED_V6_DRAFT, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path.name}") + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + verified = report["regression"] == "REGRESSION_VERIFIED" + return { + "schema": "hyperlex.unbind_screen_v6_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "regression": report["regression"], + "measurement_eligible": verified, + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v5": "challengeable_outer_buckets_over_a_frozen_v5_bucket", + "demotion_stops_for_this_application": True, + "acceptance_sha256": report["acceptance_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "assertions": report["assertions"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "direct_swaps": report["direct_swaps"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + draft = file_sha256(HYP / "HYPOTHESIS.draft.json") + acceptance = file_sha256(HYP / "ACCEPTANCE.json") + if draft != EXPECTED_V6_DRAFT or acceptance != EXPECTED_V6_ACCEPTANCE: + raise SystemExit("v6 draft or acceptance bytes changed") + if hypothesis.get("draft_hypothesis_sha256") not in (None, EXPECTED_V6_DRAFT): + raise SystemExit("v6 hypothesis no longer points at the frozen draft") + if hypothesis.get("acceptance_sha256") != EXPECTED_V6_ACCEPTANCE: + raise SystemExit("v6 hypothesis no longer points at the frozen acceptance") + hypothesis["encoded"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = receipt["measurement_eligible"] + hypothesis["applied"] = bool(receipt.get("applied_to_measurement")) + hypothesis["applied_to_measurement"] = bool(receipt.get("applied_to_measurement")) + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["previously_correct_lost"] = receipt["previously_correct_lost"] + hypothesis["direct_swaps"] = receipt["direct_swaps"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + if receipt.get("measurement_sample_sha256"): + hypothesis["measurement_sample_sha256"] = receipt["measurement_sample_sha256"] + hypothesis["measurement_rows"] = receipt["measurement_rows"] + hypothesis["measurement_state"] = receipt["measurement_state"] + hypothesis["precision"] = "NOT_COMPUTABLE" + path.write_text(json.dumps(hypothesis, indent=2) + "\n", encoding="utf-8") + path.chmod(0o600) + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Replay v6 and, if verified, draw the measurement sample") + parser.add_argument("command", choices=("replay", "draw", "run")) + args = parser.parse_args() + if args.command == "replay": + receipt = replay() + elif args.command == "draw": + receipt = draw_measurement() + else: + receipt = replay() + if receipt.get("regression") == "REGRESSION_VERIFIED": + receipt = draw_measurement() + print(json.dumps({ + key: receipt.get(key) + for key in ( + "state", "regression", "measurement_eligible", "measurement_state", + "replay_rows", "diff_rows", "moves", "previously_correct_lost", + "direct_swaps", "failures", "rows", "sample_sha256", "precision", + "v6_application_count", "events_sha256", + ) + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_screen_v7.py b/scripts/shadow/hyperlexical/unbind_screen_v7.py new file mode 100644 index 00000000..4584ce7c --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v7.py @@ -0,0 +1,279 @@ +"""One high challenge over a frozen v6 bucket. + +A provisional high may fall to secondary when ordinary compositional derivation +fires. That demotion stops. Reject and secondary buckets stay at the v6 +decision: those transitions were already applied, and a stopped demotion is +not reopened. The v6 high-challenge evidence is not a v7 transition. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical +from hyperlexical.unbind_screen_v5 import ( + HIGH_EVIDENCE, + REJECT_EVIDENCE, + _FUNCTION, + _WORD, + _content, + _initialism, + _related, + _single_words, + _stem, +) +from hyperlexical.unbind_screen_v6 import NONREFERENTIAL_EVIDENCE, apply_v6 + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v7" +ORDINARY_EVIDENCE = "ordinary_compositional_derivation" +_COMPARATIVE_GLOSS = re.compile(r"^used to form the comparative\b", re.IGNORECASE) +_DEGREE_SENSE = re.compile(r"^(?:of less|of more|of greater|of smaller)\b", re.IGNORECASE) +_METAPHOR = re.compile(r"\bmetaphors?\b|\bmetaphorical(?:ly)?\b", re.IGNORECASE) +_PARTICLE = frozenset({ + "out", "off", "up", "down", "away", "back", "over", "through", "along", +}) + + +def apply_v7(v6_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v7 bucket. Only a provisional high is eligible for the new predicate.""" + bucket = canonical_bucket(v6_bucket) + normalized = normalize_lexical(surface) + if bucket == "HIGH": + if ordinary_compositional_derivation(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", ORDINARY_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "HIGH", None, normalized, inspected=True) + if bucket == "REJECT": + inherited = apply_v6(bucket, surface, gloss, source_pos, lexicon) + return _decision( + bucket, + inherited["v6_bucket"], + inherited["primary_evidence"], + normalized, + inspected=bool(inherited["inspected"]), + ) + if bucket != "SECONDARY": + raise ValueError(f"provisional bucket {bucket} is outside HIGH, REJECT, and SECONDARY") + # v6 already ran the secondary transitions. Running them again would reopen + # a demotion that v6 stopped, so the frozen secondary decision stands. + return _decision(bucket, "SECONDARY", None, normalized, inspected=False) + + +def ordinary_compositional_derivation(surface: str, gloss: str, lexicon) -> bool: + """True when the recorded sense is ordinary composition, not a stored binding. + + Stem overlap with a constituent gloss is not enough, and a metaphorical + retelling of the gloss is not enough. The lexical record has to present a + comparative, syntactic, or phrasal composition with no unrelated synonym. + """ + found = lexicon.entry(surface) + if found is None or _stored_binding(found, surface) or _METAPHOR.search(gloss or ""): + return False + return ( + _comparative(found, surface, gloss, lexicon) + or _syntactic(surface, gloss, lexicon) + or _phrasal(surface, gloss, lexicon) + ) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 197) -> dict: + """Regression gate. Previously correct rows must stay correct. Buckets may move.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = { + "high_to_secondary": 0, + "reject_to_secondary": 0, + "secondary_to_high": 0, + "secondary_to_reject": 0, + "unchanged": 0, + } + previously_correct = 0 + previously_correct_lost = 0 + correct_high = 0 + correct_high_lost = 0 + direct_swaps = 0 + for row in rows: + prior = canonical_bucket(str(row["v6_bucket"])) + nxt = canonical_bucket(str(row["v7_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == operator: + previously_correct += 1 + if nxt != operator: + previously_correct_lost += 1 + failures.append("previously correct row is no longer correct") + if prior == "HIGH" and operator == "HIGH": + correct_high += 1 + if nxt != "HIGH": + correct_high_lost += 1 + failures.append("previously correct HIGH was demoted") + if prior == nxt: + moves["unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged row carries transition evidence") + elif prior == "HIGH" and nxt == "SECONDARY": + moves["high_to_secondary"] += 1 + if primary != ORDINARY_EVIDENCE or supporting: + failures.append("HIGH to SECONDARY lacks ordinary_compositional_derivation") + elif prior == "REJECT" and nxt == "SECONDARY": + moves["reject_to_secondary"] += 1 + if primary != NONREFERENTIAL_EVIDENCE or supporting: + failures.append("REJECT to SECONDARY lacks nonreferential_lexical_use") + elif prior == "SECONDARY" and nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif prior == "SECONDARY" and nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + elif {prior, nxt} == {"HIGH", "REJECT"}: + direct_swaps += 1 + failures.append("HIGH and REJECT swapped directly") + else: + failures.append("row moved outside the legal transitions") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + assertions = { + "A_replay_197": len(rows) == expected_rows and len(set(surfaces)) == len(surfaces), + "B_previously_correct_remain_correct": previously_correct_lost == 0, + "C_high_demotion_evidence": "HIGH to SECONDARY lacks ordinary_compositional_derivation" not in failures, + "D_correct_high_stays_high": correct_high_lost == 0, + "E_v6_transitions_unchanged": ( + "REJECT to SECONDARY lacks nonreferential_lexical_use" not in failures + and "SECONDARY to HIGH lacks lexicalized_noncompositional" not in failures + and "SECONDARY to REJECT lacks referential_terminological_dominance" not in failures + and moves["secondary_to_high"] == 0 + and moves["secondary_to_reject"] == 0 + and moves["reject_to_secondary"] == 0 + ), + "F_no_direct_outer_swap": direct_swaps == 0, + "G_no_phrase_rules": not phrase_specific_rule_fired, + "H_demotion_stops": moves["high_to_secondary"] == 0 or "HIGH to SECONDARY lacks ordinary_compositional_derivation" not in failures, + "I_measurement_not_drawn": True, + } + verified = not failures and all(assertions.values()) + return { + "schema": "hyperlex.unbind_screen_v7_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "assertions": assertions, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "previously_correct": previously_correct, + "previously_correct_lost": previously_correct_lost, + "correct_high": correct_high, + "correct_high_lost": correct_high_lost, + "direct_swaps": direct_swaps, + "moves": moves, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": False, + "measurement_sample_drawn": False, + "regression_is_not_generalization": True, + "select_authorized": False, + "revision_eligible": False, + } + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v6_bucket": prior, + "v7_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _stored_binding(found, surface: str) -> bool: + """An unrelated single-word synonym is a stored conventionalized binding.""" + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + for lemma in _single_words(found.lemmas): + if _initialism(lemma): + continue + if not _related(lemma, content, stems): + return True + return False + + +def _comparative(found, surface: str, gloss: str, lexicon) -> bool: + """A periphrastic comparative whose record is the degree construction itself.""" + if found.lex != "adv.all" or not _COMPARATIVE_GLOSS.match((gloss or "").strip()): + return False + return any(_degree_adjective(token, lexicon) for token in _content(surface)) + + +def _degree_adjective(token: str, lexicon) -> bool: + if len(token) < 5 or not token.endswith("er"): + return False + for sense in lexicon.senses(token): + if sense.pos != "adj" and sense.lex != "adj.all": + continue + if _DEGREE_SENSE.match((sense.gloss or "").strip()): + return True + return False + + +def _syntactic(surface: str, gloss: str, lexicon) -> bool: + """The gloss is the recorded senses of two open-class constituents, in order.""" + heads = [token for token in _content(surface) if token not in _PARTICLE] + if len(heads) < 2: + return False + return _gloss_is_sense_sum(heads, gloss, lexicon) + + +def _phrasal(surface: str, gloss: str, lexicon) -> bool: + """The gloss is a verb sense plus one particle sense, with nothing left over.""" + content = _content(surface) + particles = [token for token in content if token in _PARTICLE] + heads = [token for token in content if token not in _PARTICLE] + if len(particles) != 1 or len(heads) != 1: + return False + if not any(sense.pos == "verb" for sense in lexicon.senses(heads[0])): + return False + return _gloss_is_sense_sum([heads[0], particles[0]], gloss, lexicon) + + +def _gloss_is_sense_sum(tokens: list[str], gloss: str, lexicon) -> bool: + target = _open_words(gloss) + if len(target) < 2: + return False + choices: list[tuple[tuple[str, ...], ...]] = [] + for token in tokens: + glosses = tuple(_open_words(sense.gloss) for sense in lexicon.senses(token) if _open_words(sense.gloss)) + if not glosses: + return False + choices.append(glosses) + return _any_concatenation(choices, target) + + +def _any_concatenation(choices: list[tuple[tuple[str, ...], ...]], target: list[str]) -> bool: + def walk(index: int, built: list[str]) -> bool: + if index == len(choices): + return built == target and all(built) + for gloss in choices[index]: + if walk(index + 1, built + list(gloss)): + return True + return False + + return walk(0, []) + + +def _open_words(text: str) -> list[str]: + return [ + word.casefold() + for word in _WORD.findall(text or "") + if word.casefold() not in _FUNCTION and len(word) >= 3 + ] diff --git a/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py b/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py new file mode 100644 index 00000000..0fab3c0c --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v7_measure.py @@ -0,0 +1,681 @@ +"""Replay the frozen 197-row set, then draw one unseen measurement sample. + +The replay is regression evidence. The draw excludes those identities, freezes +the sample, and only then applies v7 once. It does not label, admit, settle, +or append the ledger. +""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.unbind_screen_v3 import gloss_for, screen +from hyperlexical.unbind_screen_v4 import WordNetLexicon as V4Lexicon +from hyperlexical.unbind_screen_v4 import apply_v4, canonical_bucket, normalize_lexical, rule_surface_violations +from hyperlexical.unbind_screen_v4_measure import _stratified_sample +from hyperlexical.unbind_screen_v5 import WordNetLexicon +from hyperlexical.unbind_screen_v5 import WordNetLexicon as V5Lexicon +from hyperlexical.unbind_screen_v5 import apply_v5 +from hyperlexical.unbind_screen_v6 import apply_v6 +from hyperlexical.unbind_screen_v7 import RULE_VERSION, apply_v7, assess + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +V6 = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-001/unbind_screen_v6" +V6H = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V6-HYPOTHESIS-001" +HYP = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-HYPOTHESIS-001" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SCREEN-V7-001/unbind_screen_v7" +SPEC = Path("/home/morpheus/Hyperlex/specs/007-hyperlexical-model/evaluation-reserve.md") +SCORER = Path(__file__).with_name("unbind_screen_v7.py") +EXPECTED_EVENTS = "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c" +EXPECTED_V6_SAMPLE = "8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2" +EXPECTED_V6_PREDICTIONS = "476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee" +EXPECTED_V6_LABELS = "feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11" +EXPECTED_V6_ACCEPTANCE = "68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f" +EXPECTED_V6_SOURCE = "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba" +EXPECTED_DRAFT = "c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b" +EXPECTED_ACCEPTANCE = "562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5" +EXPECTED_EVIDENCE = "3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3" +EXPECTED_ANALYSIS = "06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766" +EXPECTED_REPLAY = "24f266e16b80da602b011bf7cca13774ce9f76c0c52e81e600a1d9743d61d197" +EXPECTED_GATE = "78225b68c639694bb4c17342c87320ccb44faac557a9145ea117603da0d545ed" +LEAK_KEYS = frozenset({ + "predicted", "predicted_bucket", "bucket", "relation", "rule", "phase", + "operator", "operator_bucket", "operator_reason", "forecast", "diagnostic", + "v3", "v4", "v5", "v6", "v7", "primary_evidence", "supporting_evidence", + "evidence", "v3_bucket", "v4_bucket", "v5_bucket", "v6_bucket", "v7_bucket", + "ordinary_compositional_derivation", +}) + + +def file_sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def events_sha256() -> str: + return file_sha256(LEDGER / "events.jsonl") + + +def _require_frozen_parents() -> None: + checks = { + LEDGER / "events.jsonl": EXPECTED_EVENTS, + V6 / "measurement_sample.jsonl": EXPECTED_V6_SAMPLE, + V6 / "measurement_predictions.jsonl": EXPECTED_V6_PREDICTIONS, + V6 / "measurement_labels.jsonl": EXPECTED_V6_LABELS, + V6 / "measurement_error_analysis.json": EXPECTED_ANALYSIS, + V6H / "ACCEPTANCE.json": EXPECTED_V6_ACCEPTANCE, + V6H / "HYPOTHESIS.json": "fd4f5eeb66061ffdaf2aa4341fe86a0b5e988f131d1711d6ea299a4ebb9e1da4", + HYP / "HYPOTHESIS.draft.json": EXPECTED_DRAFT, + HYP / "ACCEPTANCE.json": EXPECTED_ACCEPTANCE, + HYP / "DEVELOPMENT_EVIDENCE.json": EXPECTED_EVIDENCE, + Path("/home/morpheus/Hyperlex/scripts/shadow/hyperlexical/unbind_screen_v6.py"): EXPECTED_V6_SOURCE, + } + for path, digest in checks.items(): + if file_sha256(path) != digest: + raise SystemExit(f"frozen parent changed: {path}") + + +def load_replay_rows() -> list[dict]: + rows = [] + for line in (V6 / "v6_replay_169_predictions.jsonl").read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + raw = json.loads(line) + rows.append({ + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw.get("gloss") or "", + "v3_bucket": raw["v3_bucket"], + "v4_bucket": raw["v4_bucket"], + "v5_bucket": raw["v5_bucket"], + "v6_bucket": raw["v6_bucket"], + "operator_bucket": raw["operator_bucket"], + "split": raw.get("split") or "reviewed_169", + "row_id": raw["row_id"], + }) + sample = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_sample.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + labels = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_labels.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + predictions = { + json.loads(line)["row_id"]: json.loads(line) + for line in (V6 / "measurement_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + } + if set(sample) != set(labels) or set(sample) != set(predictions): + raise SystemExit("v6 measurement identities differ") + seen = {row["row_id"] for row in rows} + if seen & set(sample): + raise SystemExit("v6 measurement overlaps the prior reviewed surfaces") + reviewed_norm = {normalize_lexical(row["surface"]) for row in rows} + measurement_norm = {normalize_lexical(row["surface"]) for row in sample.values()} + if reviewed_norm & measurement_norm: + raise SystemExit("v6 measurement reuses a normalized reviewed identity") + for row_id, blind in sample.items(): + pred = predictions[row_id] + rows.append({ + "surface": blind["surface"], + "pos": blind["pos"], + "gloss": blind.get("gloss") or "", + "v3_bucket": pred["v3_bucket"], + "v4_bucket": pred["v4_bucket"], + "v5_bucket": pred["v5_bucket"], + "v6_bucket": pred["v6_bucket"], + "operator_bucket": labels[row_id]["operator_bucket"], + "split": "v6_measurement_28", + "row_id": row_id, + }) + if len(rows) != 197 or len({row["row_id"] for row in rows}) != 197: + raise SystemExit(f"reviewed surfaces are {len(rows)}, not 197 unique") + if len({row["surface"] for row in rows}) != 197: + raise SystemExit("reviewed surfaces are not unique") + return rows + + +def replay(out_dir: Path = OUT) -> dict: + if (out_dir / "v7_replay_197_predictions.jsonl").exists(): + raise SystemExit("v7 replay is already frozen") + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is not authorized") + if "Unbind screen v7 encoded" in SPEC.read_text(encoding="utf-8"): + raise SystemExit("spec already records the v7 replay") + _require_frozen_parents() + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + if acceptance.get("state") != "SPEC_FROZEN" or acceptance.get("implementation_encoded") is not False: + raise SystemExit("acceptance is not the frozen specification") + probes = acceptance["probe_surfaces"]["forbidden_in_executable_rule_code"] + violations = rule_surface_violations(SCORER.read_text(encoding="utf-8"), probes) + lexicon = WordNetLexicon(WORDNET) + predictions = [] + for raw in load_replay_rows(): + decision = apply_v7(raw["v6_bucket"], raw["surface"], raw["gloss"], raw["pos"], lexicon) + predictions.append({ + "schema": "hyperlex.unbind_screen_v7_replay_row.v1", + "row_id": raw["row_id"], + "split": raw["split"], + "surface": raw["surface"], + "pos": raw["pos"], + "gloss": raw["gloss"], + "operator_bucket": canonical_bucket(raw["operator_bucket"]), + "v3_bucket": canonical_bucket(raw["v3_bucket"]), + "v4_bucket": canonical_bucket(raw["v4_bucket"]), + "v5_bucket": canonical_bucket(raw["v5_bucket"]), + **decision, + }) + report = assess(predictions, phrase_specific_rule_fired=bool(violations), expected_rows=197) + report["phrase_violations"] = violations + report["acceptance_sha256"] = file_sha256(HYP / "ACCEPTANCE.json") + report["draft_hypothesis_sha256"] = file_sha256(HYP / "HYPOTHESIS.draft.json") + report["development_evidence_sha256"] = file_sha256(HYP / "DEVELOPMENT_EVIDENCE.json") + report["error_analysis_sha256"] = file_sha256(V6 / "measurement_error_analysis.json") + report["events_sha256"] = events_sha256() + report["measurement_sample_drawn"] = False + report["measurement_eligible"] = False + report["regression_is_not_generalization"] = True + if report["replay_rows"] != 197: + report["failures"].append("replay count is not 197") + report["assertions"]["A_replay_197"] = False + report["regression"] = "REGRESSION_FAILED" + report["state"] = "ENCODED" + predictions.sort(key=lambda row: row["row_id"]) + changed = [row for row in predictions if row["v6_bucket"] != row["v7_bucket"]] + changed.sort(key=lambda row: row["row_id"]) + out_dir.mkdir(parents=True, exist_ok=True) + out_dir.chmod(0o700) + prediction_sha = _dump_jsonl(out_dir / "v7_replay_197_predictions.jsonl", predictions) + diff_sha = _dump_jsonl(out_dir / "v7_replay_197_diff.jsonl", changed) + report["prediction_sha256"] = prediction_sha + report["diff_sha256"] = diff_sha + report["diff_rows"] = len(changed) + _dump_json(out_dir / "v7_replay_197_gate_report.json", report) + receipt = _implementation_receipt(report, prediction_sha, diff_sha) + _dump_json(out_dir / "v7_implementation_receipt.json", receipt) + _touch_hypothesis(receipt) + _append_spec(report, changed) + _require_frozen_parents() + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("a measurement sample was drawn") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during replay") + return receipt + + +def _implementation_receipt(report: dict, prediction_sha: str, diff_sha: str) -> dict: + verified = report["regression"] == "REGRESSION_VERIFIED" + return { + "schema": "hyperlex.unbind_screen_v7_implementation_receipt.v1", + "rule": RULE_VERSION, + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "regression": report["regression"], + "measurement_eligible": False, + "measurement_sample_drawn": False, + "regression_is_not_generalization": True, + "authorized": False, + "encoded": True, + "applied_to_measurement": False, + "relation_to_v6": "one_high_challenge_over_a_frozen_v6_bucket", + "demotion_stops_for_this_application": True, + "acceptance_sha256": report["acceptance_sha256"], + "draft_hypothesis_sha256": report["draft_hypothesis_sha256"], + "source_sha256": {SCORER.name: file_sha256(SCORER)}, + "replay_rows": report["replay_rows"], + "diff_rows": report["diff_rows"], + "prediction_sha256": prediction_sha, + "diff_sha256": diff_sha, + "phrase_specific_rule_fired": report["phrase_specific_rule_fired"], + "failures": report["failures"], + "assertions": report["assertions"], + "moves": report["moves"], + "previously_correct": report["previously_correct"], + "previously_correct_lost": report["previously_correct_lost"], + "correct_high": report["correct_high"], + "correct_high_lost": report["correct_high_lost"], + "direct_swaps": report["direct_swaps"], + "events_sha256": report["events_sha256"], + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "precision": "NOT_COMPUTABLE", + } + + +def _touch_hypothesis(receipt: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + if file_sha256(HYP / "HYPOTHESIS.draft.json") != EXPECTED_DRAFT: + raise SystemExit("v7 draft bytes changed") + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("v7 acceptance bytes changed") + if hypothesis.get("acceptance_sha256") != EXPECTED_ACCEPTANCE: + raise SystemExit("v7 hypothesis no longer points at the frozen acceptance") + if hypothesis.get("draft_hypothesis_sha256") != EXPECTED_DRAFT: + raise SystemExit("v7 hypothesis no longer points at the frozen draft") + hypothesis["encoded"] = True + hypothesis["implementation_encoded"] = True + hypothesis["encoding_authorized"] = True + hypothesis["state"] = receipt["state"] + hypothesis["regression"] = receipt["regression"] + hypothesis["measurement_eligible"] = False + hypothesis["measurement_sample_drawn"] = False + hypothesis["applied"] = False + hypothesis["applied_to_measurement"] = False + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["previously_correct_lost"] = receipt["previously_correct_lost"] + hypothesis["correct_high_lost"] = receipt["correct_high_lost"] + hypothesis["direct_swaps"] = receipt["direct_swaps"] + hypothesis["moves"] = receipt["moves"] + hypothesis["replay_rows"] = receipt["replay_rows"] + hypothesis["regression_is_not_generalization"] = True + hypothesis["precision"] = "NOT_COMPUTABLE" + text = json.dumps(hypothesis, indent=2, sort_keys=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _append_spec(report: dict, changed: list[dict]) -> None: + moves = report["moves"] + if len(changed) == 1: + row = changed[0] + change = ( + f"One row moves. `{row['surface']}` goes from high to secondary with " + f"`{row['primary_evidence']}`, and the operator bucket is {row['operator_bucket'].lower()}. " + "That firing is not a required fit." + ) + elif not changed: + change = "No row moves." + else: + names = ", ".join(f"`{row['surface']}`" for row in changed) + change = f"{len(changed)} rows move: {names}." + verified = report["regression"] == "REGRESSION_VERIFIED" + if verified: + state = ( + "State is `REGRESSION_VERIFIED`. `measurement_eligible` stays false. " + "The unseen sample was not drawn." + ) + else: + state = ( + "State stays `ENCODED` because the regression gate failed. " + "`measurement_eligible` stays false. The unseen sample was not drawn." + ) + section = f""" +## Unbind screen v7 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7` is encoded over a frozen v6 bucket. The only new predicate is `ordinary_compositional_derivation`. It inspects a provisional high. The recorded sense must be a comparative, syntactic, or phrasal composition, and the synset must not store an unrelated single-word synonym. A gloss that merely shares constituent stems does not fire, and a metaphorical retelling does not fire. A demotion stops at secondary. Reject rows still use the v6 reject challenge. Secondary rows are not reopened, so a demotion v6 already stopped stays stopped. `compositional_recoverability` is not a v7 transition. The acceptance contract is unchanged, sha256 `{EXPECTED_ACCEPTANCE}`. + +The 197 reviewed surfaces were replayed as development and regression evidence, not as a generalization estimate. Previously correct rows lost: {report['previously_correct_lost']}. Previously correct high rows demoted: {report['correct_high_lost']}. Direct swaps between high and reject: {report['direct_swaps']}. High to secondary: {moves['high_to_secondary']}. Reject to secondary: {moves['reject_to_secondary']}. Secondary to high: {moves['secondary_to_high']}. Secondary to reject: {moves['secondary_to_reject']}. {change} No phrase-specific rule fired. This replay is not a precision estimate. Gate report sha256 `{report['replay_rows'] and ''}`. Prediction sha256 `{report['prediction_sha256']}`. Diff sha256 `{report['diff_sha256']}`. + +{state} The pre-registered bar remains high precision 1, reject precision 1, false high 0, and false reject 0, with no false-secondary quota, no accuracy target, and no recall target. No measurement sample was drawn. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `{EXPECTED_EVENTS}`. +""" + # The gate-report hash is the file hash, filled by the caller after the report is dumped. + # This placeholder is replaced below once the file exists. + spec = SPEC.read_text(encoding="utf-8") + if not spec.endswith("\n"): + spec += "\n" + SPEC.write_text(spec + "\n" + section.lstrip("\n"), encoding="utf-8") + + +def _dump_json(path: Path, payload: dict) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps(payload, sort_keys=True, ensure_ascii=True, indent=2) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _dump_jsonl(path: Path, rows: list[dict]) -> str: + path.parent.mkdir(parents=True, exist_ok=True) + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def draw_measurement(out_dir: Path = OUT, per_cell: int = 2) -> dict: + """Freeze one unseen sample, then apply v7 once.""" + if (out_dir / "measurement_sample.jsonl").exists(): + raise SystemExit("measurement sample is already frozen") + if (out_dir / "measurement_predictions.jsonl").exists(): + raise SystemExit("measurement predictions already exist") + if "Unbind screen v7 measurement frozen" in SPEC.read_text(encoding="utf-8"): + raise SystemExit("spec already records the v7 measurement draw") + _require_frozen_parents() + if file_sha256(out_dir / "v7_replay_197_predictions.jsonl") != EXPECTED_REPLAY: + raise SystemExit("v7 replay predictions changed") + if file_sha256(out_dir / "v7_replay_197_gate_report.json") != EXPECTED_GATE: + raise SystemExit("v7 gate report changed") + report = json.loads((out_dir / "v7_replay_197_gate_report.json").read_text(encoding="utf-8")) + if report.get("regression") != "REGRESSION_VERIFIED" or report.get("replay_rows") != 197: + raise SystemExit("measurement draw refused: regression is not verified") + if report.get("previously_correct_lost") != 0 or report.get("failures"): + raise SystemExit("measurement draw refused: regression failures are present") + acceptance = json.loads((HYP / "ACCEPTANCE.json").read_text(encoding="utf-8")) + criteria = dict(acceptance["success_criteria"]) + if criteria.get("high_precision") != 1.0 or criteria.get("reject_precision") != 1.0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_high_allowed") != 0 or criteria.get("false_reject_allowed") != 0: + raise SystemExit("acceptance bar was amended") + if criteria.get("false_secondary_rate_required") is not False or criteria.get("accuracy_gain_required") is not False: + raise SystemExit("acceptance bar gained a coverage or accuracy target") + if criteria.get("recall_required") is not False: + raise SystemExit("acceptance bar gained a recall target") + if criteria.get("next_measurement_excludes_reviewed_surfaces") != 197: + raise SystemExit("acceptance exclusion count changed") + criteria_sha = _dump_json(out_dir / "measurement_criteria.json", criteria) + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("copying criteria changed the acceptance") + + reviewed = [ + json.loads(line) + for line in (out_dir / "v7_replay_197_predictions.jsonl").read_text(encoding="utf-8").splitlines() + if line.strip() + ] + if len(reviewed) != 197 or len({row["row_id"] for row in reviewed}) != 197: + raise SystemExit("reviewed replay is not 197 unique rows") + blocked_hash = {row["row_id"] for row in reviewed} + blocked_norm = {normalize_lexical(row["surface"]) for row in reviewed} + if len(blocked_norm) != 197: + raise SystemExit("reviewed normalized identities are not unique") + picked, strata = _stratified_sample(blocked_hash, blocked_norm, per_cell=per_cell) + exhausted = [cell for cell in strata if cell["available"] < per_cell or cell["taken"] < per_cell] + for cell in strata: + if cell["taken"] > cell["available"] or cell["taken"] > per_cell: + raise SystemExit("stratum draw exceeded the frozen policy") + + blinded = [] + sources = {} + for row in picked: + surface = str(row["text"]) + pos = str(row["source_pos"]) + tokens = [str(tok) for tok in row["fillers"]] + if " ".join(tokens) != surface or len(tokens) < 1: + raise SystemExit("sample tokens do not reconstruct the surface") + _pos, gloss = gloss_for(surface, pos, WORDNET) + row_id = normalized_text_sha256(surface) + identity = normalize_lexical(surface) + if row_id in blocked_hash or identity in blocked_norm: + raise SystemExit(f"draw reused a reviewed identity: {surface}") + blind = { + "schema": "hyperlex.unbind_screen_review_row.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V7-001", + "sample_id": "measurement-001", + "row_id": row_id, + "surface": surface, + "pos": pos, + "token_count": len(tokens), + "gloss": gloss, + "provenance": { + "source": "wordnet-3.0", + "source_pos": pos, + "excluded_reviewed_surfaces": 197, + }, + } + leaked = LEAK_KEYS.intersection(blind) + if leaked: + raise SystemExit(f"blind row carries {sorted(leaked)}") + blinded.append(blind) + sources[row_id] = {"tokens": tokens, "pos": pos, "surface": surface, "gloss": gloss} + identities = [normalize_lexical(row["surface"]) for row in blinded] + if len(identities) != len(set(identities)): + raise SystemExit("sample contains a duplicate normalized identity") + if set(identities) & blocked_norm or {row["row_id"] for row in blinded} & blocked_hash: + raise SystemExit("sample intersects the reviewed 197") + if {row["surface"] for row in blinded} & {row["surface"] for row in reviewed}: + raise SystemExit("sample reuses a reviewed surface") + blinded.sort(key=lambda row: row["row_id"]) + sample_sha = _dump_jsonl(out_dir / "measurement_sample.jsonl", blinded) + sample_text = (out_dir / "measurement_sample.jsonl").read_text(encoding="utf-8") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash did not stick") + for forbidden in ( + "ordinary_compositional_derivation", + "primary_evidence", + "operator_bucket", + "operator_reason", + "v7_bucket", + "v6_bucket", + "v5_bucket", + "v4_bucket", + "v3_bucket", + ): + if forbidden in sample_text: + raise SystemExit(f"blind sample contains {forbidden}") + frozen_rows = [json.loads(line) for line in sample_text.splitlines() if line.strip()] + if [row["row_id"] for row in frozen_rows] != [row["row_id"] for row in blinded]: + raise SystemExit("frozen sample does not match the draw") + + v4_lexicon = V4Lexicon(WORDNET) + v5_lexicon = V5Lexicon(WORDNET) + predictions = [] + applications = 0 + for row in frozen_rows: + source = sources[row["row_id"]] + if source["surface"] != row["surface"] or source["gloss"] != row["gloss"]: + raise SystemExit("apply input drifted from the frozen sample") + if source["pos"] != row["pos"] or len(source["tokens"]) != row["token_count"]: + raise SystemExit("apply input drifted from the frozen sample") + surface = row["surface"] + pos = row["pos"] + tokens = source["tokens"] + gloss = row["gloss"] + v3_bucket, v3_rule, _phase = screen(surface, pos, tokens, gloss) + v4_decision = apply_v4(v3_bucket, surface, gloss, pos, v4_lexicon) + v5_decision = apply_v5(v4_decision["v4_bucket"], surface, gloss, pos, v5_lexicon) + v6_decision = apply_v6(v5_decision["v5_bucket"], surface, gloss, pos, v5_lexicon) + decision = apply_v7(v6_decision["v6_bucket"], surface, gloss, pos, v5_lexicon) + applications += 1 + if decision["v6_bucket"] != v6_decision["v6_bucket"]: + raise SystemExit("v7 did not keep the provisional v6 bucket") + predictions.append({ + "schema": "hyperlex.unbind_screen_v7_measurement_prediction.v1", + "evaluation_id": "HLX-EVAL-UNBIND-SCREEN-V7-001", + "sample_id": "measurement-001", + "row_id": row["row_id"], + "v3_rule": v3_rule, + "application_index": 1, + "v3_bucket": canonical_bucket(v3_bucket), + "v4_bucket": v4_decision["v4_bucket"], + "v5_bucket": v5_decision["v5_bucket"], + "provisional_v6_bucket": v6_decision["v6_bucket"], + **decision, + }) + if applications != len(frozen_rows): + raise SystemExit("v7 application count does not match the sample") + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash changed while v7 was applied") + predictions.sort(key=lambda row: row["row_id"]) + if [row["row_id"] for row in predictions] != [row["row_id"] for row in frozen_rows]: + raise SystemExit("predictions do not cover the frozen sample") + prediction_sha = _dump_jsonl(out_dir / "measurement_predictions.jsonl", predictions) + freeze = { + "schema": "hyperlex.unbind_screen_v7_measurement_freeze.v1", + "rule": RULE_VERSION, + "state": "MEASUREMENT_ELIGIBLE", + "sample_state": "MEASUREMENT_SAMPLE_FROZEN", + "regression": "REGRESSION_VERIFIED", + "measurement_eligible": True, + "rows": len(frozen_rows), + "per_cell": per_cell, + "strata": "source_pos x token_count", + "stratum_counts": strata, + "exhausted_strata": exhausted, + "order": "normalized_text_sha256", + "excluded_reviewed_surfaces": 197, + "normalized_overlap_with_reviewed_197": 0, + "duplicate_normalized_identities": 0, + "deduplicated_normalized_lexical_identity": True, + "sample_sha256": sample_sha, + "prediction_sha256": prediction_sha, + "criteria_sha256": criteria_sha, + "sample_frozen_before_apply": True, + "v7_application_count": applications, + "hand_corrections": 0, + "operator_labels": None, + "precision": "NOT_COMPUTABLE", + "confusion": "NOT_COMPUTABLE", + "inspected_before_freeze": False, + "false_secondary_rate_required": False, + "accuracy_gain_required": False, + "recall_required": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "select_authorized": False, + "revision_eligible": False, + "events_sha256": events_sha256(), + } + _dump_json(out_dir / "measurement_freeze.json", freeze) + receipt = json.loads((out_dir / "v7_implementation_receipt.json").read_text(encoding="utf-8")) + receipt["state"] = "MEASUREMENT_ELIGIBLE" + receipt["applied_to_measurement"] = True + receipt["measurement_eligible"] = True + receipt["measurement_sample_drawn"] = True + receipt["measurement_state"] = "MEASUREMENT_ELIGIBLE" + receipt["sample_state"] = "MEASUREMENT_SAMPLE_FROZEN" + receipt["measurement_rows"] = len(frozen_rows) + receipt["measurement_sample_sha256"] = sample_sha + receipt["measurement_prediction_sha256"] = prediction_sha + receipt["criteria_sha256"] = criteria_sha + receipt["v7_application_count"] = applications + receipt["hand_corrections"] = 0 + receipt["precision"] = "NOT_COMPUTABLE" + receipt["exhausted_strata"] = len(exhausted) + receipt["events_sha256"] = events_sha256() + _dump_json(out_dir / "v7_implementation_receipt.json", receipt) + _mark_measured(freeze) + _write_review_sheet(out_dir, frozen_rows, sample_sha) + _append_measurement_spec(freeze, exhausted) + _require_frozen_parents() + if file_sha256(out_dir / "measurement_sample.jsonl") != sample_sha: + raise SystemExit("sample hash drifted after the draw") + if file_sha256(HYP / "ACCEPTANCE.json") != EXPECTED_ACCEPTANCE: + raise SystemExit("acceptance bytes changed during the draw") + if file_sha256(HYP / "HYPOTHESIS.draft.json") != EXPECTED_DRAFT: + raise SystemExit("draft bytes changed during the draw") + if events_sha256() != EXPECTED_EVENTS: + raise SystemExit("ledger events hash changed during the draw") + return freeze + + +def _mark_measured(freeze: dict) -> None: + path = HYP / "HYPOTHESIS.json" + hypothesis = json.loads(path.read_text(encoding="utf-8")) + if hypothesis.get("acceptance_sha256") != EXPECTED_ACCEPTANCE: + raise SystemExit("hypothesis no longer points at the frozen acceptance") + if hypothesis.get("draft_hypothesis_sha256") != EXPECTED_DRAFT: + raise SystemExit("hypothesis no longer points at the frozen draft") + hypothesis["state"] = "MEASUREMENT_ELIGIBLE" + hypothesis["measurement_state"] = "MEASUREMENT_ELIGIBLE" + hypothesis["sample_state"] = "MEASUREMENT_SAMPLE_FROZEN" + hypothesis["measurement_eligible"] = True + hypothesis["measurement_sample_drawn"] = True + hypothesis["applied"] = True + hypothesis["applied_to_measurement"] = True + hypothesis["precision"] = "NOT_COMPUTABLE" + hypothesis["measurement_rows"] = freeze["rows"] + hypothesis["measurement_sample_sha256"] = freeze["sample_sha256"] + hypothesis["measurement_prediction_sha256"] = freeze["prediction_sha256"] + hypothesis["hand_corrections"] = 0 + hypothesis["v7_application_count"] = freeze["v7_application_count"] + hypothesis["operator_labels"] = None + hypothesis["labeling_performed"] = False + hypothesis["next_legal_transition"] = "OPERATOR_LABELS" + hypothesis["select_authorized"] = False + hypothesis["authorized"] = False + hypothesis["revision_eligible"] = False + hypothesis["admitted"] = 0 + hypothesis["settled"] = 0 + hypothesis["gold"] = 0 + text = json.dumps(hypothesis, indent=2, sort_keys=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + + +def _write_review_sheet(out_dir: Path, rows: list[dict], sample_sha: str) -> None: + lines = [ + "# Unbind screen v7 blind review", + "", + "Prediction buckets and evidence codes are withheld.", + f"Sample sha256 `{sample_sha}`.", + "", + "| row_id | surface | pos | token_count | gloss |", + "|---|---|---|---|---|", + ] + for row in rows: + gloss = row["gloss"].replace("|", "\\|") + lines.append( + f"| `{row['row_id']}` | {row['surface']} | {row['pos']} | {row['token_count']} | {gloss} |" + ) + lines.append("") + path = out_dir / "measurement_review.md" + path.write_text("\n".join(lines), encoding="utf-8") + path.chmod(0o600) + text = path.read_text(encoding="utf-8") + for forbidden in ("ordinary_compositional_derivation", "primary_evidence", "v7_bucket", "operator_bucket"): + if forbidden in text: + raise SystemExit(f"review sheet contains {forbidden}") + + +def _append_measurement_spec(freeze: dict, exhausted: list[dict]) -> None: + if exhausted: + detail = "; ".join( + f"{cell['source_pos']} x {cell['token_count']} available {cell['available']}, taken {cell['taken']}" + for cell in exhausted + ) + exhausted_sentence = f"Exhausted cells, where fewer than 2 rows remained: {detail}." + else: + exhausted_sentence = "No occupied cell was exhausted. Each occupied cell contributed 2 rows." + section = f""" +## Unbind screen v7 measurement frozen — 2026-09-28 + +The 197-row regression stays `REGRESSION_VERIFIED`. One unseen measurement sample excludes those 197 normalized identities. Normalized overlap with the reviewed set is 0. Duplicate normalized identities inside the sample are 0. The draw is deterministic and stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced {freeze['rows']} rows. {exhausted_sentence} + +The sample was frozen before v7 was applied. Sample sha256 `{freeze['sample_sha256']}`. v7 was then applied once. Hand corrections are 0. Prediction sha256 `{freeze['prediction_sha256']}`. The blind rows carry surface, part of speech, token count, and gloss. They do not carry a bucket or an evidence code. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. + +The acceptance bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor, no accuracy target, and no recall target were added. Criteria sha256 `{freeze['criteria_sha256']}`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `{EXPECTED_EVENTS}`. +""" + spec = SPEC.read_text(encoding="utf-8") + if not spec.endswith("\n"): + spec += "\n" + SPEC.write_text(spec + "\n" + section.lstrip("\n"), encoding="utf-8") + + +def main() -> None: + import argparse + + parser = argparse.ArgumentParser(description="Draw the unseen v7 measurement sample") + parser.add_argument("command", choices=("draw",)) + args = parser.parse_args() + if args.command != "draw": + raise SystemExit("only the measurement draw is authorized") + freeze = draw_measurement() + print(json.dumps({ + "state": freeze["state"], + "sample_state": freeze["sample_state"], + "rows": freeze["rows"], + "exhausted_strata": freeze["exhausted_strata"], + "sample_sha256": freeze["sample_sha256"], + "prediction_sha256": freeze["prediction_sha256"], + "v7_application_count": freeze["v7_application_count"], + "hand_corrections": freeze["hand_corrections"], + "normalized_overlap_with_reviewed_197": freeze["normalized_overlap_with_reviewed_197"], + "duplicate_normalized_identities": freeze["duplicate_normalized_identities"], + "precision": freeze["precision"], + "events_sha256": freeze["events_sha256"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py new file mode 100644 index 00000000..5f7ce781 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py @@ -0,0 +1,267 @@ +"""Sense-first unbind screen. + +The class comes from the frozen WordNet record of the supplied synset. +Membership in WordNet is not a class. Absence of a signal is not secondary. +""" + +from __future__ import annotations + +import re +from pathlib import Path + +RULE_VERSION = "RUNE.UNBIND_SENSE_SCREEN.v1" +_LIFESPAN = re.compile(r"\([0-9]{4}-[0-9]{4}\)") +_COMPARATIVE = re.compile(r"^used to form the comparative\b") +_SUPERLATIVE = re.compile(r"^used to form the superlative\b") +_LEXICAL_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +_RELATION_SYMBOLS = frozenset({"+", "\\"}) +_PREDICATE_TYPES = frozenset({"r", "a", "s"}) +_FILES = { + "noun": "data.noun", + "verb": "data.verb", + "adj": "data.adj", + "adv": "data.adv", +} +_EXC = { + "noun": "noun.exc", + "verb": "verb.exc", + "adj": "adj.exc", + "adv": "adv.exc", +} +_CODES = { + "REFERENTIAL": "referential_designation", + "ORDINARY_COMPOSITIONAL": "productive_grammatical_frame", + "LEXICALIZED_NONCOMPOSITIONAL": "noncompositional_semantic_mapping", + "LEXICALIZED_COMPOSITIONAL": "compositional_lexical_unit", + "AMBIGUOUS": "insufficient_record_evidence", +} +_BUCKETS = { + "REFERENTIAL": "REJECT", + "LEXICALIZED_NONCOMPOSITIONAL": "HIGH", + "LEXICALIZED_COMPOSITIONAL": "SECONDARY", + "ORDINARY_COMPOSITIONAL": "SECONDARY", + "AMBIGUOUS": "QUARANTINE", +} + + +class Pointer: + def __init__(self, symbol: str, offset: str, pos: str, source: int, target: int): + self.symbol = symbol + self.offset = offset + self.pos = pos + self.source = source + self.target = target + + +class Synset: + def __init__(self, offset: str, ss_type: str, lemmas: tuple[str, ...], pointers: tuple[Pointer, ...]): + self.offset = offset + self.ss_type = ss_type + self.lemmas = lemmas + self.pointers = pointers + + +def parse_data_line(line: str) -> tuple[Synset, str] | None: + """Return the synset and the first gloss clause. + + The word count and the source/target word numbers are hexadecimal. + The pointer count is a decimal integer. Verb frames follow the pointers + and are not pointers. + """ + if not line or line[0] == " ": + return None + meta, bar, gloss = line.partition("|") + if not bar: + return None + tokens = meta.split() + offset = tokens[0] + ss_type = tokens[2] + word_count = int(tokens[3], 16) + index = 4 + lemmas = [] + for _ in range(word_count): + lemmas.append(tokens[index]) + index += 2 + pointer_count = int(tokens[index], 10) + index += 1 + pointers = [] + for _ in range(pointer_count): + symbol = tokens[index] + target = tokens[index + 1] + pos = tokens[index + 2] + link = tokens[index + 3] + pointers.append(Pointer(symbol, target, pos, int(link[:2], 16), int(link[2:], 16))) + index += 4 + first = gloss.strip().split(";", 1)[0].strip() + return Synset(offset, ss_type, tuple(lemmas), tuple(pointers)), first + + +def load_exceptions(root: str | Path) -> dict[str, set[str]]: + linked: dict[str, set[str]] = {} + base = Path(root) + for name in _EXC.values(): + for line in (base / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parts = line.split() + if len(parts) < 2: + continue + left = parts[0].casefold() + right = parts[1].casefold() + linked.setdefault(left, set()).add(right) + linked.setdefault(right, set()).add(left) + return linked + + +def load_wordnet(root: str | Path) -> tuple[dict[tuple[str, str], Synset], dict[tuple[str, str], str]]: + synsets: dict[tuple[str, str], Synset] = {} + glosses: dict[tuple[str, str], str] = {} + base = Path(root) + for pos, name in _FILES.items(): + for line in (base / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parsed = parse_data_line(line) + if parsed is None: + continue + synset, gloss = parsed + for key in ((pos, synset.offset), (synset.ss_type, synset.offset)): + synsets[key] = synset + glosses[key] = gloss + return synsets, glosses + + +def classify(surface: str, gloss: str, synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> dict: + """Apply the frozen procedure once. The result is one class and one bucket.""" + text = gloss or "" + tokens = _tokens(surface) + ref_yes, ref_source = _referential_yes(text, synset) + ref_no = _referential_no(text, synset, ref_yes) + unrelated = _unrelated(tokens, synset, exceptions) + pointer = _lexical_pointer(tokens, synset) + alternation = _alternation(tokens, synset) + operator = _operator(text) + unit_yes = bool(unrelated) or pointer + unit_no = alternation or operator + constituent = _constituent_relation(tokens, synset, exceptions, targets) + # The constituent signal is defined only when no unrelated co-lemma is present, + # so those two signals do not fire together. + if (ref_yes and unit_no) or (unit_yes and unit_no): + return _emit("AMBIGUOUS", "none", insufficient=True) + _ = ref_no + if ref_yes: + return _emit("REFERENTIAL", ref_source, insufficient=False) + if unit_no: + source = "synset.lemmas.productive_alternation" if alternation else "synset.gloss.grammatical_operator" + return _emit("ORDINARY_COMPOSITIONAL", source, insufficient=False) + if not unit_yes: + return _emit("AMBIGUOUS", "none", insufficient=True) + if unrelated: + return _emit("LEXICALIZED_NONCOMPOSITIONAL", "synset.lemmas.unrelated_single_word", insufficient=False) + if constituent: + return _emit("LEXICALIZED_COMPOSITIONAL", "synset.lexical_pointer.derivation_or_pertainym_to_constituent", insufficient=False) + return _emit("AMBIGUOUS", "none", insufficient=True) + + +def _emit(sense_class: str, source: str, *, insufficient: bool) -> dict: + return { + "rule": RULE_VERSION, + "sense_class": sense_class, + "bucket": _BUCKETS[sense_class], + "primary_evidence_code": _CODES[sense_class], + "evidence_source": source, + "confidence_status": "INSUFFICIENT" if insufficient else "DETERMINATE", + } + + +def _tokens(text: str) -> tuple[str, ...]: + return tuple(_norm(text).split()) + + +def _norm(text: str) -> str: + return " ".join(text.replace("_", " ").replace("-", " ").casefold().split()) + + +def _referential_yes(gloss: str, synset: Synset) -> tuple[bool, str]: + if any(pointer.symbol == "@i" for pointer in synset.pointers): + return True, "synset.instance_hypernym" + if _LIFESPAN.search(gloss): + return True, "synset.gloss.lifespan" + return False, "none" + + +def _referential_no(gloss: str, synset: Synset, ref_yes: bool) -> bool: + if ref_yes: + return False + return synset.ss_type in _PREDICATE_TYPES or _operator(gloss) + + +def _operator(gloss: str) -> bool: + return bool(_COMPARATIVE.match(gloss) or _SUPERLATIVE.match(gloss)) + + +def _surface_index(tokens: tuple[str, ...], synset: Synset) -> int: + wanted = " ".join(tokens) + for index, lemma in enumerate(synset.lemmas, start=1): + if _norm(lemma) == wanted: + return index + return 0 + + +def _unrelated(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]]) -> bool: + owned = set(tokens) + wanted = " ".join(tokens) + for lemma in synset.lemmas: + if _norm(lemma) == wanted: + continue + parts = _tokens(lemma) + if len(parts) != 1: + continue + word = parts[0] + if word in owned or _linked(word, owned, exceptions): + continue + return True + return False + + +def _linked(word: str, tokens: set[str], exceptions: dict[str, set[str]]) -> bool: + related = exceptions.get(word, set()) + for token in tokens: + if token in related or word in exceptions.get(token, set()): + return True + return False + + +def _lexical_pointer(tokens: tuple[str, ...], synset: Synset) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + return any(pointer.source == source and pointer.symbol in _LEXICAL_SYMBOLS for pointer in synset.pointers) + + +def _alternation(tokens: tuple[str, ...], synset: Synset) -> bool: + groups = [_tokens(lemma) for lemma in synset.lemmas if len(_tokens(lemma)) >= 2] + for left in range(len(groups)): + for right in range(left + 1, len(groups)): + one = groups[left] + other = groups[right] + if len(one) != len(other): + continue + if sum(token != sibling for token, sibling in zip(one, other)) != 1: + continue + if one == tokens or other == tokens: + return True + return False + + +def _constituent_relation(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + owned = set(tokens) + for pointer in synset.pointers: + if pointer.source != source or pointer.symbol not in _RELATION_SYMBOLS or pointer.target < 1: + continue + lemmas = targets.get((pointer.pos, pointer.offset), ()) + if pointer.target > len(lemmas): + continue + parts = _tokens(lemmas[pointer.target - 1]) + if len(parts) == 1 and (parts[0] in owned or _linked(parts[0], owned, exceptions)): + return True + return False diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py new file mode 100644 index 00000000..c8dd4ffe --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py @@ -0,0 +1,308 @@ +"""Replay the frozen 225-row development manifest through the encoded sense screen. + +The replay is development evidence. It does not draw a measurement sample, +revise the frozen procedure, or write a sense class back onto the manifest. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.screen_eval import _error_class, _metrics +from hyperlexical.unbind_sense_screen_v1 import classify, load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +PREDICTIONS = OUT / "development_replay_predictions.jsonl" +REPORT = OUT / "development_replay_report.json" +TRACKER = OUT / "HYPOTHESIS.json" +EVIDENCE = OUT / "DEVELOPMENT_EVIDENCE.json" +PROCEDURE = OUT / "CLASSIFICATION_PROCEDURE.json" + +EXPECTED = { + OUT / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + OUT / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + OUT / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + PROCEDURE: "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + TRACKER: "f9c4757b6eec558b5e1baf644bcf33c27c949807e7f00cd15df869eb6411de31", + LEDGER / "events.jsonl": "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", +} +SENSE_CLASSES = ( + "REFERENTIAL", + "LEXICALIZED_NONCOMPOSITIONAL", + "LEXICALIZED_COMPOSITIONAL", + "ORDINARY_COMPOSITIONAL", + "AMBIGUOUS", +) +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str | None: + if denominator == 0: + return None + return f"{numerator}/{denominator}" + + +def main() -> None: + for path, expected in EXPECTED.items(): + found = sha256(path) + if found != expected: + refuse(f"hash mismatch {path.name}: {found}") + if PREDICTIONS.exists() or REPORT.exists(): + refuse("development replay artifacts already exist") + evidence = json.loads(EVIDENCE.read_text(encoding="utf-8")) + if evidence.get("sense_classes_assigned") is not False: + refuse("development manifest already records assigned sense classes") + rows = evidence["rows"] + if len(rows) != 225 or evidence.get("row_count") != 225: + refuse("development manifest is not the frozen 225-row inventory") + if len({row["row_id"] for row in rows}) != 225: + refuse("development row ids are not unique") + if any(row.get("sense_class") is not None for row in rows): + refuse("a development row already has a sense class") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if tracker.get("state") != "PROCEDURE_FROZEN" or tracker.get("encoded") is not False: + refuse("tracker is not waiting at PROCEDURE_FROZEN") + procedure = json.loads(PROCEDURE.read_text(encoding="utf-8")) + if procedure.get("encoded") is not False or procedure.get("development_replay_run") is not False: + refuse("frozen procedure artifact is not in its sealed unencoded state") + + synsets, glosses = load_wordnet(WORDNET) + exceptions = load_exceptions(WORDNET) + targets = {key: synset.lemmas for key, synset in synsets.items()} + classified = [] + for row in rows: + key = (row["synset_pos"], row["synset_offset"]) + synset = synsets.get(key) + gloss = glosses.get(key) + if synset is None or gloss is None: + refuse(f"missing synset {row['synset_pos']} {row['synset_offset']}") + if gloss != row["gloss"]: + refuse(f"frozen gloss does not match the data-line first clause for {row['row_id']}") + decision = classify(row["surface"], row["gloss"], synset, exceptions, targets) + if decision["sense_class"] not in SENSE_CLASSES or decision["bucket"] not in BUCKETS: + refuse(f"classifier returned an unknown class for {row['row_id']}") + classified.append((row, decision)) + + predictions = [] + for row, decision in sorted(classified, key=lambda item: item[0]["row_id"]): + predictions.append( + { + "schema": "hyperlex.unbind_sense_screen_v1_development_row.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "application_index": 1, + "row_id": row["row_id"], + "surface": row["surface"], + "pos": row["pos"], + "synset": f"{row['synset_pos']}:{row['synset_offset']}", + "synset_offset": row["synset_offset"], + "synset_pos": row["synset_pos"], + "sense_class": decision["sense_class"], + "bucket": decision["bucket"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_source": decision["evidence_source"], + "confidence_status": decision["confidence_status"], + } + ) + if any("operator_bucket" in row or "historical_unbind_screen" in row for row in predictions): + refuse("prediction rows carry operator or historical fields") + prediction_sha = write_jsonl(PREDICTIONS, predictions) + + by_id = {row["row_id"]: decision for row, decision in classified} + ordered_rows = sorted(rows, key=lambda row: row["row_id"]) + pairs = [] + sense_by_operator = {op: {sense: 0 for sense in SENSE_CLASSES} for op in OPERATOR_BUCKETS} + error_rows = {"false_high": [], "false_reject": [], "false_secondary": [], "false_quarantine": []} + for row in ordered_rows: + decision = by_id[row["row_id"]] + operator = row["operator_bucket"] + pairs.append((row["row_id"], decision["bucket"], operator, decision["primary_evidence_code"])) + sense_by_operator[operator][decision["sense_class"]] += 1 + kind = _error_class(decision["bucket"], operator) + if kind: + error_rows[kind].append( + { + "row_id": row["row_id"], + "surface": row["surface"], + "operator_bucket": operator, + "bucket": decision["bucket"], + "sense_class": decision["sense_class"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_source": decision["evidence_source"], + } + ) + metrics = _metrics(pairs) + sense_counts = Counter(row["sense_class"] for row in predictions) + bucket_counts = Counter(row["bucket"] for row in predictions) + evidence_counts = Counter(row["primary_evidence_code"] for row in predictions) + source_counts = Counter(row["evidence_source"] for row in predictions) + confidence_counts = Counter(row["confidence_status"] for row in predictions) + ambiguous = sense_counts["AMBIGUOUS"] + + def bucket_block(name: str) -> dict: + stats = metrics["per_bucket"][name] + support = stats["support"] + predicted = bucket_counts[name] + true_positive = sum(1 for _i, pred, op, _r in pairs if pred == name and op == name) + return { + "precision": stats["precision"], + "precision_fraction": fraction(true_positive, predicted), + "recall": stats["recall"], + "recall_fraction": fraction(true_positive, support), + "operator_support": support, + "predicted": predicted, + } + + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + report = { + "schema": "hyperlex.unbind_sense_screen_v1_development_replay.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "role": "development evidence, not validation", + "not_a_validation_set": True, + "tuning_on_these_rows_is_not_validation": True, + "state": "DEVELOPMENT_ANALYZED", + "state_path": [ + "PROCEDURE_FROZEN", + "ENCODE_AUTHORIZED", + "ENCODED", + "225_ROW_DEVELOPMENT_REPLAY", + "DEVELOPMENT_ANALYZED", + ], + "analyzed_at": when, + "rows": 225, + "application_index": 1, + "applications_per_row": 1, + "prediction_sha256": prediction_sha, + "procedure_sha256": EXPECTED[PROCEDURE], + "procedure_mutated": False, + "acceptance_sha256": EXPECTED[OUT / "ACCEPTANCE.json"], + "acceptance_mutated": False, + "draft_hypothesis_sha256": EXPECTED[OUT / "HYPOTHESIS.draft.json"], + "development_evidence_sha256": EXPECTED[EVIDENCE], + "development_manifest_sense_classes_written": False, + "lexeme_architecture_sha256": EXPECTED[OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json"], + "lexeme_architecture_mutated": False, + "lineage_retirement_sha256": EXPECTED[OUT / "LINEAGE_RETIREMENT.json"], + "events_sha256": EXPECTED[LEDGER / "events.jsonl"], + "sense_class_counts": {name: sense_counts[name] for name in SENSE_CLASSES}, + "bucket_counts": {name: bucket_counts[name] for name in BUCKETS}, + "operator_vs_bucket_confusion": metrics["confusion_matrix"], + "operator_vs_sense_class": sense_by_operator, + "ambiguous_count": ambiguous, + "ambiguous_rate": ambiguous / 225, + "ambiguous_rate_fraction": fraction(ambiguous, 225), + "ambiguous_rate_is_expected_measurement": True, + "high_ambiguity_was_not_repaired": True, + "absence_of_evidence_is_not_secondary": True, + "evidence_code_counts": dict(sorted(evidence_counts.items())), + "evidence_source_counts": dict(sorted(source_counts.items())), + "confidence_status_counts": dict(sorted(confidence_counts.items())), + "HIGH": bucket_block("HIGH"), + "SECONDARY": bucket_block("SECONDARY"), + "REJECT": bucket_block("REJECT"), + "QUARANTINE": bucket_block("QUARANTINE"), + "quarantine_support": metrics["per_bucket"]["QUARANTINE"]["support"], + "false_high": len(error_rows["false_high"]), + "false_reject": len(error_rows["false_reject"]), + "false_secondary": len(error_rows["false_secondary"]), + "false_quarantine": len(error_rows["false_quarantine"]), + "false_secondary_is_descriptive_only": True, + "false_high_rows": error_rows["false_high"], + "false_reject_rows": error_rows["false_reject"], + "outer_bucket_metrics_are_descriptive": True, + "measurement_bar_applied": False, + "measurement_sample_drawn": False, + "measurement_eligible": False, + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "ledger_appended": False, + "next_legal_transition": "MEASUREMENT_AUTHORIZATION", + "next_transition_authorized": False, + } + report_sha = write_json(REPORT, report) + + untouched = json.loads(EVIDENCE.read_text(encoding="utf-8")) + if any(row.get("sense_class") is not None for row in untouched["rows"]): + refuse("replay wrote a sense class onto the development manifest") + if sha256(EVIDENCE) != EXPECTED[EVIDENCE] or sha256(PROCEDURE) != EXPECTED[PROCEDURE]: + refuse("replay mutated a sealed artifact") + + tracker["previous_state"] = "PROCEDURE_FROZEN" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "DEVELOPMENT_ANALYZED" + tracker["encoded"] = True + tracker["encoding_authorized"] = True + tracker["applied"] = True + tracker["development_replay_run"] = True + tracker["rows_classified"] = 225 + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["next_legal_transition"] = "MEASUREMENT_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["prediction_sha256"] = prediction_sha + tracker["report_sha256"] = report_sha + tracker["ambiguous_rate_fraction"] = report["ambiguous_rate_fraction"] + tracker["high_ambiguity_was_not_repaired"] = True + tracker_sha = write_json(TRACKER, tracker) + print(json.dumps({ + "prediction_sha256": prediction_sha, + "report_sha256": report_sha, + "tracker_sha256": tracker_sha, + "sense_class_counts": report["sense_class_counts"], + "bucket_counts": report["bucket_counts"], + "ambiguous_rate_fraction": report["ambiguous_rate_fraction"], + "HIGH": report["HIGH"], + "SECONDARY": report["SECONDARY"], + "REJECT": report["REJECT"], + "QUARANTINE": report["QUARANTINE"], + "false_high": report["false_high"], + "false_reject": report["false_reject"], + "false_secondary": report["false_secondary"], + "evidence_code_counts": report["evidence_code_counts"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py new file mode 100644 index 00000000..2a804a5d --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py @@ -0,0 +1,295 @@ +"""Sense screen procedure v2. + +Three states are recorded before the class. A whole-expression co-lemma +is lexicalization. It is not a compositional no, and it is not high. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_sense_screen_v1 import ( + Pointer, + Synset, + load_exceptions, + load_wordnet, + parse_data_line, +) + +RULE_VERSION = "RUNE.UNBIND_SENSE_SCREEN.v1" +PROCEDURE = "hyperlex.unbind_sense_screen_v1_classification_procedure.v2" +_LIFESPAN = re.compile(r"\([0-9]{4}-[0-9]{4}\)") +_COMPARATIVE = re.compile(r"^used to form the comparative\b") +_SUPERLATIVE = re.compile(r"^used to form the superlative\b") +_LEXICAL_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +_RELATION_SYMBOLS = frozenset({"+", "\\"}) +_PREDICATE_TYPES = frozenset({"r", "a", "s"}) +_BUCKETS = { + "REFERENTIAL": "REJECT", + "LEXICALIZED_NONCOMPOSITIONAL": "HIGH", + "LEXICALIZED_COMPOSITIONAL": "SECONDARY", + "ORDINARY_COMPOSITIONAL": "SECONDARY", + "AMBIGUOUS": "QUARANTINE", +} +_FAMILY_ORDER = ( + "referential_designation", + "whole_expression_lexicalization", + "compositional_semantic_relation", + "productive_grammatical_frame", + "noncompositional_semantic_mapping", + "insufficient_record_evidence", + "conflicting_record_evidence", +) +_SOURCE_ORDER = ( + "synset.instance_hypernym", + "synset.gloss.lifespan", + "synset.ss_type.predicate", + "synset.lemmas.unrelated_single_word", + "synset.lexical_pointer", + "synset.gloss.grammatical_operator", + "synset.lexical_pointer.derivation_or_pertainym_to_constituent", + "synset.lemmas.productive_alternation", +) + +__all__ = [ + "Pointer", + "Synset", + "classify", + "load_exceptions", + "load_wordnet", + "parse_data_line", +] + + +def classify(surface: str, gloss: str, synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> dict: + """Apply procedure v2 once. Compositional NO is not produced.""" + text = gloss or "" + tokens = _tokens(surface) + referential, referential_hits = _referential(text, synset) + lexicalized, lexicalized_hits = _lexicalized(tokens, synset, exceptions, text) + compositional, compositional_hits = _compositional(tokens, synset, exceptions, targets, text) + if compositional in {"NO", "CONFLICT"}: + raise RuntimeError("procedure v2 has no compositional NO signal") + hits = referential_hits + lexicalized_hits + compositional_hits + if referential == "CONFLICT" or (referential != "YES" and lexicalized == "CONFLICT"): + return _decision( + "AMBIGUOUS", + "conflicting_record_evidence", + "CONTRADICTORY", + referential, + lexicalized, + compositional, + hits, + None, + ) + if referential == "YES": + return _decision( + "REFERENTIAL", + "referential_designation", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + referential_hits[0][0], + ) + if lexicalized == "YES" and compositional == "NO": + return _decision( + "LEXICALIZED_NONCOMPOSITIONAL", + "noncompositional_semantic_mapping", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + None, + ) + if lexicalized == "YES" and compositional == "YES": + source, family = compositional_hits[0] + return _decision( + "LEXICALIZED_COMPOSITIONAL", + family, + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + source, + ) + if lexicalized == "NO" and compositional == "YES": + return _decision( + "ORDINARY_COMPOSITIONAL", + "productive_grammatical_frame", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + "synset.gloss.grammatical_operator", + ) + return _decision( + "AMBIGUOUS", + "insufficient_record_evidence", + "INSUFFICIENT", + referential, + lexicalized, + compositional, + hits, + None, + ) + + +def _decision(sense_class, primary, confidence, referential, lexicalized, compositional, hits, deciding): + families = [] + fired = [] + for source, family in hits: + if source not in fired: + fired.append(source) + if family and family not in families: + families.append(family) + sources = [name for name in _SOURCE_ORDER if name in fired and name != deciding] + if deciding: + sources.insert(0, deciding) + if not sources: + sources = ["none"] + supporting = [name for name in _FAMILY_ORDER if name in families and name != primary] + return { + "rule": RULE_VERSION, + "procedure": PROCEDURE, + "referential_state": referential, + "lexicalized_state": lexicalized, + "compositional_state": compositional, + "sense_class": sense_class, + "bucket": _BUCKETS[sense_class], + "primary_evidence_code": primary, + "supporting_evidence_codes": supporting, + "evidence_sources": sources, + "confidence_status": confidence, + } + + +def _referential(gloss: str, synset: Synset): + if any(pointer.symbol == "@i" for pointer in synset.pointers): + return "YES", [("synset.instance_hypernym", "referential_designation")] + if _LIFESPAN.search(gloss): + return "YES", [("synset.gloss.lifespan", "referential_designation")] + hits = [] + if synset.ss_type in _PREDICATE_TYPES: + hits.append(("synset.ss_type.predicate", None)) + if _operator(gloss): + hits.append(("synset.gloss.grammatical_operator", "productive_grammatical_frame")) + if hits: + return "NO", hits + return "UNKNOWN", [] + + +def _lexicalized(tokens, synset: Synset, exceptions, gloss: str): + hits = [] + if _unrelated(tokens, synset, exceptions): + hits.append(("synset.lemmas.unrelated_single_word", "whole_expression_lexicalization")) + if _lexical_pointer(tokens, synset): + hits.append(("synset.lexical_pointer", "whole_expression_lexicalization")) + operator = [("synset.gloss.grammatical_operator", "productive_grammatical_frame")] if _operator(gloss) else [] + if hits and operator: + return "CONFLICT", hits + operator + if hits: + return "YES", hits + if operator: + return "NO", operator + return "UNKNOWN", [] + + +def _compositional(tokens, synset: Synset, exceptions, targets, gloss: str): + hits = [] + if _constituent_relation(tokens, synset, exceptions, targets): + hits.append(("synset.lexical_pointer.derivation_or_pertainym_to_constituent", "compositional_semantic_relation")) + if _alternation(tokens, synset): + hits.append(("synset.lemmas.productive_alternation", "productive_grammatical_frame")) + if _operator(gloss): + hits.append(("synset.gloss.grammatical_operator", "productive_grammatical_frame")) + if hits: + return "YES", hits + return "UNKNOWN", [] + + +def _tokens(text: str) -> tuple[str, ...]: + return tuple(_norm(text).split()) + + +def _norm(text: str) -> str: + return " ".join(text.replace("_", " ").replace("-", " ").casefold().split()) + + +def _operator(gloss: str) -> bool: + return bool(_COMPARATIVE.match(gloss or "") or _SUPERLATIVE.match(gloss or "")) + + +def _surface_index(tokens: tuple[str, ...], synset: Synset) -> int: + wanted = " ".join(tokens) + for index, lemma in enumerate(synset.lemmas, start=1): + if _norm(lemma) == wanted: + return index + return 0 + + +def _unrelated(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]]) -> bool: + owned = set(tokens) + wanted = " ".join(tokens) + for lemma in synset.lemmas: + if _norm(lemma) == wanted: + continue + parts = _tokens(lemma) + if len(parts) != 1: + continue + word = parts[0] + if word in owned or _linked(word, owned, exceptions): + continue + return True + return False + + +def _linked(word: str, tokens: set[str], exceptions: dict[str, set[str]]) -> bool: + related = exceptions.get(word, set()) + for token in tokens: + if token in related or word in exceptions.get(token, set()): + return True + return False + + +def _lexical_pointer(tokens: tuple[str, ...], synset: Synset) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + return any(pointer.source == source and pointer.symbol in _LEXICAL_SYMBOLS for pointer in synset.pointers) + + +def _alternation(tokens: tuple[str, ...], synset: Synset) -> bool: + groups = [_tokens(lemma) for lemma in synset.lemmas if len(_tokens(lemma)) >= 2] + for left in range(len(groups)): + for right in range(left + 1, len(groups)): + one = groups[left] + other = groups[right] + if len(one) != len(other): + continue + if sum(token != sibling for token, sibling in zip(one, other)) != 1: + continue + if one == tokens or other == tokens: + return True + return False + + +def _constituent_relation(tokens, synset: Synset, exceptions, targets) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + owned = set(tokens) + for pointer in synset.pointers: + if pointer.source != source or pointer.symbol not in _RELATION_SYMBOLS or pointer.target < 1: + continue + lemmas = targets.get((pointer.pos, pointer.offset), ()) + if pointer.target > len(lemmas): + continue + parts = _tokens(lemmas[pointer.target - 1]) + if len(parts) == 1 and (parts[0] in owned or _linked(parts[0], owned, exceptions)): + return True + return False diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py new file mode 100644 index 00000000..19501fc8 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v2_replay.py @@ -0,0 +1,362 @@ +"""Replay the frozen 225-row development manifest through procedure v2. + +The replay is a development experiment. It does not draw a measurement sample, +restore the co-lemma shortcut, or write a class back onto the manifest. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.screen_eval import _error_class, _metrics +from hyperlexical.unbind_sense_screen_v2 import classify, load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +PREDICTIONS = OUT / "development_replay_v2_predictions.jsonl" +REPORT = OUT / "development_replay_v2_report.json" +TRACKER = OUT / "HYPOTHESIS.json" +EVIDENCE = OUT / "DEVELOPMENT_EVIDENCE.json" +PROCEDURE_V1 = OUT / "CLASSIFICATION_PROCEDURE.json" +PROCEDURE_V2 = OUT / "CLASSIFICATION_PROCEDURE.v2.json" +V1_PREDICTIONS = OUT / "development_replay_predictions.jsonl" +V1_REPORT = OUT / "development_replay_report.json" + +EXPECTED = { + PROCEDURE_V1: "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + PROCEDURE_V2: "3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662", + OUT / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + OUT / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + OUT / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + OUT / "PROCEDURE_V1_ERROR_ANALYSIS.json": "471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e", + OUT / "PROCEDURE_V1_TO_V2_CHANGE_NOTE.json": "443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c", + V1_PREDICTIONS: "69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef", + V1_REPORT: "38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0", + TRACKER: "af11d20ebefec5629718617ebacc07d8b4cc36c61e59e829d153b05b9397ca40", + LEDGER / "events.jsonl": "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_sense_screen_v1.py": "531b58422e6f18b42276c6dde36493c7d0f8841556785b4b8911017879f93ad0", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", +} +SENSE_CLASSES = ( + "REFERENTIAL", + "LEXICALIZED_NONCOMPOSITIONAL", + "LEXICALIZED_COMPOSITIONAL", + "ORDINARY_COMPOSITIONAL", + "AMBIGUOUS", +) +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) +STATES = ("YES", "NO", "UNKNOWN", "CONFLICT") + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str | None: + if denominator == 0: + return None + return f"{numerator}/{denominator}" + + +def main() -> None: + for path, expected in EXPECTED.items(): + found = sha256(path) + if found != expected: + refuse(f"hash mismatch {path.name}: {found}") + if PREDICTIONS.exists() or REPORT.exists(): + refuse("procedure v2 replay artifacts already exist") + evidence = json.loads(EVIDENCE.read_text(encoding="utf-8")) + rows = evidence["rows"] + if len(rows) != 225 or any(row.get("sense_class") is not None for row in rows): + refuse("development manifest is not the unclassified 225-row inventory") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if tracker.get("state") != "DEVELOPMENT_ANALYZED" or tracker.get("procedure_v2_state") != "PROCEDURE_V2_FROZEN": + refuse("tracker is not waiting at PROCEDURE_V2_FROZEN") + if tracker.get("procedure_v2_encoded") is not False or tracker.get("measurement_sample_drawn") is not False: + refuse("procedure v2 is already encoded or a sample exists") + procedure = json.loads(PROCEDURE_V2.read_text(encoding="utf-8")) + if procedure.get("encoded") is not False or procedure.get("state") != "PROCEDURE_V2_FROZEN": + refuse("procedure v2 artifact is not the sealed unencoded freeze") + + synsets, glosses = load_wordnet(WORDNET) + exceptions = load_exceptions(WORDNET) + targets = {key: synset.lemmas for key, synset in synsets.items()} + classified = [] + for row in rows: + key = (row["synset_pos"], row["synset_offset"]) + synset = synsets.get(key) + gloss = glosses.get(key) + if synset is None or gloss is None: + refuse(f"missing synset {row['synset_pos']} {row['synset_offset']}") + if gloss != row["gloss"]: + refuse(f"frozen gloss does not match the data-line first clause for {row['row_id']}") + decision = classify(row["surface"], row["gloss"], synset, exceptions, targets) + if decision["sense_class"] not in SENSE_CLASSES or decision["compositional_state"] == "NO": + refuse(f"classifier left the frozen catalog for {row['row_id']}") + if decision["bucket"] == "HIGH" or decision["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL": + refuse("encoder emitted HIGH from the empty compositional NO catalog") + classified.append((row, decision)) + + by_surface = {row["surface"]: decision for row, decision in classified} + alternation = by_surface.get("as far as possible") + if alternation is None or ( + alternation["referential_state"], + alternation["lexicalized_state"], + alternation["compositional_state"], + alternation["sense_class"], + ) != ("NO", "UNKNOWN", "YES", "AMBIGUOUS"): + refuse("as far as possible did not follow the frozen v2 illustration") + damascus = by_surface.get("road to damascus") + if damascus is None or ( + damascus["referential_state"], + damascus["lexicalized_state"], + damascus["compositional_state"], + damascus["sense_class"], + ) != ("UNKNOWN", "UNKNOWN", "UNKNOWN", "AMBIGUOUS"): + refuse("road to damascus did not follow the frozen v2 illustration") + + predictions = [] + for row, decision in sorted(classified, key=lambda item: item[0]["row_id"]): + predictions.append( + { + "schema": "hyperlex.unbind_sense_screen_v2_development_row.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "procedure": "hyperlex.unbind_sense_screen_v1_classification_procedure.v2", + "application_index": 1, + "row_id": row["row_id"], + "surface": row["surface"], + "pos": row["pos"], + "synset": f"{row['synset_pos']}:{row['synset_offset']}", + "synset_offset": row["synset_offset"], + "synset_pos": row["synset_pos"], + "referential_state": decision["referential_state"], + "lexicalized_state": decision["lexicalized_state"], + "compositional_state": decision["compositional_state"], + "sense_class": decision["sense_class"], + "bucket": decision["bucket"], + "primary_evidence_code": decision["primary_evidence_code"], + "supporting_evidence_codes": decision["supporting_evidence_codes"], + "evidence_sources": decision["evidence_sources"], + "confidence_status": decision["confidence_status"], + } + ) + prediction_sha = write_jsonl(PREDICTIONS, predictions) + + by_id = {row["row_id"]: decision for row, decision in classified} + ordered_rows = sorted(rows, key=lambda row: row["row_id"]) + pairs = [] + sense_by_operator = {op: {sense: 0 for sense in SENSE_CLASSES} for op in OPERATOR_BUCKETS} + error_rows = {name: [] for name in ("false_high", "false_reject", "false_secondary", "false_quarantine")} + for row in ordered_rows: + decision = by_id[row["row_id"]] + operator = row["operator_bucket"] + pairs.append((row["row_id"], decision["bucket"], operator, decision["primary_evidence_code"])) + sense_by_operator[operator][decision["sense_class"]] += 1 + kind = _error_class(decision["bucket"], operator) + if kind: + error_rows[kind].append( + { + "row_id": row["row_id"], + "surface": row["surface"], + "operator_bucket": operator, + "bucket": decision["bucket"], + "sense_class": decision["sense_class"], + "referential_state": decision["referential_state"], + "lexicalized_state": decision["lexicalized_state"], + "compositional_state": decision["compositional_state"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_sources": decision["evidence_sources"], + } + ) + metrics = _metrics(pairs) + sense_counts = Counter(row["sense_class"] for row in predictions) + bucket_counts = Counter(row["bucket"] for row in predictions) + evidence_counts = Counter(row["primary_evidence_code"] for row in predictions) + state_counts = { + "referential": Counter(row["referential_state"] for row in predictions), + "lexicalized": Counter(row["lexicalized_state"] for row in predictions), + "compositional": Counter(row["compositional_state"] for row in predictions), + } + triples = Counter( + f"{row['referential_state']}|{row['lexicalized_state']}|{row['compositional_state']}" + for row in predictions + ) + + def bucket_block(name: str) -> dict: + stats = metrics["per_bucket"][name] + support = stats["support"] + predicted = bucket_counts[name] + true_positive = sum(1 for _i, pred, op, _r in pairs if pred == name and op == name) + return { + "precision": stats["precision"], + "precision_fraction": fraction(true_positive, predicted), + "recall": stats["recall"], + "recall_fraction": fraction(true_positive, support), + "operator_support": support, + "predicted": predicted, + } + + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + report = { + "schema": "hyperlex.unbind_sense_screen_v2_development_replay.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "procedure": "hyperlex.unbind_sense_screen_v1_classification_procedure.v2", + "role": "development experiment, not validation", + "question": "Can the frozen WordNet-only evidence model instantiate the intended classes, especially LEXICALIZED_NONCOMPOSITIONAL, without the co-lemma shortcut?", + "not_a_validation_set": True, + "state": "DEVELOPMENT_ANALYZED_V2", + "state_path": [ + "PROCEDURE_V2_FROZEN", + "ENCODE_PROCEDURE_V2_AUTHORIZATION", + "ENCODED", + "225_ROW_DEVELOPMENT_REPLAY", + "DEVELOPMENT_ANALYZED_V2", + ], + "analyzed_at": when, + "rows": 225, + "applications_per_row": 1, + "prediction_sha256": prediction_sha, + "procedure_v2_sha256": EXPECTED[PROCEDURE_V2], + "procedure_v2_mutated": False, + "procedure_v1_sha256": EXPECTED[PROCEDURE_V1], + "procedure_v1_mutated": False, + "acceptance_sha256": EXPECTED[OUT / "ACCEPTANCE.json"], + "development_evidence_sha256": EXPECTED[EVIDENCE], + "development_manifest_sense_classes_written": False, + "v1_prediction_sha256": EXPECTED[V1_PREDICTIONS], + "v1_report_sha256": EXPECTED[V1_REPORT], + "lexeme_architecture_sha256": EXPECTED[OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json"], + "lexeme_architecture_mutated": False, + "events_sha256": EXPECTED[LEDGER / "events.jsonl"], + "sense_class_counts": {name: sense_counts[name] for name in SENSE_CLASSES}, + "bucket_counts": {name: bucket_counts[name] for name in BUCKETS}, + "evidence_state_counts": { + dimension: {state: counts[state] for state in STATES} + for dimension, counts in state_counts.items() + }, + "evidence_state_triples": dict(sorted(triples.items())), + "operator_vs_bucket_confusion": metrics["confusion_matrix"], + "operator_vs_sense_class": sense_by_operator, + "ambiguous_count": sense_counts["AMBIGUOUS"], + "ambiguous_rate_fraction": fraction(sense_counts["AMBIGUOUS"], 225), + "ambiguous_rate_is_not_a_success_criterion": True, + "high_support": bucket_counts["HIGH"], + "high_support_is_zero": bucket_counts["HIGH"] == 0, + "lexicalized_noncompositional_support": sense_counts["LEXICALIZED_NONCOMPOSITIONAL"], + "lexicalized_noncompositional_instantiated": sense_counts["LEXICALIZED_NONCOMPOSITIONAL"] > 0, + "colemma_shortcut_restored": False, + "compositional_no_emitted": state_counts["compositional"]["NO"], + "evidence_code_counts": dict(sorted(evidence_counts.items())), + "HIGH": bucket_block("HIGH"), + "SECONDARY": bucket_block("SECONDARY"), + "REJECT": bucket_block("REJECT"), + "QUARANTINE": bucket_block("QUARANTINE"), + "referential_precision_fraction": bucket_block("REJECT")["precision_fraction"], + "referential_recall_fraction": bucket_block("REJECT")["recall_fraction"], + "lexicalized_compositional_support": sense_counts["LEXICALIZED_COMPOSITIONAL"], + "lexicalized_compositional_support_is_a_readiness_signal_only": True, + "false_high": len(error_rows["false_high"]), + "false_reject": len(error_rows["false_reject"]), + "false_secondary": len(error_rows["false_secondary"]), + "false_quarantine": len(error_rows["false_quarantine"]), + "false_high_rows": error_rows["false_high"], + "false_reject_rows": error_rows["false_reject"], + "diagnostics_are_not_pass_fail": True, + "measurement_bar_applied": False, + "measurement_sample_drawn": False, + "measurement_eligible": False, + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "ledger_appended": False, + "next_legal_transition": "V2_DEVELOPMENT_RESULT_REVIEW", + "next_transition_authorized": False, + } + report_sha = write_json(REPORT, report) + if any(row.get("sense_class") is not None for row in json.loads(EVIDENCE.read_text(encoding="utf-8"))["rows"]): + refuse("replay wrote a sense class onto the development manifest") + for path, expected in EXPECTED.items(): + if path == TRACKER: + continue + if sha256(path) != expected: + refuse(f"replay mutated {path.name}") + + tracker["previous_state"] = "DEVELOPMENT_ANALYZED" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["v1_screen_state"] = "DEVELOPMENT_ANALYZED" + tracker["state"] = "DEVELOPMENT_ANALYZED_V2" + tracker["procedure_v2_state"] = "DEVELOPMENT_ANALYZED_V2" + tracker["procedure_v2_encoded"] = True + tracker["procedure_v2_encoding_authorized"] = True + tracker["procedure_v2_applied"] = True + tracker["procedure_v2_development_replay_run"] = True + tracker["procedure_v2_rows_classified"] = 225 + tracker["procedure_v2_prediction_sha256"] = prediction_sha + tracker["procedure_v2_report_sha256"] = report_sha + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["next_legal_transition"] = "V2_DEVELOPMENT_RESULT_REVIEW" + tracker["next_transition_authorized"] = False + tracker["high_support_v2"] = bucket_counts["HIGH"] + tracker["lexicalized_noncompositional_support_v2"] = sense_counts["LEXICALIZED_NONCOMPOSITIONAL"] + tracker_sha = write_json(TRACKER, tracker) + if sha256(PROCEDURE_V2) != EXPECTED[PROCEDURE_V2] or sha256(EVIDENCE) != EXPECTED[EVIDENCE]: + refuse("tracker update mutated a sealed artifact") + print(json.dumps({ + "prediction_sha256": prediction_sha, + "report_sha256": report_sha, + "tracker_sha256": tracker_sha, + "sense_class_counts": report["sense_class_counts"], + "bucket_counts": report["bucket_counts"], + "evidence_state_counts": report["evidence_state_counts"], + "ambiguous_rate_fraction": report["ambiguous_rate_fraction"], + "HIGH": report["HIGH"], + "SECONDARY": report["SECONDARY"], + "REJECT": report["REJECT"], + "false_high": report["false_high"], + "false_reject": report["false_reject"], + "false_secondary": report["false_secondary"], + "lexicalized_compositional_support": report["lexicalized_compositional_support"], + "lexicalized_noncompositional_instantiated": report["lexicalized_noncompositional_instantiated"], + "evidence_code_counts": report["evidence_code_counts"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/specs/007-hyperlexical-model/evaluation-reserve.md b/specs/007-hyperlexical-model/evaluation-reserve.md index 7f96f856..d86a06ce 100644 --- a/specs/007-hyperlexical-model/evaluation-reserve.md +++ b/specs/007-hyperlexical-model/evaluation-reserve.md @@ -351,3 +351,487 @@ Fresh operator authorization for `HLX-EXP-2026-09-26-SELECT-004` only. It does n Authorization artifact sha256 `7389783b52a5d5cf14ceca874cc12351d3f6571b6be8a8666c2040be38972625`. Post-authorization admission used `HYPERLEX_ALLOW_TRAIN=1` and `HLX_ADMISSION_ONLY=1`. Preflight and the trainer entrypoint share environment hash `a2b18512be8329ee63ad06e98f894c6138cf7eac05f6e82d53d4132e99a27f5d`. `admission_result` is `ADMISSION_PASS`. Status is `TRAINING_READY` because the scientific contract is sealed, the decision rule is sealed, and the real entrypoint passed. `optimizer_loaded` is false. Epochs 0. Gradient steps 0. Training was not started. Output directory stayed absent. BEST sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1` was not moved. Trunk sha256 `340ac08b74eef0d7bdec2d7981a6a3d4249bf0e6aab60634b72ad02c2b8023a9`. `training_launch_authorized` is false. `HLX_ALLOW_NO_HOLDOUT` stayed unset. Representation completeness is `PASS`. Evaluation quality is `LIMITED`. The decision covers classify 241, classify_observed 123, classify_non_none 188, and unbind_clean 250. It does not establish performance for relationship-dating, conflict-aggression, sports-competition, fashion-aesthetic, regional-cultural, or spiritual-mystic. Clean unbind stays limited to the sealed WordNet slice. Vendor calls: 0. Do not train. The next decision is launch authorization only. + +## SELECT-004 scored — HLX-EXP-2026-09-26-SELECT-004 + +One run finished under the sealed contract. Container `hlx-train-select004-1790448703` exited 0 after 40 epochs. The training schedule was not shortened. Checkpoint selection used `classify_macro_f1_nonnone` on the training-export validation surface. The selected checkpoint is epoch 3 at 0.64448782942204. Epoch 39 scored 0.5864567286803334. Candidate weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Consumed export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430`, 9150 rows, reserve filter removed 0. The reserve ledger was not modified during training. + +Reserve scoring used the sealed twelve-family universe `227b782011aad7e693fde253e103a24b3ca0bd6b04e090d446656fa943bf0175`. seed-morph78 macro-F1 0.022395727019119547, accuracy 0.14107883817427386, OBSERVED accuracy 0.032520325203252036, unbind clean exact 0.092. The candidate macro-F1 0.07245710784313726, accuracy 0.15767634854771784, OBSERVED accuracy 0.08130081300813008, unbind clean exact 0.100. Deltas are +0.05006138082401771, +0.01659751037344398, +0.048780487804878044, and +0.008. The char 3–5 baseline macro-F1 was 0.0, so CHAR_WINS did not occur. Decision **PROMOTE**. State `PROMOTION_ELIGIBLE`. `--apply-best` was not run. BEST remains `seed-morph78` / `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. The production head still emits the nine training families, so nine of the twelve gold families cannot be named and pull the absolute macro-F1 down for both checkpoints. The claim stays inside the sealed slices. Vendor calls: 0. The 40-epoch schedule was left as sealed. A later duration study can measure best-epoch locations; it is not a change to this result. + +## SELECT-004 promoted — HLX-EXP-2026-09-26-SELECT-004 + +Operator authorization applied the selected epoch-3 checkpoint as BEST. The promoted object is `hyperlex-encoder-modernbert-base-seed-select004/model.safetensors`, sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. `model.final.safetensors` (`d5be46be6611e382d837bab8868bb373cbead4f9caa3a666ca06f2b2cda1925b`) was not promoted. Epoch 39 was not promoted. No training ran. The reserve was not rescored. + +Previous BEST `hyperlex-encoder-modernbert-base-seed-morph78` remains in place, sha256 `fc53676bd347cccd4d0ac9a429f3469c36436f8eb0e09954e0c347c7b133a4a1`. + +Decision receipt sha256 `ee493b7d73b5e00ae27ec681bb52405bbcf2983a16a0297edad225aec52b7225`. Decision `PROMOTE`. Selection metric `classify_macro_f1_nonnone` `0.64448782942204`. Primary macro-F1 baseline `0.022395727019119547`, candidate `0.07245710784313726`, delta `0.05006138082401771`. Preservation deltas: classification accuracy `0.01659751037344398`, OBSERVED accuracy `0.048780487804878044`, unbind clean exact `0.008`. CHAR_WINS did not occur. Integrity `PASS`. + +The 491 sealed reserve identities moved `EVAL_RESERVE` to `EVAL_BOUND` to `EVAL_SPENT` through `IdentityLedger.transition` and `persist_append`. Events sha256 before `8223ae11bb42bd1a98ebcd739d1cfbc470085e241826b662703faefdfe752da6`, after `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. Projection sha256 before `d071b7aec8154203ce7f9ae9531639b8d638f86c2ac0c3af38bead9b3c4a48f9`, after `77e22433203879b252f7a9e309d2013d7550101d1c4a014b494d2c96df87d0e0`. The sealed binding file `c16e69559dd6582f687532ce6e2a2e9b52a70db1aa066d11e7e68e723936b371` was not rewritten. Historical consumed, spent, and abandoned identities outside this reserve were not changed. + +Representation completeness is `PASS`. Evaluation quality is `LIMITED`. This promotion does not establish performance across the eighteen-family ontology. The production head cannot name nine of the twelve sealed gold families. That limitation is separate from the checkpoint pointer. Vendor calls: 0. No further experiment was started. + +Promotion receipt sha256 `2e1f84b476f7355e31380419afad00d0b16d372235a0da6be85adc17fcf92d01`. + +## SELECT-005 blocked — HLX-EXP-2026-09-27-SELECT-005 + +`HLX-EXP-2026-09-27-SELECT-005` is the next unclaimed experiment id. Claimed ids are `HLX-EXP-2026-09-26-SELECT-001` through `HLX-EXP-2026-09-26-SELECT-004`. This id is recorded and not sealed. No preregistration file, arm directory, reserve, or admission receipt was created. + +`TRAINING_BLOCKED`. The proposed variable is the composite `train_schedule`. Control would be `max_epochs=40`, early stopping disabled, restore best. Candidate would be `max_epochs=12`, `minimum_epochs=4` scored epochs through epoch index 3, `early_stopping_patience=4`, strict increase, ties keep the earlier checkpoint, restore best. Both arms would pin `classify_macro_f1_nonnone`, warm start `hyperlex-encoder-modernbert-base-seed-morph65`, and export sha256 `64b7d3dede25047cb6dd2e5b663f7fa72946ec82ac1a8816ae34622d1aaac430` at 9150 rows. + +The canonical trainer does not implement that candidate. `scripts/shadow/hyperlexical/loop.py` scores `for ep in range(epochs)` and never stops early. `score_now > best_macro` already keeps the earlier checkpoint on ties, and best weights are written back after the loop when the selection metric is `classify_macro_f1_nonnone`. `epoch-progress.jsonl` records the metric and not wall-clock. Spark and public `main` have the same `loop.py` sha256 `f51e7aaff69e9033cc9ba16eee7225bfeefcf521e32236bdb791cc7900e130a5`. Spark HEAD `098ece4d9e8ebb27b0b0d3410b1280ed072d4847` was clean. Public `main` is `2f73f30010cc16ee014ed8d88de73131eb80d0a9`. + +The smallest separate change is an optional break in that epoch loop, default off: after a scored epoch, stop when at least 4 epochs have been scored and `epoch - best_epoch >= 4`, still capped by `max_epochs`, and append per-epoch wall-clock seconds to `epoch-progress.jsonl`. Setting `HYPERLEX_TRAIN_EPOCHS=12` is not that schedule. This record does not apply the change. + +SELECT-004 artifacts were not modified. Its reserve stays `EVAL_SPENT` and was not reused. No new reserve was allocated. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. No optimizer was constructed. Epochs and gradient steps for this id are zero. Launch is not authorized. + +## SELECT-005 still blocked — trainer synced, admission refused + +Spark now carries public `main` `a7d254e8981695072f5be36ab7ebdcc46cbd672c` for the trainer. `scripts/shadow/hyperlexical/loop.py` sha256 `1aa395081d7709be3844bf2568d12100d73d931ecbc51af43c3d6e01dba77e2a`. Early stopping remains default-off. `HLX-EXP-2026-09-27-SELECT-005` is still not sealed. No preregistration file, arm directory, reserve, or admission receipt was created. The private ledger sections above were not replaced by the public projection. + +`TRAINING_BLOCKED`. The trainer can express the candidate schedule. Canonical admission cannot seal it. + +`admit_training_run` allows exactly one scientific difference, and that difference must be `HLX_SELECT_METRIC`. A non-mutating probe pinned both arms to `classify_macro_f1_nonnone` and changed only `HYPERLEX_TRAIN_EPOCHS` (`40` versus `12`) and `HYPERLEX_EARLY_STOP` (`0` versus `1`). Patience and minimum epochs were the same on both arms. The gate `single_variable` failed: `scientific variable count is 2: HYPERLEX_EARLY_STOP,HYPERLEX_TRAIN_EPOCHS`. + +The spent ledger cannot supply a new `EVAL_RESERVE`. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. Live reserve counts are classify 0, classify_observed 0, classify_non_none 0, unbind_clean 0. The same probe failed at `holdout_reserve`: `reserve slice classify is absent`. The reserve scan also includes every identity with `evaluation_reserved` set. That scan is 491 identities, all derived `EVAL_SPENT`. A new reserve written onto this ledger would still fail that lifecycle check. The flag was not cleared. The SELECT-004 reserve was not reused. + +No optimizer was constructed. Epochs and gradient steps for this id are zero. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Launch is not authorized. + +The smallest separate change is an admission-contract patch: accept one declared schedule variable while both arms share `classify_macro_f1_nonnone`, and treat historical `EVAL_SPENT` identities as outside the current reserve without clearing `evaluation_reserved`. Do not allocate a reserve before that contract exists. + + +## SELECT-005 still unsealed — admission contract synced, fresh reserve unavailable + +Public `main` is `fab0de03d75e3280dc34feed425c4783ccf7e2b5`, the squash merge of the admission contract. Spark carries that `admission.py` sha256 `6ae7b63ca1b6e0459d9b795c731668b3eff524269d81bcb92181c0c9d407ab78` and `identity_ledger.py` sha256 `ac49b7240a895952b99b3ef9046f7a8185c8647badac7a45c8640d592057d1c5`. `scripts/shadow/hyperlexical/loop.py` remains sha256 `1aa395081d7709be3844bf2568d12100d73d931ecbc51af43c3d6e01dba77e2a`. The private ledger file was not replaced. + +`HLX-EXP-2026-09-27-SELECT-005` is still not sealed. No preregistration file, arm directory, reserve binding, or admission receipt was created. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `evaluation_reserved` was not cleared. The SELECT-004 reserve stays `EVAL_SPENT` and was not reused. + +A fresh reserve needs all four slices from text that is absent from the ledger and from the pinned export. The ledger has no `EVAL_RESERVE`, `EVAL_BOUND`, or `AVAILABLE` identities. The remaining settled hashes from the held-out stream that are absent from the ledger are `UNRESOLVED`, `NONE`, `RECLASSIFY`, or `ACCEPT` with `RIGHTS_UNRESOLVED`. Unresolved-rights rows stay out of `EVAL_RESERVE`. No novel rights-cleared classify settlement remains. A classify-absent reserve was not written. The WordNet unbind source was not admitted by itself. + +No optimizer was constructed. Epochs and gradient steps for this id are zero. BEST was not moved. It still names `hyperlex-encoder-modernbert-base-seed-select004`, weights sha256 `9fba0f66b1d5de6492470f53577d1447bfac1d29b9ac03869268abb70bbd97f6`. Launch is not authorized. + + +## SELECT-005 census — fresh reserve still unavailable + +A second join of the 339 settlement events to the 7964 ledger identities found no novel rights-cleared classify row. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The ledger was not appended. `evaluation_reserved` was not cleared. The SELECT-004 reserve was not reused. + +All 6761 unique hashes in the pinned export are already in the ledger. Of the 98 settlement events absent from the ledger, none are in that export. Those 98 are 82 `UNRESOLVED` with cleared rights, 2 `UNRESOLVED` with unresolved rights, 7 `ACCEPT` with unresolved rights, 6 `NONE` with unresolved rights, and 1 `RECLASSIFY` with unresolved rights. `UNRESOLVED` means the operator reviewed the row and declined to settle it. Unresolved-rights rows stay out of `EVAL_RESERVE`. There are 0 novel `CC-BY-SA` rows with decision `ACCEPT`, `RECLASSIFY`, or `NONE`. + +The 28 promoted-accept files are training gold under an older family set and were not remapped. WordNet can still supply `unbind_clean` and was not admitted alone, because a classify-absent reserve fails the slice gate. No preregistration, arm directory, reserve binding, or admission receipt was created. The optimizer was not constructed. Epochs and gradient steps for `HLX-EXP-2026-09-27-SELECT-005` remain 0. `training_launch_authorized` is false. BEST was not moved. + + +## SELECT-005 classify candidates blocked — HLX-EVAL-REVIEW-2026-09-27-001 + +Admission requires each active slice count to be at least 1. The classify floors are classify 1, classify_observed 1, and classify_non_none 1. Planning targets 606, 287, and 604 are not that floor. This pass did not lower the floor. + +Packet `HLX-EVAL-REVIEW-2026-09-27-001` is an operator-review packet, not a reserve. Ready rows: 0. Other screened rows: label unresolved 10, rights blocked 69, provenance blocked 4, cohort duplicate 1. Previously declined rights-cleared rows and unresolved-rights events were not reopened. WordNet and the older promoted-accept files were not used. No operator decision was written. + +Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The ledger was not appended. `HLX-EXP-2026-09-27-SELECT-005` was not sealed. BEST was not moved. + + +## ai-native evaluation family — 2026-09-27 + +Public `main` is `ca9403efe8470d46566abbcd098640f07b33b759`. `ai-native` is taxonomy-active. `evaluation.enabled` stays false. The production head stays nine names. No row was settled. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The review packet `HLX-EVAL-REVIEW-2026-09-27-001` now has 10 ready rows proposing `ai-native`. Their stored class stays `INFERRED`. `classify_observed` is still short by 1 until an operator attests `OBSERVED`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. + + +## Operator attested the ready rows observed + +The operator attested `OBSERVED` on all 10 ready rows in packet `HLX-EVAL-REVIEW-2026-09-27-001`. The proposed family is `ai-native`. Rights-blocked, provenance-blocked, and duplicate rows were not attested. No `ACCEPT`, `RECLASSIFY`, or `NONE` was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. + +## Unbind preview refused as slang — 2026-09-27 + +The operator reviewed the first 20 positional surfaces from the novel WordNet unbind pool. None are admitted as slang. Eighteen are refused. `give a damn` and `in one's birthday suit` are quarantined for provenance review. Slang, idiom, colloquialism, and profanity stay distinct. No `ACCEPT`, `REJECT`, `CORRECT_TARGET`, or `UNRESOLVED` was written to the unbind settlement log. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. These rows are not classify gold and must not be trained as slang. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Disposition packet `HLX-EVAL-UNBIND-PREVIEW-2026-09-27-001`. + +## Unbind preview roles — 2026-09-27 + +Slang classification and structure unbinding stay orthogonal. On the same 20-surface preview, the operator marked 7 high-value unbind candidates, 8 secondary candidates, and 5 rejects. The rejects are `beta vulgaris`, `gulf of oman`, `u. s. air force`, `department of the federal government`, and `court of assize and nisi prius`. The slang disposition is unchanged: none admitted, 18 refused, and `give a damn` plus `in one's birthday suit` quarantined as slang. No unbind settlement decision was written. The 15 candidates are not admitted. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-PREVIEW-2026-09-27-001`. + +## Unbind screen specified — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v1` is a candidate-selection rule, not an admission filter. Specification agreement with the 20-item preview is 20/20. That figure is specification fit, not held-out precision. A held-out validation sample of 32 admissible positional surfaces, excluding those 20, is recorded with proposed buckets and `gold` null. Provisional screen counts on 63882 admissible positional surfaces are high-value 1965, secondary 60559, reject 1358. Those counts are not a draw. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-001`. + +## Unbind screen v2 — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v2` replaces the v1 proposal as the screening hypothesis. It is not authorized as the SELECT-005 screen. Surface patterns remain candidate-generation heuristics. Hard exclusions are proper name, titled entity, taxonomy, productive number, and unconstrained free composition. Agreement after the revision is 20/20 on the first preview and 32/32 on the reviewed sample. That agreement is fit, not held-out precision. A new validation sample excludes all 52 reviewed surfaces and carries `gold` null. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-002`. + +## Unbind screen v3 — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v3` is a hypothesis. It is not authorized as the SELECT-005 screen. Candidate-generation patterns do not assign high-value. The automatic screen emits reject or unresolved only. v2 pool counts stay frozen at high-value 1977, secondary 55427, reject 6478, over 63882 positional surfaces, and were not recomputed. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-003`. + +## Unbind screen v3 held out — 2026-09-27 + +`RUNE.UNBIND_SCREEN.v3` stays a proposed refinement. Fifty-two surfaces are frozen as development data and thirty-two as validation-development data. Agreement on those eighty-four is fit, not held-out precision. A fresh sample of 29 admissible positional surfaces excludes all 84. Predictions were not hand-corrected. Held-out precision is not computed. v2 pool counts stay frozen at high-value 1977, secondary 55427, reject 6478, over 63882 positional surfaces. The v3 screen was not run on that pool. No settlement was written. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not sealed. Packet `HLX-EVAL-UNBIND-SCREEN-2026-09-27-004`. + +## Unbind held-out frozen — 2026-09-27 + +The 29-row v3 application is frozen. Sample sha256 `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. It was produced by one application of `RUNE.UNBIND_SCREEN.v3` and was not hand-corrected. Operator labels are pending. Held-out precision is `NOT_COMPUTABLE`. v3 was not revised. Development data remain 52 rows. Validation-development data remain 32 rows. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `HLX-EXP-2026-09-27-SELECT-005` is not authorized and is not sealed. + +## Unbind screen evaluation lane — 2026-09-27 + +`HLX-EVAL-UNBIND-SCREEN-V3-001` evaluates the frozen 29-row v3 application. Source sample sha256 remains `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. Prediction, operator judgment, gold, admission, and settlement are separate artifacts. The blind review does not carry the predicted bucket. Operator labels are pending. Held-out precision and the confusion matrix are `NOT_COMPUTABLE`. A scored report does not authorize `HLX-EXP-2026-09-27-SELECT-005`. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen held-out scored — 2026-09-27 + +Operator labels for HLX-EVAL-UNBIND-SCREEN-V3-001 are frozen. Label sha256 `4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5`. The source sample sha256 remains `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e`. Resolved accuracy is 16/29. High precision is 1 and recall is 6/13. Secondary precision is 4/17 and recall is 1. Reject precision is 1 and recall is 6/12. Quarantine support is 0. Every error is false secondary: 7 operator-high and 6 operator-reject. False high and false reject are 0. revision_eligible stays false. SELECT-005 is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 hypothesis — 2026-09-28 + +The v3 held-out score stays 16/29. All 13 errors are false secondary: 7 operator-high and 6 operator-reject. False high and false reject are 0. Those 29 rows are now v4 development evidence, not a validation set. `RUNE.UNBIND_SCREEN.v4` is drafted and not encoded, applied, or authorized. It would only add coverage around the secondary basin: normalized productive numbers, multi-token personal names, organization glosses, species common names, and medical technical phrases on one side; nonliteral and conventionalized gloss evidence on the other. Existing high and reject decisions stay in place. `revision_eligible` stays false. SELECT-005 is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v4.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. v4 may only move additional secondary fall-throughs, and only by semantic evidence classes. It must not reinterpret v3 high or reject logic, redefine secondary, or train on a future validation sample. Phrase-specific exceptions are prohibited. The 113 reviewed surfaces are the later regression set. The next measurement sample must exclude them and be frozen before inspection. `revision_eligible` stays false. SELECT-005 is not authorized. The v3 sample and label hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v4 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v4` is encoded as a wrapper over a frozen v3 bucket. It inspects a row only when that bucket is secondary. Patch A may move secondary to reject. Patch B may move secondary to high. Existing high and reject decisions are not reopened. The acceptance contract is unchanged, sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. + +The 113 reviewed surfaces were replayed as a regression suite. Gate A through Gate E passed. High rows unchanged: 39. Reject rows unchanged: 28. Secondary moves: 13 to high and 6 to reject. Each move has one Patch A or Patch B evidence code. No phrase-specific rule fired. Operator conflict on those moves: 0. This replay is not a new precision estimate. + +Success criteria were frozen before the measurement draw, sha256 `4fcbebfad7797eb22393fa94a389f1ef41402612e359b63d3a3a4b031c6cf8be`. High and reject precision floors are the v3 held-out floors of 1. The false-secondary rate must be strictly below 13/29. False high and false reject are not allowed. Perfect accuracy is not required. + +The measurement sample excludes all 113 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. v4 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The v3 sample sha256 `8af5644061a7a60fc5620c217e15a4ec8145f170edee9d8ff4e2999e7b86605e` and label sha256 `4e7bae5986e6345de62086af270a1d1a6902103d69a50d8f0b1e4e0fe01ecde5` are unchanged. + +## Unbind screen v4 measurement scored — 2026-09-28 + +Operator labels for the 28-row v4 measurement sample are frozen. Label sha256 `023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463`. The sample sha256 remains `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. The prediction sha256 remains `a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f`. Labels were recorded at `2026-09-28T02:17:00Z`, after the sample freeze at `2026-09-28T01:07:34Z`. Hand corrections are 0. v4 was not applied again. + +Resolved accuracy is 16/28. High precision is 1 (7/7) and recall is 7/13. Reject precision is 1 (5/5) and recall is 5/11. Secondary precision is 4/16 and recall is 4/4. Quarantine support is 0. False high is 0. False reject is 0. False secondary is 12/28, which is below the frozen floor of 13/29. Every error is a secondary fall-through. The pre-registered success criteria pass, so `revision_eligible` is true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 hypothesis — 2026-09-28 + +The v4 measurement artifact and score receipt stay frozen. Sample sha256 `dae8851134aa960a13e072ae017428054c68b988c8d7e6f86d8cab2d16c2586b`. Prediction sha256 `a854847e516fbcd37fbb221456e8caf8c420552795ac7fc6225960bb5434084f`. Label sha256 `023691f8349f0dda12c234691f235ae109289fcf9eab86ec20be1e23bfed9463`. Acceptance sha256 `ffb39e38784a56ae15bae51718c61b78fc861e48399936dbed57fb7d0754c55b`. Resolved accuracy remains 16/28. High precision remains 1. Reject precision remains 1. False secondary remains 12/28. `revision_eligible` on that measurement remains true. + +Those 28 rows are now v5 development evidence, not a validation set. Evidence sha256 `dcab832038c3209a54a4159b23caf4eecfb94864cef7188af28f1ae3b0ff80c0`. The twelve secondary fall-throughs are two escape routes only: referential or terminological rows that stayed secondary, and lexicalized noncompositional rows that stayed secondary. They are not a fit list. + +`RUNE.UNBIND_SCREEN.v5` is drafted and not encoded, applied, or authorized. It is a wrapper over a frozen v4 bucket. It inspects a row only when that bucket is secondary. One referential/terminological dominance test may move secondary to reject. One lexicalized noncompositionality test may move secondary to high, and only when the surface is conventionalized and the gloss is not compositionally recoverable. A lexicalized and mostly compositional surface stays secondary. Existing high and reject decisions stay in place. A separate rule for each miss is prohibited. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v5.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. The draft hypothesis sha256 is `c3d3f8081e18582158eab7a1bf154a02aed38614b9801df6a83f7eaf3875da7b`. v5 may only move additional secondary fall-throughs, and only through the two general tests named in the contract. It must not reinterpret v3 or v4 high or reject logic, redefine secondary, redesign the classifier, or train on a future validation sample. Phrase-specific exceptions are prohibited. One executable rule per observed miss is prohibited. + +The later regression replays all 141 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, and 28 from the v4 measurement. The 113-row v4 replay and the 28-row measurement sample are disjoint. Every previously correct high stays high. Every previously correct reject stays reject. Every new move originates from secondary. The next measurement sample must exclude all 141 and be frozen before inspection. v5 is applied once. Operator labels are collected independently. + +The pre-registered bar is high precision 1, reject precision 1, and a false-secondary rate strictly below 12/28. No accuracy-gain target is registered. Perfect accuracy is not required. `revision_eligible` on the v4 measurement stays true and does not authorize `HLX-EXP-2026-09-27-SELECT-005`. The v4 sample, prediction, label, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v5` is encoded as a wrapper over a frozen v4 bucket. It inspects a row only when that bucket is secondary. Referential or terminological dominance may move secondary to reject. A conventionalized surface whose gloss is not recoverable from the ordinary first senses of its constituents may move secondary to high. Existing high and reject decisions are not reopened. The acceptance contract is unchanged, sha256 `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. + +The 141 reviewed surfaces were replayed as a regression suite. High rows unchanged: 59. Reject rows unchanged: 39. Secondary moves: 4 to high and 7 to reject. Each move has the one evidence code for its patch. No phrase-specific rule fired. Operator conflict on those moves: 0. This replay is not a new precision estimate. Gate report sha256 `c70214ac1ea3f24b0b8c5590b015a64de4f5a7a632cf372f02aa9e88bd5b2977`. + +The pre-registered bar is unchanged: high precision 1, reject precision 1, and a false-secondary rate strictly below 12/28. Criteria sha256 `49e179db537a738bb7370404167c7ec1c7019fcd251d77e6a916f4426a0ecccb`. No accuracy-gain target was added. + +The measurement sample excludes all 141 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. Prediction sha256 `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. v5 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. `revision_eligible` on this screen stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v5 measurement scored — 2026-09-28 + +Operator labels for the 28-row v5 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f`. The row table is high 11, secondary 8, reject 9, quarantine 0, unresolved 0. The seal line said high 10, secondary 8, reject 10. The frozen artifact follows the row table. + +The sample sha256 remains `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. The prediction sha256 remains `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. Labels were recorded at `2026-09-28T03:03:27Z`, after the sample freeze at `2026-09-28T02:44:48Z`. Hand corrections are 0. v5 was not applied again. The 141-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 17/28. High precision is 6/7 and recall is 6/11. Reject precision is 5/6 and recall is 5/9. Secondary precision is 6/15 and recall is 6/8. Quarantine support is 0. False high is 1. False reject is 1. False secondary is 9/28: 5 operator-high and 4 operator-reject. Accuracy was recorded and was not part of the bar. + +False high is 1 and false reject is 1, so the outer-bucket bar fails. The false high is `keep out`, which v4 had already marked high and v5 passed through. The false reject is `on the job`, which v5 moved from secondary to reject on referential/terminological dominance. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `752fd459658f636df91a8da9f1a701a0de8e3240d6df12072bbd2b7ac4cc0b7a`. + +## Unbind screen v6 hypothesis — 2026-09-28 + +The v5 measurement stays scored as an outer-bucket failure. Sample sha256 `a462f08307e62b09fdfe1dfb6e9a86ec3ea207db27e917730b0d244c47c73359`. Prediction sha256 `51bcf051fb1a4b9bf67a28fe7d1235c8922e3a887bb908de32e5ac0d78a6408f`. Label sha256 `b05dc1c35d3d10eed4b615ca1ee8251a875a450f57560fef4b5a84b3e1fc075f`. False high is 1. False reject is 1. High precision is 6/7. Reject precision is 5/6. False secondary is 9/28. Resolved accuracy is 17/28 and is descriptive only. `revision_eligible` on v5 stays false. + +The error analysis is frozen and authorizes no fix. Analysis sha256 `ecb229dccec627af958d12537ab153d35de9e3e58d71bb8d3f0a8835077b26ac`. One inherited high overreach: `keep out` was already high at v4, and v5 passed it through against an operator secondary label. One new reject overreach: `on the job` moved from secondary to reject under referential/terminological dominance against an operator secondary label. Secondary undercoverage remains 5 operator-high rows and 4 operator-reject rows. These 28 rows are v6 development evidence, not a validation set. Evidence sha256 `daf512dd331216fecb87ec4d4890c83a03afaf78b5fc1ca6dae6e46e1a9f63a9`. + +`RUNE.UNBIND_SCREEN.v6` is drafted and not encoded, applied, or authorized. Draft sha256 `942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5`. It is a structural revision, not another secondary-only wrapper. A frozen v5 bucket is provisional. High may fall to secondary only with compositional recoverability. Reject may fall to secondary only with nonreferential lexical use. A demotion stops at secondary for that application. A provisional secondary row may still move by the two existing evidences: lexicalized noncompositional to high, and referential/terminological dominance to reject. High and reject do not swap directly. The motivating surfaces are not a required fit. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v6.ACCEPTANCE` is frozen and the screen is not encoded. Acceptance sha256 `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. The draft hypothesis sha256 is `942eeab825d1c89c21e91af84da0d753281a91012e9ff219a633d27fe6648aa5`. The error analysis sha256 is `ecb229dccec627af958d12537ab153d35de9e3e58d71bb8d3f0a8835077b26ac`. The development evidence sha256 is `daf512dd331216fecb87ec4d4890c83a03afaf78b5fc1ca6dae6e46e1a9f63a9`. + +The replay set is all 169 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, 28 from the v4 measurement, and 28 from the v5 measurement. The v5 measurement is disjoint from the prior 141. Previously operator-correct rows must remain operator-correct. High falls to secondary only with compositional recoverability. Reject falls to secondary only with nonreferential lexical use. Secondary rises to high only with lexicalized noncompositional evidence. Secondary falls to reject only with referential/terminological dominance. Direct movement between high and reject is prohibited. Phrase-specific rules are prohibited. + +The pre-registered measurement bar is high precision 1 and reject precision 1, with false high and false reject at 0. No false-secondary quota and no accuracy-gain target are registered. The next measurement sample must exclude all 169 reviewed surfaces and be frozen before inspection. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. The v5 sample, prediction, label, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v6` is encoded over a frozen v5 bucket. High may fall to secondary only when the gloss is recoverable from the ordinary first senses of at least two constituents and the synset has no unrelated single-word mapping. Reject may fall to secondary only when a pertainym is used as a state or relation whose gloss meets those ordinary senses, and not when the gloss is a relational designation. A demotion stops at secondary for that application. A provisional secondary row may still move by lexicalized noncompositional evidence or referential/terminological dominance. High and reject do not swap. The acceptance contract is unchanged, sha256 `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. + +The 169 reviewed surfaces were replayed. Previously correct rows lost: 0. Direct swaps between high and reject: 0. High to secondary: 0. Reject to secondary: 1. That row is `on the job`, evidence `nonreferential_lexical_use`, and the operator label is secondary. Secondary to high: 0. Secondary to reject: 0. No phrase-specific rule fired. `keep out` stays high. Its gloss is not covered by the ordinary first senses of its constituents, and the synset maps it to an unrelated word, so compositional recoverability does not fire. That miss is not a required fit. Unchanged disagreements: 16. This replay is not a precision estimate. Gate report sha256 `66e4f3859738dfaa63a21ec2e69025b568ea097c17f509ceed4277022d979ffb`. Prediction sha256 `7398254bd62253129153e7b57c99122ed7cc7157520788cc60bb0e80f0b356e8`. Diff sha256 `3af430bb9261c1cedc1bd6181f5e928e01eef22bccea0b81c3c435073fabff59`. + +The pre-registered bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor and no accuracy target are registered. Criteria sha256 `3a511ffb49f32930ead14ca2f1e2f6639122c47a7ceec9d1bc29815dd7b490dd`. + +The measurement sample excludes all 169 reviewed surfaces and duplicate normalized lexical identities. It is stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. Sample sha256 `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. Prediction sha256 `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. v6 was applied once. Hand corrections are 0. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. `revision_eligible` on this screen stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v6 measurement scored — 2026-09-28 + +Operator labels for the 28-row v6 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11`. The row table is high 12, secondary 4, reject 12, quarantine 0, unresolved 0. The seal line said high 11, secondary 4, reject 13. The frozen artifact follows the row table. + +The sample sha256 remains `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. The prediction sha256 remains `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. Labels were recorded at `2026-09-28T03:34:48Z`, after the sample freeze at `2026-09-28T03:25:47Z`. Hand corrections are 0. v6 was not applied again. The 169-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 21/28 and is descriptive only. High precision is 9/10 and recall is 9/12. Reject precision is 1 (9/9) and recall is 9/12. Secondary precision is 3/9 and recall is 3/4. Quarantine support is 0. False high is 1. False reject is 0. False secondary is 6/28: 3 operator-high and 3 operator-reject. Recall and accuracy were not part of the bar. + +A false high or false reject is present, so the outer-bucket bar fails. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `68c60dfe9efaf9f79bc445b974c5436d31b60b77fe7374b17103f3b74865610f`. + +## Unbind screen v7 hypothesis — 2026-09-28 + +The v6 measurement stays scored as an outer-bucket failure. Sample sha256 `8ee516423f002a355759eed788bfe3bef81332fad61b1f19c7e46bd0722a29f2`. Prediction sha256 `476c5526b18fcb999f2a5397d0674ec07e29b199d5a5259afb555bf608d245ee`. Label sha256 `feed0ee131a513f2801bd1725f553bb6274df1378b28fdea1cb13e238ea9eb11`. High precision is 9/10. Reject precision is 1 (9/9). False high is 1. False reject is 0. False secondary is 6/28. Resolved accuracy is 21/28 and is descriptive only. `revision_eligible` on v6 stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. + +The only false high in that measurement is `to a lesser extent`. The operator bucket is secondary. v3, v4, v5, and v6 are high, and v6 recorded no evidence code. A prior inherited high disagreement, `keep out`, stays high in the 169-row replay. Its operator bucket is secondary and `compositional_recoverability` did not fire. That v6 predicate is not redefined. Secondary fall-throughs are 13/29 at v3, 12/28 at v4, 9/28 at v5, and 6/28 at v6. Coverage is not the next question. + +The error analysis is frozen and authorizes no fix. Analysis sha256 `06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766`. These 28 rows are v7 development evidence, not a validation set. Evidence sha256 `3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3`. The six false-secondary rows stay unresolved and are not a fit list: `to the letter`, `with child`, `dressed to the nines`, `union jack`, `atomic number 98`, and `law of definite proportions`. Known reviewed inventory is 197: the prior 169 plus these 28. Normalized overlap is 0. + +`RUNE.UNBIND_SCREEN.v7` is specified and not encoded, applied, or authorized. State is `SPEC_FROZEN`. Draft sha256 `c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b`. It keeps the frozen v6 phase order and replaces only the high challenge. A provisional high may fall to secondary only with `ordinary_compositional_derivation`: the WordNet sense can be derived by ordinary grammatical, syntactic, comparative, or phrasal composition without a stored conventionalized or nonliteral lexical binding. The test does not require recovery from only the ordinary first sense of each token. It asks whether lexicalized binding is required at all. A demotion stops at secondary for that application. Reject may still fall to secondary only with `nonreferential_lexical_use`. A provisional secondary row may still move by the unchanged v6 evidences. High and reject do not swap. `to a lesser extent` and `keep out` motivate the question and are not required fits, and neither is special-cased. No SELECT-005 admission surface exists. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 acceptance frozen — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7.ACCEPTANCE` is frozen and the screen is not encoded. State is `SPEC_FROZEN`. Acceptance sha256 `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. The draft hypothesis sha256 is `c168bf975e26804a032956d570e39a3d2c7078b407ef966da33083adecf5585b`. The error analysis sha256 is `06a2f10493640d6536b45ef3d9405470c12623bfa787f7fba49b3f64498fb766`. The development evidence sha256 is `3d457801203c69ceb13c6cefbf5d82fe9cac0daea48f26eb2642d8673585a3a3`. + +The replay set is all 197 reviewed surfaces: 52 development, 32 validation-development, 29 from the v3 held-out application, 28 from the v4 measurement, 28 from the v5 measurement, and 28 from the v6 measurement. The v6 measurement is disjoint from the prior 169. Previously operator-correct rows must remain operator-correct. A previously correct high stays high. High falls to secondary only with ordinary compositional derivation. The frozen v6 transitions for secondary to high, secondary to reject, and reject to secondary are unchanged. Direct movement between high and reject is prohibited. A high demotion stops at secondary and is not re-promoted in the same application. Phrase-specific rules, row-id rules, and surface-hash rules are prohibited. + +No measurement sample was drawn. The bar proposed for a later unseen sample is high precision 1 and reject precision 1, with false high and false reject at 0. No false-secondary quota, no accuracy target, and no recall target are registered. That sample must exclude all 197 reviewed normalized identities and be frozen before inspection. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. No admission surface for that experiment exists. The v6 sample, prediction, label, criteria, and acceptance hashes are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 encoded — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v7` is encoded over a frozen v6 bucket. The only new predicate is `ordinary_compositional_derivation`. It inspects a provisional high. The recorded sense must be a comparative, syntactic, or phrasal composition, and the synset must not store an unrelated single-word synonym. A gloss that merely shares constituent stems does not fire, and a metaphorical retelling does not fire. A demotion stops at secondary. Reject rows still use the v6 reject challenge. Secondary rows are not reopened, so a demotion v6 already stopped stays stopped. `compositional_recoverability` is not a v7 transition. The acceptance contract is unchanged, sha256 `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. + +The 197 reviewed surfaces were replayed as development and regression evidence, not as a generalization estimate. Previously correct rows lost: 0. Previously correct high rows demoted: 0. Direct swaps between high and reject: 0. High to secondary: 1. Reject to secondary: 0. Secondary to high: 0. Secondary to reject: 0. One row moves. `to a lesser extent` goes from high to secondary with `ordinary_compositional_derivation`, and the operator bucket is secondary. That firing is not a required fit. `keep out` stays high. No phrase-specific rule fired. This replay is not a precision estimate. Gate report sha256 `78225b68c639694bb4c17342c87320ccb44faac557a9145ea117603da0d545ed`. Prediction sha256 `24f266e16b80da602b011bf7cca13774ce9f76c0c52e81e600a1d9743d61d197`. Diff sha256 `b86ae6611a5fcd233b6e92b3def4cdd0680018902532a0a0d5fb02659ad092ae`. + +State is `REGRESSION_VERIFIED`. `measurement_eligible` stays false. The unseen sample was not drawn. The pre-registered bar remains high precision 1, reject precision 1, false high 0, and false reject 0, with no false-secondary quota, no accuracy target, and no recall target. No measurement sample was drawn. `revision_eligible` stays false on v7, v6, and v5. `revision_eligible` on the v4 measurement stays true. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 measurement frozen — 2026-09-28 + +The 197-row regression stays `REGRESSION_VERIFIED`. One unseen measurement sample excludes those 197 normalized identities. Normalized overlap with the reviewed set is 0. Duplicate normalized identities inside the sample are 0. The draw is deterministic and stratified by source part of speech and token count, two rows from each occupied cell. Occupied cells produced 28 rows. No occupied cell was exhausted. Each occupied cell contributed 2 rows. + +The sample was frozen before v7 was applied. Sample sha256 `2b4414e2a6db3acf2967603e3ef4552c631803285633fbae5f3f3d50451724b1`. v7 was then applied once. Hand corrections are 0. Prediction sha256 `d201affe0c87206227bc211a6071a25e639c0ffb7cbcfdc8dac0572155a4da47`. The blind rows carry surface, part of speech, token count, and gloss. They do not carry a bucket or an evidence code. Operator labels are absent. Precision is `NOT_COMPUTABLE`. State is `MEASUREMENT_ELIGIBLE`. + +The acceptance bar is unchanged: high precision 1, reject precision 1, false high 0, and false reject 0. No false-secondary floor, no accuracy target, and no recall target were added. Criteria sha256 `3b22449874f2384d843e4f16cb7e5c28b6acfc83273aec44542e0bc5b0c3ef37`. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind screen v7 measurement scored — 2026-09-28 + +Operator labels for the 28-row v7 measurement sample are frozen. Each label carries only the row identity, the surface, the operator bucket, and the operator reason. Label sha256 `0ce39dffad49b28b5adebfbe3514fcf81f9ca39689d3eadf0695975496b6adb5`. The row table is high 14, secondary 5, reject 9, quarantine 0, unresolved 0. The seal line said high 13, secondary 5, reject 10. The frozen artifact follows the row table. + +The sample sha256 remains `2b4414e2a6db3acf2967603e3ef4552c631803285633fbae5f3f3d50451724b1`. The prediction sha256 remains `d201affe0c87206227bc211a6071a25e639c0ffb7cbcfdc8dac0572155a4da47`. Labels were recorded at `2026-09-28T04:14:27Z`, after the sample freeze at `2026-09-28T04:05:23Z`. Hand corrections are 0. v7 was not applied again. The 197-row regression remains a regression result and is not a generalization estimate. + +Resolved accuracy is 21/28 and is descriptive only. High precision is 10/11 and recall is 10/14. Reject precision is 7/8 and recall is 7/9. Secondary precision is 4/9 and recall is 4/5. Quarantine support is 0. False high is 1. False reject is 1. False secondary is 5/28: 3 operator-high and 2 operator-reject. Recall, accuracy, and false secondary were not part of the bar. + +A false high or false reject is present, so the outer-bucket bar fails. `revision_eligible` stays false. The screen was not retuned. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `revision_eligible` on v6 stays false. `revision_eligible` on v5 stays false. `revision_eligible` on the v4 measurement stays true. Acceptance sha256 remains `562c0756b0337e2fb10643f4fd6689ea4421977a345d12fe33d8a1504161cac5`. + +## Unbind screen lineage retired — 2026-09-28 + +`RUNE.UNBIND_SCREEN.v3` through `RUNE.UNBIND_SCREEN.v7` stay sealed. They are a completed experimental lineage. The lineage showed that further surface-pattern patches are the wrong instrument. No `RUNE.UNBIND_SCREEN.v8` exists. Historical scores are not results for the sense-first screen. v7 remains `SCORED`, outcome `OUTER_BUCKET_FAILURE`, `revision_eligible` false. v7 was not retuned. v6 and v5 `revision_eligible` stay false. The v4 measurement `revision_eligible` stays true. Source sha256 values remain v3 `179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6`, v4 `f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377`, v5 `70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948`, v6 `59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba`, v7 `73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab`. Retirement sha256 `fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f`. + +The v7 measurement disagreements are not one failure class. The frozen error analysis separates three questions: whether the bound sense designates a referent, whether that sense requires conventionalized lexical binding, and whether that sense is compositionally recoverable. Error analysis sha256 `ebc56d4d4499efee19bc368365b0d6d3a7afc27ede4e78f40fb9d0fd15fcb9c8`. The canonical v7 operator counts from the row artifact are high 14, secondary 5, reject 9. The handwritten seal does not override those rows. + +## Unbind sense screen v1 specified — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` is `SPEC_FROZEN`. It is a new screen family. It is not encoded, not applied, and no measurement sample was drawn. The unit of analysis is the specific lexical sense: surface, part of speech, WordNet synset, and the frozen first-sense gloss. Another sense of the same surface, a historical v3-v7 bucket, and a referential origin that is not the predicated sense are outside that unit. + +The five classes are `REFERENTIAL`, `LEXICALIZED_NONCOMPOSITIONAL`, `LEXICALIZED_COMPOSITIONAL`, `ORDINARY_COMPOSITIONAL`, and `AMBIGUOUS`. The procedure assigns exactly one class. Empty or unmatched sense evidence is `AMBIGUOUS`. A figurative gloss is not referential merely because the wording mentions a place or story; `road to damascus` is the protected illustration and is not a required fit. Ordinary syntax, comparison, degree, and phrasal composition are `ORDINARY_COMPOSITIONAL`; a WordNet lemma by itself is not lexicalization. `as far as possible` motivates that distinction and is not a required fit. Insufficient evidence stays `AMBIGUOUS`. + +The frozen mapping is referential to reject, lexicalized noncompositional to high, both compositional classes to secondary, and ambiguous to quarantine. Ordinary composition does not map to reject. That would hide a second classifier. The two compositional classes remain distinct before the mapping. No historical bucket may override the sense class. + +The reviewed positional inventory is 225: the prior 197 plus the 28 v7 measurement rows. Normalized overlap between those sets is 0. The 28 rows are development evidence, not validation, and not a required fit. 225 rows bind a synset offset whose first-sense gloss equals the frozen gloss. 0 rows keep the frozen gloss with no offset. A future measurement must exclude all 225 normalized identities and be frozen before inspection. The proposed pass/fail bar is high precision 1, reject precision 1, false high 0, and false reject 0. Sense-class confusion, bucket confusion, per-class support, the ambiguous rate, high recall, reject recall, and secondary precision and recall are descriptive. This pass does not make them pass/fail criteria. + +Draft hypothesis sha256 `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. Acceptance sha256 `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. Development evidence sha256 `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. Tracker sha256 `4145848a2e1368b23f28f645249a9a8c55bf113a43763d31a434f1389b9e1154`. No JSON Schema exists for this hypothesis family. The older unbind sample, label, and report schemas are a different contract and were not applied. The next named transition is `ENCODED`. This pass does not authorize it. + +## Lexeme structure screen v1 architecture — 2026-09-28 + +`RUNE.LEXEME_STRUCTURE_SCREEN.v1` is an architecture note for a sibling lane. It evaluates one orthographic lexeme for internal structure. Candidate classes are atomic, compound, affixed, blend, clipping, respelling, reduplicated, borrowed, and unknown. A single token is not evidence of an atomic lexeme. A string split is not evidence of a valid morphological decomposition. Conceptual illustrations are not a corpus and not a required fit. This lane is not mixed into positional unbind. It is not encoded. No corpus was populated, no sample was drawn, and no gold was created. Architecture sha256 `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. + +`HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +## Unbind sense screen v1 procedure frozen — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` moves from `SPEC_FROZEN` to `PROCEDURE_FROZEN`. The screen is not encoded and not applied. No development replay was run. No row received a sense class. No measurement sample was drawn. + +The companion artifact is `hyperlex.unbind_sense_screen_v1_classification_procedure.v1`, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The sealed acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The sealed draft stays `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. The acceptance does not embed this procedure, so its bytes were not revised. The tracker now points at the procedure. Tracker sha256 `f9c4757b6eec558b5e1baf644bcf33c27c949807e7f00cd15df869eb6411de31`. + +WordNet membership is not evidence that a surface is a conventional lexical unit. A conventional unit requires an unrelated single-word co-lemma or a lexical pointer on the whole lemma. A productive frame requires a one-token lemma alternation stored on the synset, or a gloss that starts with the comparative or superlative operator formula. Anything else at that test is `AMBIGUOUS`. A stored unrelated equivalent is the noncompositional mapping. A derivation or pertainym from the whole lemma to one of its constituents, with no unrelated equivalent, is the compositional lexical unit. Referential designation requires an instance-hypernym pointer or a parenthetical four-digit lifespan. A geographic or religious allusion does not qualify, and neither does the capital letter in `road_to_Damascus`. Ordinary composition still maps only to secondary. Ambiguity maps to quarantine and is a normal class. + +The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. Reviewed positional inventory remains 225. v3 through v7 sources are unchanged. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `ENCODE_AUTHORIZATION`. This pass does not authorize it. + +## Unbind sense screen v1 development replay — 2026-09-28 + +`RUNE.UNBIND_SENSE_SCREEN.v1` is encoded and the 225-row development replay is analyzed. The frozen classification procedure is unchanged, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The sealed acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The sealed draft stays `93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38`. The development manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. No sense class was written onto that manifest. + +The encoder follows the frozen tests. An instance-hypernym pointer or a parenthetical four-digit lifespan is referential. An unrelated single-word co-lemma is lexicalized noncompositional. A derivation or pertainym from the whole lemma back to a constituent, with no unrelated equivalent, is lexicalized compositional. A stored one-token lemma alternation, or a gloss that starts with the comparative or superlative operator formula, is ordinary compositional. Anything else is ambiguous. A yes-signal together with a no-signal is a conflict and stays ambiguous. Absence of a signal is not secondary. No gloss-keyword list, technical-term dictionary, or phrase exception was added. A high ambiguous rate was not repaired. + +Each development row was classified once from its frozen synset. The replay is development evidence, not validation. The measurement bar was not applied. Prediction sha256 `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef`. Report sha256 `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. Tracker sha256 `b3690de5c957f019224c7ea980bf6e27b25ba1f1c519347f72a327fa3ea399a2`. + +Sense classes are referential 26, lexicalized noncompositional 72, lexicalized compositional 0, ordinary compositional 25, and ambiguous 102. The ambiguous rate is 102/225. Buckets are high 72, secondary 25, reject 26, and quarantine 102. Evidence codes are referential designation 26, noncompositional semantic mapping 72, compositional lexical unit 0, productive grammatical frame 25, and insufficient record evidence 102. The determinate sources are instance hypernym 26, unrelated single-word co-lemma 72, and productive alternation 25. No lifespan marker and no grammatical-operator gloss fired on this inventory. + +Lexicalized compositional support is 0. One development row stores a pertainym from the whole lemma to a constituent token and also stores a one-token lemma alternation. The frozen conflict rule returns ambiguous. That zero records the frozen test. It is not a missing dictionary. + +Operator labels already on the manifest, joined after classification: of 101 operator-high rows, 32 stay high, 17 go to secondary, and 52 go to quarantine. Of 46 operator-secondary rows, 22 go to high, 3 stay secondary, and 21 go to quarantine. Of 77 operator-reject rows, 18 go to high, 4 go to secondary, 26 stay reject, and 29 go to quarantine. The one operator-quarantine row goes to secondary. Descriptive precision and recall, not pass/fail criteria: high 32/72 and 32/101, secondary 3/25 and 3/46, reject 26/26 and 26/77. Operator quarantine support is 1. False high is 40. False reject is 0. False secondary is 22 and is descriptive only. + +`road to damascus` is ambiguous and quarantined. The bound record has no instance pointer and no lifespan marker. `as far as possible` is ordinary compositional and secondary, from the stored far/much alternation. Neither row is a required fit. + +State is `DEVELOPMENT_ANALYZED`. The path was `PROCEDURE_FROZEN`, `ENCODE_AUTHORIZED`, `ENCODED`, `225_ROW_DEVELOPMENT_REPLAY`, `DEVELOPMENT_ANALYZED`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. v3 through v7 sources are unchanged. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `MEASUREMENT_AUTHORIZATION`. This pass does not authorize it. + +## Unbind sense screen procedure v2 frozen — 2026-09-28 + +The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v1 stays frozen, sha256 `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. The companion procedure `hyperlex.unbind_sense_screen_v1_classification_procedure.v2` is `PROCEDURE_V2_FROZEN`. It is not encoded and not applied. No v2 development replay was run. No row received a v2 sense class. No measurement sample was drawn. + +The v1 development replay mapped an unrelated single-word co-lemma to lexicalized noncompositional, and therefore to high. All 40 development false highs used that path. A stored whole-expression synonym can show that the phrase is a lexical unit. It does not show that the sense is semantically noncompositional. + +Procedure v2 keeps the five classes and the sealed bucket map. Before the class, it records three states: referential, lexicalized, and compositional. Each state is yes, no, unknown, or conflict. An unknown state is not a no. Referential yes, from an instance-hypernym pointer or a parenthetical four-digit lifespan, remains referential and reject. A co-lemma or a whole-lemma lexical pointer is lexicalized yes. A derivation or pertainym back to a constituent, a stored one-token alternation, or a comparative or superlative operator gloss is compositional yes. The alternation does not set lexicalized to no. The operator gloss sets lexicalized to no and compositional to yes, so when referential is not yes the class is ordinary compositional. Lexicalized yes together with compositional yes, when referential is not yes, is lexicalized compositional. That cell is reachable. The procedure does not force rows into it. + +WordNet 3.0 has no structural noncompositionality field once the co-lemma shortcut is retired. The compositional no-signal set is empty, so the lexicalized-noncompositional cell stays defined and does not fire. A later encoder must not invent a replacement signal. Missing lexicalization evidence stays unknown. On the record already frozen in procedure v1, `as far as possible` is therefore ambiguous under v2, and `road to damascus` stays ambiguous. Neither illustration is a required fit, and neither class was written onto the development manifest. An insufficient record stays ambiguous and maps to quarantine. A lower ambiguous rate is not a success criterion. Referential precision is not traded for coverage, and no technical-term gloss list was added. + +No JSON Schema document exists for procedure v2. The schema name on the artifact is not a JSON Schema file. Older unbind sample, label, and report schemas were not applied. Procedure sha256 `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. v1 error analysis sha256 `471bc27b89f550fae36b3471daaad282a6dd8735414846cb18aafe1195e0a52e`. Change note sha256 `443ce2964d4e4fcd8257055cb1404965faa70b838264b1f623be192d1cae085c`. Tracker sha256 `af11d20ebefec5629718617ebacc07d8b4cc36c61e59e829d153b05b9397ca40`. + +The reviewed positional inventory remains 225. A future encoded replay of those same rows may report lexicalized-compositional support, the false-high count, the ambiguous rate, referential precision and coverage, the sense-class distribution, the evidence-state distribution, and the operator confusion tables. This freeze sets no accuracy threshold. Lexicalized-compositional support above zero would be a structural readiness signal, not a hard pass/fail rule. + +The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. The procedure records the shared invariant that lexicalized identity, compositionality, and semantic shift are different questions. That note was not edited, and the lexeme screen was not implemented. v3 through v7 sources are unchanged. The v1 implementation, v1 predictions, and v1 report are unchanged. Acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. Prediction sha256 remains `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef`. Report sha256 remains `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. + +The next named transition is `ENCODE_PROCEDURE_V2_AUTHORIZATION`. This pass does not authorize it. An unseen measurement stays deferred until an encoded v2 replay of these 225 rows has been examined. + +## Unbind sense screen procedure v2 development replay — 2026-09-28 + +Procedure v2 is encoded and the same 225 development rows were replayed once. State is `DEVELOPMENT_ANALYZED_V2`. The path was `PROCEDURE_V2_FROZEN`, `ENCODE_PROCEDURE_V2_AUTHORIZATION`, `ENCODED`, `225_ROW_DEVELOPMENT_REPLAY`, `DEVELOPMENT_ANALYZED_V2`. The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v2 bytes stay `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. Procedure v1 stays `4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad`. No sense class was written onto the development manifest. No measurement sample was drawn. + +The replay asks whether the frozen WordNet record can instantiate the five classes, and `LEXICALIZED_NONCOMPOSITIONAL` in particular, without treating a single-word co-lemma as compositional NO. It does not instantiate that class. High support is 0. Compositional NO count is 0. False high is 0. A co-lemma sets lexicalized YES and leaves the row ambiguous when compositionality is unknown. Ninety-five rows are lexicalized YES. One hundred eighty-two rows are compositionally unknown. + +Sense classes are referential 27, lexicalized noncompositional 0, lexicalized compositional 16, ordinary compositional 0, and ambiguous 182. The ambiguous rate is 182/225. Buckets are high 0, secondary 16, reject 27, and quarantine 182. Of the 16 lexicalized-compositional rows, 15 are a stored one-token alternation on a lexicalized synset and 1 is a derivation or pertainym back to a constituent. That support is a readiness signal. It is not a pass/fail result. Ordinary compositional support is 0 because no row has lexicalized NO. The one grammatical-operator gloss also has an unrelated co-lemma, so lexicalized state is conflict and the class is ambiguous. + +Referential yes remains an instance-hypernym pointer. All 27 referential rows use that pointer. Descriptive reject precision is 27/27 and recall is 27/77. False reject is 0. One of those 27 was quarantined by procedure v1 because the instance pointer shared the synset with a one-token alternation. Procedure v2 keeps the instance pointer decisive and records the alternation as compositional yes. Secondary precision is 7/16 and recall is 7/46. False secondary is 9 and is descriptive only. High precision is not computable, because no row was predicted high. High recall is 0/101. + +Evidence states: referential yes 27, no 97, unknown 101. Lexicalized yes 95, no 0, unknown 129, conflict 1. Compositional yes 43, no 0, unknown 182, conflict 0. `as far as possible` is referential no, lexicalized unknown, compositional yes, and ambiguous. `road to damascus` is unknown on all three states and ambiguous. Neither illustration was written onto the manifest. + +Prediction sha256 `1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a`. Report sha256 `93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5`. Tracker sha256 `6ecb7e16bbe2ea3be2efca51c46cc970f2a5ba8ec8cd9c101eaef8398f5a0e61`. Acceptance stays `cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0`. The v1 prediction and report hashes stay `69ea6b8714f3cb6105222d636af3f17bd5c5caac7b290c3c3d87e4efaeedd0ef` and `38ada8bc32d8b19361cc974346d5972f6020eb0c32c2ca537abff4d17f66c7f0`. The development manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. v3 through v7 sources and the v1 implementation are unchanged. + +These figures are development evidence. No accuracy threshold was applied. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. The next named transition is `V2_DEVELOPMENT_RESULT_REVIEW`. This pass does not authorize it, and it does not authorize an unseen measurement. + +## Semantic evidence source v1 specified — 2026-09-28 + +The procedure-v2 development result is reviewed and frozen. WordNet 3.0 structural evidence supports referentiality, whole-expression lexicalization, and some positive compositionality. Under the frozen procedure it does not provide a deterministic structural signal for semantic noncompositionality at useful coverage. That is a source-capability limitation. Procedure v2 was not retuned. + +Observed on the 225 development rows: compositional NO 0, lexicalized noncompositional 0, high 0, ambiguous 182/225, whole-expression lexicalization 95, positive compositionality 43, referential 27. Descriptive reject precision remains 27/27. False reject is 0. False high is 0. No measurement bar was applied. Review sha256 `77ae2c0491def0b75cd4213cc23fdcb6f2eec18dc2d0641764a276a583ee537d`. Limitation sha256 `3c05cd9d6301fab0791e31b542d767cc757307cf3e304065362b479cc40e964a`. Status is `WORDNET_STRUCTURAL_SOURCE_LIMITATION_CONFIRMED`. + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` is `SPEC_FROZEN`. It is an evidence-source evaluation framework, not an unbind classifier. The research question is which additional source can establish semantic noncompositionality independently of WordNet whole-expression lexicalization, with provenance and precision enough to become eligible input to a future sense-first screen. No source is selected. Nothing was encoded or applied. No external resource was downloaded. The only local lexical source is WordNet 3.0, and that source is the one whose limitation was just confirmed. + +Five candidate families are specified and left unselected: explicit lexical-semantic resources, a deterministic comparison of the bound sense with a composition of its parts, a curated linguistic annotation, a model-based judgment, and a later hybrid of WordNet structure plus one independent semantic source. An LLM completion is not runtime truth. A source must pass provenance, target alignment, sense alignment, label independence, reproducibility, abstention, a ban on phrase exceptions, development evaluation before integration, and isolation of any later generalization rows. The future output can be YES, NO, or UNKNOWN. UNKNOWN stays valid. No hard accuracy target is frozen. A high unknown rate is acceptable when false semantic claims stay low. Reducing ambiguity is not a reason to select a source. + +The same 225 rows remain development evidence for a future candidate test. They were not mutated and no source was run on them. False high and false reject stay the safety priority for a later integration specification. That specification does not exist yet, and no unseen-measurement bar was registered. + +A shared primitive, `RUNE.SEMANTIC_COMPOSITIONALITY`, is architecture only. The phrase lane would eventually use constituent words, phrase structure, and the bound sense. The single-lexeme lane would use morphemes or compound constituents and the bound sense. Lexicalized identity, compositionality, and semantic shift remain different questions. `RUNE.LEXEME_STRUCTURE_SCREEN.v1` was not implemented. Its architecture note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. + +No JSON Schema document exists for this source-evaluation family. Hypothesis sha256 `39127a810d38ede96d7947c33dbc3e5491c9e1cc9b3f76b1064d9e0dd04a7787`. Acceptance sha256 `1252c8c20ce3f49fe61ed8aeeec3157df7f4185b3aa7c468938ff47342d81b94`. Evaluation plan sha256 `472b3819c050bbc9b1dd2eec3183cdb27c3659521408c321acb12b9c1b69dc8a`. Architecture sha256 `180b6721c4e19847516364f441ecc2101ed9a7758643889673dfdb7be6f41d36`. Tracker sha256 `521eccb8bd068d4697a3ffdc06e3b44feb4eec3c8a3bd6aebcea11dd288b7dfc`. + +Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`, predictions `1f7fc03547d24de851326a4848d93f1dbef16714e74e3e9f86d8c8aa6f8aaa8a`, and report `93d8fb76da8aa7155fb0ce57b0841ca455eca3e904edf50b9f76e595dd095ca5`. Procedure v1, the sense-screen acceptance, and the 225-row manifest are unchanged. The encoded v1 screen stays `DEVELOPMENT_ANALYZED`. Procedure v2 stays `DEVELOPMENT_ANALYZED_V2`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. No procedure v3 was created. + +The next named transition is `CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. Selecting or integrating a source is a later decision. + +## MAGPIE candidate evaluation — 2026-09-28 + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` is `CANDIDATE_SOURCE_EVALUATED` for one candidate, MAGPIE. The candidate remains unselected. `selected_source` is `none`. Runtime integration is false. The encoded sense screen and procedure v2 were not modified. + +The evaluated artifact is the author corpus `MAGPIE_unfiltered.jsonl` from `https://github.com/hslh/magpie-corpus` at commit `7fa677b82b9a772dfa54bbdd0fb414412d73db3b` (2020-06-07T09:57:12Z). The file was acquired from that commit on 2026-09-28T05:50:00Z. It contains 56,622 instances and 1,756 potentially idiomatic expression types. Assigned labels are idiomatic 40,011, literal 16,168, other 436, and unclear 7. The filtered split files were not acquired and were not mixed into the evaluation. No Hugging Face package was used. No other repository named magpie was used. + +Licenses are recorded per artifact. The LREC 2020 PDF, sha256 `8247c926909772ce317d5f33cddb83caa51969eb5ef928bcbadbbcf05e39c979`, states on page 1 that the ELRA proceedings text is licensed under CC-BY-NC. That statement is the publication license. The dataset LICENSE file in the pinned commit is Creative Commons Attribution 4.0 International, sha256 `05ab88f3f9da1d05f9c5bf0a7c45c49a9007f877dd9c237a5bf668276fe04c3b`. That file is also the only code license in the commit. The evaluated jsonl sha256 is `541ee535e93d71eff85351351665115e2a9f22ad736423881da5774a93bc880e`. Provenance sha256 `bf0dd1dd747a6423406d97393f375f99620894a20af6bd4758e2de738d5c82dd`. License receipt sha256 `8813818aa3704ba1e764121d2f66ff1630862a66d6c0c0b959b6afa35e0c3972`. + +Surface comparison is orthographic. Case and separator folding produced 25 exact matches. One normalized match folds a comma: `day in day out` corresponds to the MAGPIE type `day in, day out`. `to a t` casefolds onto `to a T` and is counted as exact. Recorded variant categories such as inflection, dashes, and possessive describe occurrences of a type. They are not alternate expression strings, so variant matches are 0. Ambiguous collisions are 0. Unmatched rows are 199. Surface coverage is 26/225. + +Sense alignment requires equality between a source sense identifier and the supplied WordNet synset. The pinned instances have no synset, sense key, or gloss field. Alignment coverage is 0/225. Every row is UNKNOWN. Semantic noncompositionality is YES 0, NO 0, UNKNOWN 225. Abstention is 225/225. Evidence codes are `magpie_surface_only` 26 and `magpie_no_match` 199. A surface hit does not assign YES. Fifteen matched rows have only idiomatic instance labels, including `throw in the towel` and `dressed to the nines`, and they stay UNKNOWN. Ten matched types contain both literal and idiomatic instances, including `round the bend` and `to the letter`. Those counts stay on the match record. They are not a MIXED sense alignment and they are not semantic YES. `as luck would have it` is an exact match with only idiomatic labels, and the historical operator bucket is secondary. The idiomatic labels were not converted into YES. + +Operator labels were joined after the semantic evidence file was written. They are not runtime evidence. The descriptive proxy, stated as a proxy, treats operator HIGH as the comparison class for semantic YES and operator SECONDARY as the comparison class for semantic NO. YES precision is `NOT_COMPUTABLE`. NO precision is `NOT_COMPUTABLE`. False YES is 0. False NO is 0. Both counts are zero because no YES or NO claim was emitted. Operator REJECT is its own row in the joint table: 77 unknown. It was not read as compositional NO or as semantic YES. The joint table is HIGH 101 unknown, SECONDARY 46 unknown, REJECT 77 unknown, and quarantine 1 unknown. + +Development status is `CANDIDATE_INSUFFICIENT`. Coverage of familiar phrases is not a reason to select the source. The readiness question, non-zero semantic YES with a defensible sense alignment, is not met. Match sha256 `84c847cfa545883de5a31979133fed87b0cdf9a7d13074c74cf227d9bfcadc83`. Sense-alignment sha256 `037b0f4d96d463aa7c5fbecdbef06a530ffbf770735232c92bd6abd0dd71fc66`. Semantic-evidence sha256 `89f7227e1098407c7aaae6d9876b1f5780dbfe5b6a2eb3c3b09357d57822b577`. Evaluation sha256 `74e2174d15c486dc60e9ad6be338199ff66a2d0950ed268105119323711108ba`. Decision sha256 `6eaa968b6260946998dba13e5c423f178d3349cdfe06e5ea401717f5a9bcdd0d`. Tracker sha256 `c3fe6f2fd21d07b8b45d9f26ffeebbcfbe3cb68a1f5a7952dee94d9a2c995936`. + +No JSON Schema document exists for this source-evaluation family. The schema names on the artifacts name those artifacts. Older unbind sample, label, and report schemas were not applied. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. No sense class was written onto it. v3 through v7 sources are unchanged. The lexeme-structure note stays `529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0`. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. A later authorization may name one curated resource whose entries carry a WordNet synset offset or sense key for the expression sense. PARSEME, STREUSLE, PIE, EPIE, and NCS were not downloaded and are not selected. + +## Korkontzelos–Manandhar candidate evaluation — 2026-09-28 + +`RUNE.UNBIND_SEMANTIC_EVIDENCE_SOURCE.v1` stays `CANDIDATE_SOURCE_EVALUATED`. A second candidate, the Korkontzelos–Manandhar 2009 compositionality set, was evaluated and not selected. `selected_source` remains `none`. Runtime integration remains false. MAGPIE stays `CANDIDATE_INSUFFICIENT`. Its artifacts were not modified. + +The pinned source is Table 1 of Korkontzelos and Manandhar, “Detecting Compositionality in Multi-Word Expressions,” ACL-IJCNLP 2009 short papers, pages 65–68, anthology `P09-2017`. Canonical URL `https://aclanthology.org/P09-2017/`. PDF sha256 `046da9fc26cfdf220e41ad914314f0703ea3c84cd86e045b20146b34189113d1`, acquired 2026-09-28T06:41:05Z. The paper states that the items were drawn from WordNet 3.0. The bibliography cites Miller 1995 for WordNet in general. That citation is not a second version number. Reconstruction used the local Princeton WordNet 3.0 data files and no other WordNet release. The suggested counts of about 60 compositional and 56 noncompositional items do not match this table. The verified inventory is 19 compositional and 19 noncompositional. A later 2010 sample by the same authors was not acquired. + +The PDF header reads `c©2009 ACL and AFNLP`. The ACL Anthology states that materials prior to 2016 are licensed under CC-BY-NC-SA 3.0 and that copies may be made for teaching and research. That is the publication license of the acquired PDF. No separate dataset license exists. The evaluation list is the table inside the paper. No code artifact was published with the table, and none was acquired. Bold, underline, and italic marks in the table report system detections. They were not read as gold labels. + +Table 1 preserves the surface and the section label. It does not preserve a synset offset, sense key, lemma key, part of speech, or gloss. Source sense identity is therefore absent on all 38 items. A lookup key folds case, underscore, and apostrophe shape, and it keeps hyphens. That key is not a sense choice. The key matches exactly one PWN 3.0 synset for 30 items: `EXACT_UNIQUE_RECONSTRUCTION`. Those 30 are the 19 compositional items and 11 noncompositional items. Eight noncompositional items match two synsets each: `black maria`, `dead end`, `dutch oven`, `goat's rue`, `green light`, `high jump`, `living rock`, and `prince Albert`. They are `AMBIGUOUS_MULTIPLE_SYNSETS`. No gloss, operator label, or frequency was used to pick one. `EXACT_SOURCE_ID` is 0. `NO_PWN3_MATCH` is 0. `VERSION_CONFLICT` is 0. On the inventory, the conservative map gives semantic YES 11, NO 19, and UNKNOWN 8. Only the unique reconstructions carry YES or NO. + +None of those 30 synsets occurs in the frozen 225 development rows. Surface keys also do not meet any development row. The join is synset identity, so the Hyperlex result is YES 0, NO 0, UNKNOWN 225. Sense-aligned coverage is 0/225. Abstention is 225/225. Evidence codes on the 225 rows are `km_no_match` 225. Operator labels were joined after the evidence file was written. Under the stated proxy, YES precision and NO precision are `NOT_COMPUTABLE`. False YES is 0. False NO is 0. The joint table is HIGH 101 unknown, SECONDARY 46 unknown, REJECT 77 unknown, and quarantine 1 unknown. Reject and quarantine were not read as compositionality. + +Development status is `CANDIDATE_INSUFFICIENT`. The inventory can reconstruct monosemous PWN 3.0 identity, which MAGPIE could not, and that reconstruction still supplies no YES row on this development set. The stop-condition finding `WORDNET_DERIVED_BUT_SENSE_IDENTITY_NOT_PRESERVED` was not frozen. Eight of 38 items are polysemous, not most of them, and 30 items do reconstruct one synset. The development sample and this table are disjoint. MAGPIE remains surface coverage 26/225, sense-aligned coverage 0/225, YES 0, NO 0, abstention 225/225. This candidate is surface coverage 0/225, sense-aligned coverage 0/225, YES 0, NO 0, abstention 225/225. Coverage is not the comparison. Neither candidate puts a sense-aligned YES on a development row. + +Provenance sha256 `88fbfa077d2394b8ce631ec700c482ad98a0f62f0ab06964f4f43a77d906f9aa`. License receipt sha256 `eb4c9406aab7f9021d346ebd24634cad1dcf0e6076f069e73b1d4b2702dbe2f8`. Inventory sha256 `3a35ddc0b2c04a5386c6112a2bb3cdf22735edbe2fd791f0c2ec542fe9184d81`. PWN 3.0 alignment sha256 `531d13cf1bdbc939fc11d9ef5864c6878a9f58210368c596c0aad08ff75e61c9`. Hyperlex semantic evidence sha256 `4e98f8f06713ffcf02549305aef140e92b9b8b2790f471f3ce1f41fe208b2f7a`. Evaluation sha256 `00a1d1f667635354e20e5002c4ece846fe3a8125a7ca12ebe09bb7e28dedd1a7`. Comparison sha256 `240b3ea468d22a80ac5e5765521cc691b80f914b7baff6bb1ae4819475f54965`. Decision sha256 `93a07e3c78b53a69965497410c34ddb52c2a5d3add3fb2f3cd3fd9ca84eb3d9f`. Tracker sha256 `b3546102410058d3596c4563604998685753a698b2bc533b353e1f039440f704`. + +No JSON Schema document exists for this source-evaluation family. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. v3 through v7, the lexeme-structure note, and the WordNet source-limitation finding are unchanged. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. A later candidate has to carry a PWN synset identity that can meet the development rows. Surface-only MWE corpora are not the next step. `CANDIDATE_SOURCE_SELECTION_REVIEW` is not the next step, because this candidate is not promising and is not selected. + +## Semantic compositionality residual candidate evaluation — 2026-09-28 + +`NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION` applies only to `RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1`. The candidate is an evaluation of a continuous residual. It is not selected, it is not integrated into `UNBIND_SENSE_SCREEN`, and it does not create procedure v3. Procedure v2 stays `3f4071640d0c9f29cf56f53969a88ec25c635444b87765e77e1b9158470e5662`. No unseen measurement sample is drawn. No training gold is created. No row is admitted or settled. `HLX-EXP-2026-09-27-SELECT-005` is not authorized. MAGPIE and the 2009 Korkontzelos–Manandhar artifacts stay frozen, and both remain `CANDIDATE_INSUFFICIENT`. + +The design was hashed before any development row was encoded. The whole sense is the supplied surface, its POS, and the frozen gloss, in the form `{surface} ({pos}): {gloss}`. A constituent uses the same form on the resolved PWN 3.0 lemma, POS, and first gloss clause. Whitespace tokenization keeps a hyphen inside one token. A frozen closed class of determiners, prepositions, pronouns, auxiliaries, and other structural tokens is ignored. A row needs two content tokens. Sense resolution accepts one derivation or pertainym pointer whose target word number is nonzero and whose lemma matches the constituent or a one-hop exception neighbor. That result is `EXACT`. With no exact target, one lexical synset across noun, verb, adjective, and adverb is `UNIQUE`. Several targets are `AMBIGUOUS`. None is `UNRESOLVED`. A target word number of zero does not name a lemma. The source word number is not a filter. An ambiguous or unresolved content constituent makes the row `UNKNOWN`. The pass does not use an arbitrary first sense, gloss similarity, an embedding to choose a sense, an operator label, or a manual sense choice. Composition is one operator, `normalized_mean_v1`: binary64 L2-normalize each constituent vector, take the mean, and L2-normalize the mean. Duplicate synsets stay, one vector per content token. The residual is `1 - cosine`, recorded with `format(value, '.10f')` and no clamp. A vector hash is sha256 of little-endian binary32 bytes. No semantic-noncompositionality threshold is chosen, and the residual is not turned into YES or NO. + +The pinned model is `sentence-transformers/all-MiniLM-L6-v2`, revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`, acquired locally at `2026-09-28T06:52:00Z` from the Hugging Face revision. The model license is Apache-2.0. Weights sha256 `53aa51172d142c89d9012cce15ae4d6cc0ca6895895114379cacb4fab128d9db`. `tokenizer.json` sha256 `be50c3628f2bf5bb5e3a7f17b1f74611b2561a3a27eeab05e5aa30f411572037`. `vocab.txt` sha256 `07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3`. Pooling is mean tokens into 384 dimensions, and the stack includes a normalize module. Encode uses CPU, float32, `normalize_embeddings=True`, batch size 1, seed 0, one thread, and eval mode. `use_deterministic_algorithms` stays false. The effective maximum sequence length is 256; overflow would abstain rather than truncate, and no row overflowed. Runtime versions are Python 3.12.3, torch `2.14.0+cpu` (Apache-2.0 with additional component licenses), sentence-transformers 6.1.0 (Apache-2.0), transformers 5.17.0 (Apache 2.0), tokenizers 0.23.2 (Apache), and NumPy 2.5.3 (BSD-3-Clause and other component licenses). ONNX, OpenVINO, and TensorFlow weight copies were not acquired. No API embedding service was called. Constituent glosses come from Princeton WordNet 3.0, whose license is separate from the model license. The encoder was run twice and the float32 hashes matched before the score file was written. + +Constituent extraction is `EXTRACTED` 166 and `UNKNOWN` 59. The 59 are rows with fewer than two content tokens. Across content constituents the resolution counts are `EXACT` 4, `UNIQUE` 60, `AMBIGUOUS` 370, and `UNRESOLVED` 70. Row abstentions are ambiguous content 145, fewer than two content tokens 59, and unresolved content 18. Three rows are `SCORED` and 222 are `UNKNOWN`. Ten representation texts were encoded. Six content tokens contain a character outside letters, digits, hyphen, and apostrophe; the extractor did not strip them. The three scored rows are operator `REJECT`: `california fern` residual `0.2139784896`, `monoamine oxidase inhibitor` residual `0.1768095281`, and `.22 caliber` residual `0.1309395496`. Reject is a referential axis, not semantic no. Their descriptive distribution is count 3, min `0.1309395496`, p25 `0.1538745388`, median `0.1768095281`, p75 `0.1953940088`, max `0.2139784896`, mean `0.1739091891`. Operator `HIGH` has 0 scored rows of 101, so the primary comparison is `NOT_COMPUTABLE`. `SECONDARY` has 0 of 46. `QUARANTINE` has 0 of 1. No threshold was fit to these labels. + +Development status is `CANDIDATE_DISTRIBUTION_FROZEN` because the preregistered rule uses that status whenever the scored count is greater than zero. The status freezes the score distribution. It does not say the residual separates noncompositionality, and the empty HIGH distribution supplies no such comparison. `selected_source` remains `none`. Candidate spec sha256 `39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d`. Provenance sha256 `2c34a7d0d480cde564bda694dbaa349550814c5fb7c647bfa3bbbc9db5e26886`. License receipt sha256 `323a5bad21ac74d6a1fd94c54dba95a0046b633cd7328dd6680972525abc0909`. Score artifact sha256 `cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7`. Evaluation sha256 `ba782622d4c68d23c53ae0ffb5f54f1e43c8cf059b44b7adb34e0eb356ef3896`. Decision sha256 `45922eba294b0b7d7238ce71c6157641259ea292f65743ba2ee1324ff9b52617`. Tracker sha256 `14d4daaab1e4740a596acaef7f4dbd9ed11edaf2182ff22f7aa339325febf591`. Every score row carries the spec hash, and the spec contains no residual. Operator labels were joined only after the score file was hashed. + +No JSON Schema document exists for this source-evaluation family. The 225-row manifest stays `0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03`. Sense classes on those rows stay null. v3 through v7, the lexeme-structure note, and the WordNet source-limitation finding are unchanged. `measurement_sample_drawn` stays false. `measurement_eligible` stays false. `revision_eligible` stays false. Admitted 0. Settled 0. Gold 0. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION`. This pass does not authorize it. A threshold is not supported by a HIGH distribution, because no HIGH row was scored. `CANDIDATE_SOURCE_SELECTION_REVIEW` is not the next step. `selected_source` remains `none`. + +## Residual v1 coverage limitation — 2026-09-28 + +`RESIDUAL_DEVELOPMENT_RESULT_REVIEW` is authorized for the frozen residual distribution only. `RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION` is not authorized. The residual has not been falsified. The experiment could not test HIGH against SECONDARY because constituent sense ambiguity collapsed coverage. Scored rows are 3, all operator REJECT. Unknown rows are 222. HIGH scored 0 of 101. SECONDARY scored 0 of 46. REJECT scored 3 of 77. Ambiguous content abstains 145 rows, fewer than two content tokens abstain 59, and unresolved content abstains 18. `threshold_eligible` is false. `source_selection_eligible` is false. The primary limitation is `constituent_sense_resolution`. The finding is `SEMANTIC_RESIDUAL_V1_COVERAGE_LIMITATION_CONFIRMED`: the dominant blocker is deterministic constituent sense resolution, not the residual model. The model, revision, normalized mean, and `1 - cosine` residual stay frozen. No residual score was recomputed. + +## Constituent sense resolution v1 — 2026-09-28 + +`RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` starts from that limitation. The question is whether a frozen deterministic resolver can give enough content constituents a PWN 3.0 sense to make a later residual replay testable. Operator labels, residual scores, unbind buckets, phrase exceptions, manual choices, an LLM, and the residual embedding model are not inputs. The specification and procedure were hashed before the replay. The resolver then ran twice from that specification. The two artifacts matched, so determinism is `IDENTICAL`. + +Tier 1 is structural. A lexical pointer in the frozen set `! + \ ^ * & < $` whose target word number is nonzero and whose lemma matches the constituent, or one exception hop, is structural evidence. One such synset is `EXACT` by `STRUCTURAL_EXACT`. More than one is `AMBIGUOUS` and Extended Lesk does not override it. With no structural target, one lemma synset across noun, verb, adjective, and adverb is `EXACT` by `UNIQUE_LEMMA`. No synset is `UNRESOLVED`. The extra pointer symbols did not raise structural exact above the residual baseline: structural exact stays 4 and unique lemma stays 60. + +Extended Lesk v1 runs only when several lemma synsets remain. Candidate text is the full PWN 3.0 gloss, including quoted examples, plus the first gloss clause of each depth-1 synset reached by `@ + \ & ^ =`. Context is the parent frozen gloss plus the first gloss clause of other constituents in the row that tier 1 already resolved. Lesk output is not reused as context. Normalization casefolds, folds apostrophes, and splits on characters outside letters, digits, apostrophe, and hyphen. Stopwords are the residual v1 structural class. Tokens shorter than two characters are dropped. There is no stemmer and no lemmatizer. The score is the sum of squares of greedy longest contiguous overlaps, and matched tokens are not reused. Scores are integers. The minimum margin is 1, so a tie, including a zero-overlap tie, is `AMBIGUOUS`. A strict win is `RESOLVED`. The parent gloss is local context only. Parent sense is not constituent sense. + +Every one of the 504 content constituents was attempted. Status counts are `EXACT` 64, `RESOLVED` 121, `AMBIGUOUS` 249, and `UNRESOLVED` 70. Against the residual baseline of `EXACT` 4, `UNIQUE` 60, `AMBIGUOUS` 370, and `UNRESOLVED` 70, the 60 unique lemmas are the same senses under the new `EXACT` label, unresolved stays 70, and 121 of the 370 ambiguous constituents become `RESOLVED`. The other 249 stay `AMBIGUOUS`. The tie rate on Lesk attempts is 249/370. Exact proportion is 64/504. Context-resolved proportion is 121/504. Unresolved rate is 70/504. Abstention, ambiguous plus unresolved, is 319/504. There is no constituent-sense gold, so these figures are coverage, determinism, tie behavior, and provenance. They are not accuracy. + +A row is `RESIDUAL_READY` only when it has at least two content constituents and every one is `EXACT` or `RESOLVED`. That holds for 20 rows. The other 205 are `UNKNOWN`. After the resolution file was hashed, operator labels give HIGH 4 ready and 97 unknown, SECONDARY 4 ready and 42 unknown, REJECT 12 ready and 65 unknown, and quarantine 0 ready and 1 unknown. The necessary condition, at least one ready HIGH row and one ready SECONDARY row, is met. It is not sufficient. Because 249 of 370 baseline-ambiguous constituents remain AMBIGUOUS, which is more than half, the preregistered finding is `CONSTITUENT_WSD_COVERAGE_INSUFFICIENT`. No second resolver was added. A projection, not a replay, says 20 rows could enter a later residual pass: HIGH 4, SECONDARY 4, REJECT 12. The residual was not rerun. + +Resolver state is `DEVELOPMENT_ANALYZED`. Candidate status is `COVERAGE_INSUFFICIENT`. `selected_source` remains `none`. Residual state remains `CANDIDATE_DISTRIBUTION_FROZEN` with coverage limitation `CONFIRMED`. Limitation sha256 `fc8839c15a7638b2bfca1cf0548bfb4d5f433434bae0fea2944a528dd15d6142`. Review sha256 `1c1b69856dd88567167fd5c958cd8db6d39ab9ec74a4e0ed3e667a521c82e6fa`. Resolver spec sha256 `176e6219ddc3127814a25d39ad26e3571817f7ea8323d685e081d2e0fd867acb`. Procedure sha256 `9f76b64aa6aac09bd56ca9cc8a847cda12b31a58eea8c4b51426733917d54248`. Replay sha256 `a0c707ab55e02f627a698c33ddc0ca398e34aa0bafc19b13e72422bc26d97d0a`. Analysis sha256 `6d47610014f74094394355a50419fe24441103bec09458b6f25ea392fb57151c`. Decision sha256 `ff8b90ebe5d48151dc68ddbf676e1f27d4cee5e3be2730f999c9b089f1392caa`. Tracker sha256 `c72e55c8096143b8675d4aa995c5d4b19de95d256dd234d32a421ba098bc7d4f`. + +No JSON Schema document exists for this source-evaluation family. No residual threshold was created. No semantic YES or NO was emitted. No unseen sample was drawn. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. The 225-row manifest, procedure v2, MAGPIE, Korkontzelos–Manandhar, and the residual v1 artifacts are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION`. This pass does not authorize it. `RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION` is not the next step while the constituent-coverage finding stands. `selected_source` remains `none`. + +## Model-based constituent WSD candidate v1 — 2026-09-28 + +`MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION` evaluates one WordNet-native model on the 249 constituents that Extended Lesk v1 left `AMBIGUOUS`. The model is `kanishka/GlossBERT`, family GlossBERT, canonical source `https://huggingface.co/kanishka/GlossBERT`, revision `0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a`. It is a BERT sequence classifier fine-tuned on SemCor 3.0. Each decision is a score over a supplied Princeton WordNet 3.0 gloss, so the output is a candidate synset or an abstention. The card license is MIT. The checkpoint is a third-party Hugging Face upload, not the authors' Google Drive file. The original code repository `https://github.com/HSLCY/GlossBERT` is MIT. Inference is local. Device is CPU. Dtype is float32. Batch size is 1. One thread is used and mkldnn is disabled. Runtime is Python 3.12.3, torch 2.14.0+cpu, transformers 5.17.0, tokenizers 0.23.2, and numpy 2.5.3. The positive class is index 1. A non-Hyperlex polarity control fixed that index before any development constituent was scored: a financial sentence selects `noun:08420278`, and a river sentence selects `noun:09213565`. + +The candidate specification was hashed before those 249 constituents were scored. Spec sha256 `c861ff7fff11ae6a790531267229c18d6e6e0a171a9bf6c34cfb6f7e7b14498c`. Weight sha256 `60706c7618f8ccbfa7d0a6d1d1009765a7146ea5f4232924ed9f1c46d521c898`. Vocab sha256 `07eced375cec144d27c900241f3e339478dec958f92fddbc551f295c992038a3`. Tokenizer config sha256 `09e49d0e788d25991da77d37b10eaa6a86a4e94e2127de8bedc94eb45baf2d84`. Each constituent is scored only among its frozen PWN 3.0 candidate synsets. The context is the parent surface with the target content token in double quotes. The paired text is the matched lemma, a colon, and the first gloss clause. The parent gloss is not appended. Operator labels, residual scores, and the residual embedding are not inputs. The score is the class-1 softmax probability, quantized to six decimal places, half even. The abstention rule, frozen before scoring, requires a top probability of at least 0.50 and a top-versus-second margin of at least 0.10. An equal top score stays `AMBIGUOUS`. A sense outside the candidate set would be `INVALID`. The same inference then ran a second time. The two raw artifacts matched, so determinism is `IDENTICAL`. + +Tier 3 attempted 249 constituents. It resolved 153, left 96 ambiguous, and produced 0 invalid outputs and 0 errors. The 153 resolutions use evidence `model_margin`. The 96 abstentions use evidence `model_abstention`. Exact score ties are 0. Top-score bins are 81 below 0.50, 30 from 0.50 to 0.60, 35 from 0.60 to 0.70, 26 from 0.70 to 0.80, 32 from 0.80 to 0.90, and 45 from 0.90 to 1.00. Margin bins are 50 below 0.10, 56 from 0.10 to 0.25, 56 from 0.25 to 0.50, and 87 at 0.50 or more. There is no constituent-sense gold on these rows, so the figures are coverage, validity, determinism, confidence, and abstention. They are not accuracy, precision, recall, or F1. + +Combined with the frozen earlier tiers, constituent counts are `EXACT` 64, `LESK_RESOLVED` 121, `MODEL_RESOLVED` 153, `AMBIGUOUS` 96, and `UNRESOLVED` 70. Tier 3 did not replace a structural exact, a unique lemma, or an Extended Lesk decision that had already cleared its margin. A row is projected `RESIDUAL_READY` only when it has at least two content constituents and every one is exact, Lesk-resolved, or model-resolved. That projection was hashed before operator labels were joined. Projected ready rows are 73. Unknown rows are 152. After the join, HIGH is 28 ready and 73 unknown, SECONDARY is 11 ready and 35 unknown, REJECT is 34 ready and 43 unknown, and quarantine is 0 ready and 1 unknown. Against the lexical baseline of HIGH 4, SECONDARY 4, and total 20, the deltas are +24, +7, and +53. Invalid outputs are 0 and the repeat is identical, so the preregistered coverage gate returns `CANDIDATE_PROMISING`. The gate is a readiness projection. It does not say the selected senses are correct, and it does not replay the residual. + +`selected_source` remains `none`. The model is not integrated into runtime. `RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` stays `DEVELOPMENT_ANALYZED` with lexical status `COVERAGE_INSUFFICIENT`. `RUNE.SEMANTIC_COMPOSITIONALITY_RESIDUAL.v1` stays `CANDIDATE_DISTRIBUTION_FROZEN`. `threshold_eligible` stays false. No residual score was recomputed and no residual threshold was created. No semantic YES or NO was emitted. No unseen sample was drawn. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. The resolver artifacts, residual artifacts, MAGPIE artifacts, Korkontzelos–Manandhar artifacts, procedure v2, and the 225-row manifest are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +No JSON Schema document exists for this source-evaluation family. Provenance sha256 `6d91283244a88f44477c836e6549118d5d96a36bb878bf448fcadb13ff765e11`. License receipt sha256 `67d45243959e3e76643a675d49b508a73d0f7f0bf8fbf060ce7b83c9200c621b`. Raw output sha256 `a0e508c225e6db4cdbcae701682202b1d854f3762546dbfe10b84a15e0e9e17c`. Resolution sha256 `ed945989cf4947ac84633ba2c4aa10c1ba381d2396da0b573a844f83ec367a18`. Readiness projection sha256 `c75834faf4a84d36e83246244e0aa7c6c7788c3a57cfdb7f77c7628a52023328`. Analysis sha256 `8ca0d8c8dd7e510a04daef6b3fbd78ace6d20b173e88d2a3d0c2e10fdaafe30b`. Decision sha256 `c8ed0b8acaaa415277c5f9bdbf982d075136b795b5d95fd8813d5a3010072dbb`. Tracker sha256 `6da1e9730c785d2784d23433455515989d312ed8c76f4763b3a2dd6a1f2c4b2d`. + +The next named transition is `RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION`. This pass does not authorize it. The residual is not replayed. `selected_source` remains `none`. +## Residual replay with model-resolved senses — 2026-09-28 + +`RESIDUAL_REPLAY_WITH_MODEL_RESOLVED_SENSES_AUTHORIZATION` runs one residual replay on the frozen Tier 1, Tier 2, and Tier 3 constituent senses. The question is whether operator-HIGH rows then show larger frozen semantic-compositionality residuals than operator-SECONDARY rows. This is a development distribution. It is not a classifier, and it does not train a composition function. The residual model stays `sentence-transformers/all-MiniLM-L6-v2` at revision `1110a243fdf4706b3f48f1d95db1a4f5529b4d41`. Settings stay CPU, float32, seed 0, batch size 1, one thread, eval mode, normalized embeddings, and maximum sequence length 256. Composition stays `normalized_mean_v1`. The residual stays `1 - cosine_similarity(whole_sense_vector, normalized_mean(constituent_sense_vectors))`, stored at 10 decimal places. Constituent extraction, structural exact, unique lemma, Extended Lesk v1, and GlossBERT are not retuned. No confidence or margin rule changes. No second embedding, pooling, distance, or weight scheme is tried. + +The integrated resolution is built from the frozen resolver replay and the frozen GlossBERT resolution, then checked against the frozen readiness projection before any vector is computed. The projection reproduces: 73 `RESIDUAL_READY` rows and 152 `UNKNOWN` rows. The integrated file is hashed before scoring. Integrated sha256 `0f5dafc3676a4071ce8c889e58958b90203589aa3e91d78111b4e3292bdd87fb`. Only ready rows are scored. Representation text, lemma choice, gloss fields, example inclusion, normalization, and the residual formula are the frozen residual v1 functions. The encoder then runs twice. Row readiness, representation texts, vector hashes, and 10-decimal scores match, so determinism is `IDENTICAL`. No ready row exceeds 256 tokens. Scored rows are 73 and scoring abstentions are 0. The score file is hashed before operator labels are read. The receipt records that sequence with `operator_labels_joined` false. Score sha256 `16c0a9eaa918ac4a6e8223cafcbf4b1918cb212769063e262cf29a279f1048f6`. Receipt sha256 `675b1b8b8f1b1e6f19c7d320e7b8fe516eae92b60e966afbce407c1f0482e736`. + +After that hash, the historical operator labels join. Ready counts are HIGH 28, SECONDARY 11, REJECT 34, and quarantine 0. REJECT stays out of the primary comparison. HIGH residuals have count 28, minimum 0.1169573790, p10 0.2930005820, p25 0.3564342144, median 0.4040327275, p75 0.4404865614, p90 0.5184032344, maximum 0.5908804826, mean 0.3943616308, and sample standard deviation 0.1017789927. SECONDARY residuals have count 11, minimum 0.1768283745, p10 0.2079496807, p25 0.2436642596, median 0.3307873412, p75 0.4000929338, p90 0.4561366947, maximum 0.5587792172, mean 0.3299980313, and sample standard deviation 0.1186719529. The HIGH mean exceeds the SECONDARY mean by 0.0643635995. The HIGH median exceeds the SECONDARY median by 0.0732453863. Mann-Whitney U for HIGH, with half credit for ties, is 207. There are 207 pairs in which the HIGH residual is larger, 101 in which it is smaller, and 0 ties, out of 308 pairs. The rank-biserial correlation is 0.344156. Descriptive ROC AUC, with HIGH as the positive class and a larger residual as the HIGH-like score, is 0.672078. The hypothesized direction `HIGH residual > SECONDARY residual` is `SUPPORTED_DIRECTION` on these point estimates. That direction is not a semantic YES or NO. + +Uncertainty uses the bootstrap frozen before the label join: seed 0 and 10000 resamples, with interpolated 2.5 and 97.5 percentiles. The AUC interval is 0.451299 to 0.870130. The mean-difference interval is -0.0155777495 to 0.1373246447. The median-difference interval is -0.0445612032 to 0.1742972287. Each interval includes a null or reversed value. The preregistered status rule does not require the interval to exclude the null, and no stronger AUC floor is added after seeing the scores. + +Tier 3 dependence, among ready rows, is HIGH with Tier 3: 24, HIGH without Tier 3: 4, SECONDARY with Tier 3: 7, and SECONDARY without Tier 3: 4. The no-Tier-3 group has HIGH n=4 and SECONDARY n=4, median difference 0.1435511609, rank-biserial 0.500000, and AUC 0.750000, direction `SUPPORTED_DIRECTION`. The Tier-3 group has HIGH n=24 and SECONDARY n=7, median difference 0.0595389352, rank-biserial 0.083333, and AUC 0.541667, also `SUPPORTED_DIRECTION`. The gap is not concentrated in the GlossBERT subgroup. The within-Tier-3 separation is small. Dropping the largest HIGH residual and the smallest SECONDARY residual does not remove the overall directional result, so the comparison is not extreme-driven. HIGH and SECONDARY are not confined to disjoint parts of speech. The only part of speech with at least three rows in each class is adverb: HIGH 5 and SECONDARY 6, median difference 0.0797654836, AUC 0.733333. Adjective, noun, and verb cells are `NOT_COMPUTABLE`. Whitespace token count has HIGH median 4 and SECONDARY median 3, with Spearman against the residual of 0.092940 for HIGH and 0.603202 for SECONDARY. Character length has HIGH median 17 and SECONDARY median 12, with Spearman -0.059115 and 0.165145. Content-constituent count is 2 for every SECONDARY row and for almost every HIGH row. Maximum candidate-synset count has Spearman 0.390077 for HIGH and 0.073060 for SECONDARY. Where GlossBERT supplies a constituent, minimum confidence and minimum margin have Spearman near 0 for HIGH. These diagnostics do not retune the residual. + +REJECT is reported separately and is not semantic compositionality NO. REJECT count is 34, minimum 0.1036592522, p10 0.1240175180, p25 0.1569692084, median 0.2174387191, p75 0.2995541264, p90 0.3474576871, maximum 0.5180572821, mean 0.2343664094, and sample standard deviation 0.0963240560. The largest HIGH residual is `on the other hand` (`adv:00119578`, row `1d19d96bfedbade746610820599e28166b4ef6c291fd4d186b95055e0cf22074`) at 0.5908804826, tiers GlossBERT then Extended Lesk. The smallest HIGH residual is `naked as a jaybird` (`adj:00458266`, row `d479d3da8dd2b87a7b30c9710abefd1e87f2706a7ac032dde17fc337bef8ac4c`) at 0.1169573790, tiers Extended Lesk then unique lemma. The largest SECONDARY residual is `at one time` (`adv:00153261`, row `0f1c15024e09410a3336e3910351d2cfe6143b0a98b88ab561b8dedcbe64a5fc`) at 0.5587792172, both constituents GlossBERT. The smallest SECONDARY residual is `sneak thief` (`noun:10616204`, row `00152611a0b327fd51c12b1c9a80ab838874fb68692c7e988e7970e41557c9bb`) at 0.1768283745, tiers Extended Lesk then unique lemma. The largest REJECT residual is `three times` (`adv:00476680`, row `0e17c819392a00600e3089202fef9b2e74036b73b6429724dfe91d14f7334b2b`) at 0.5180572821, both Extended Lesk. The smallest REJECT residual is `family ascaphidae` (`noun:01644542`, row `00181b47ec5143e4625569c7b41909401ccb940250f8138c775be0000ec3ff93`) at 0.1036592522, tiers GlossBERT then unique lemma. These rows are descriptive. They do not become phrase rules. + +All six preregistered conditions hold: readiness reproduces, the replay is deterministic, the HIGH median is larger, the rank separation is positive, AUC is above one half, and the separation is not solely one preregistered confound. Residual development state is `RESIDUAL_DEVELOPMENT_ANALYZED_V2`. Candidate status is `CANDIDATE_PROMISING`. `threshold_eligible` stays false because these 225 rows remain a reused development surface. No threshold is frozen. No row receives `semantic_noncompositional` YES or NO. The original residual state stays `CANDIDATE_DISTRIBUTION_FROZEN`. `MODEL_BASED_WSD_CANDIDATE_V1` stays `CANDIDATE_PROMISING`. `RUNE.CONSTITUENT_SENSE_RESOLUTION.v1` stays `DEVELOPMENT_ANALYZED` with lexical status `COVERAGE_INSUFFICIENT`. `selected_source` remains `none`. There is no runtime integration. `measurement_sample_drawn` and `measurement_eligible` stay false. SELECT-005 stays unauthorized. Admitted 0. Settled 0. Gold 0. + +No JSON Schema document exists for this replay family. Original residual spec sha256 remains `39c2914e32557ffe1a456a56f8742ea4fe8f1aaec1dc1da451656cd22f0db32d`. Original residual scores sha256 remains `cea638679faeee1bc1c689823e7c0c08562c4d7ef1f8230bbbf4079239e7c3e7`. Analysis sha256 `4367930648a68c2f84a1fd8e011fa07d9f3bf111303079f3ec07688f7d5425fb`. Confound analysis sha256 `d0f2496ca7623050eb5519969abcf7c5e2d0e23e0c1961859c40cae4dcdb7021`. Decision sha256 `ed0296fe6888e7c9fe864a2c6c7ab6cecbd4f490d6b7d650e442dafc0bd0976d`. Tracker sha256 `6500394d24543d1797eb9a2a05ba31868e035ffcbc7ae568f53116d4b016e62a`. The 225-row manifest, procedure v2, GlossBERT artifacts, resolver artifacts, MAGPIE artifacts, and Korkontzelos–Manandhar artifacts are unchanged. The ledger was not appended. Events sha256 remains `96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c`. + +The next named transition is `RESIDUAL_THRESHOLD_PREREGISTRATION_AUTHORIZATION`. This pass does not authorize it. A threshold still requires a later preregistered stage. `selected_source` remains `none`. diff --git a/specs/007-hyperlexical-model/schemas/unbind_screen_operator_label.v1.schema.json b/specs/007-hyperlexical-model/schemas/unbind_screen_operator_label.v1.schema.json new file mode 100644 index 00000000..73821dab --- /dev/null +++ b/specs/007-hyperlexical-model/schemas/unbind_screen_operator_label.v1.schema.json @@ -0,0 +1,48 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "hyperlex.unbind_screen_operator_label.v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema", + "evaluation_id", + "row_id", + "operator_bucket", + "reason_code", + "labeled_at" + ], + "properties": { + "schema": { + "const": "hyperlex.unbind_screen_operator_label.v1" + }, + "evaluation_id": { + "type": "string" + }, + "row_id": { + "type": "string", + "minLength": 64, + "maxLength": 64 + }, + "operator_bucket": { + "enum": [ + "HIGH", + "SECONDARY", + "REJECT", + "QUARANTINE", + "UNRESOLVED" + ] + }, + "reason_code": { + "type": "string" + }, + "note": { + "type": [ + "string", + "null" + ] + }, + "labeled_at": { + "type": "string" + } + } +} diff --git a/specs/007-hyperlexical-model/schemas/unbind_screen_report.v1.schema.json b/specs/007-hyperlexical-model/schemas/unbind_screen_report.v1.schema.json new file mode 100644 index 00000000..d7f6c5c8 --- /dev/null +++ b/specs/007-hyperlexical-model/schemas/unbind_screen_report.v1.schema.json @@ -0,0 +1,51 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "hyperlex.unbind_screen_report.v1", + "type": "object", + "required": [ + "schema", + "status", + "select_authorized", + "admitted", + "settled", + "gold" + ], + "properties": { + "schema": { + "const": "hyperlex.unbind_screen_report.v1" + }, + "status": { + "enum": [ + "NOT_COMPUTABLE", + "SCORED" + ] + }, + "select_authorized": { + "const": false + }, + "revision_eligible": { + "type": "boolean" + }, + "admitted": { + "const": 0 + }, + "settled": { + "const": 0 + }, + "gold": { + "const": 0 + }, + "held_out_precision": {}, + "confusion_matrix": {}, + "overall_accuracy": {}, + "per_bucket": { + "type": "object" + }, + "error_classes": { + "type": "object" + }, + "unresolved_count": { + "type": "integer" + } + } +} diff --git a/specs/007-hyperlexical-model/schemas/unbind_screen_sample.v1.schema.json b/specs/007-hyperlexical-model/schemas/unbind_screen_sample.v1.schema.json new file mode 100644 index 00000000..25c39d51 --- /dev/null +++ b/specs/007-hyperlexical-model/schemas/unbind_screen_sample.v1.schema.json @@ -0,0 +1,46 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "hyperlex.unbind_screen_sample_row.v1", + "type": "object", + "additionalProperties": false, + "required": [ + "schema", + "evaluation_id", + "sample_id", + "row_id", + "surface", + "pos", + "token_count", + "provenance" + ], + "properties": { + "schema": { + "const": "hyperlex.unbind_screen_sample_row.v1" + }, + "evaluation_id": { + "type": "string" + }, + "sample_id": { + "type": "string" + }, + "row_id": { + "type": "string", + "minLength": 64, + "maxLength": 64 + }, + "surface": { + "type": "string" + }, + "pos": { + "type": "string" + }, + "token_count": { + "type": "integer", + "minimum": 2, + "maximum": 6 + }, + "provenance": { + "type": "object" + } + } +} diff --git a/tests/shadow/test_constituent_sense_resolution_v1.py b/tests/shadow/test_constituent_sense_resolution_v1.py new file mode 100644 index 00000000..939f182e --- /dev/null +++ b/tests/shadow/test_constituent_sense_resolution_v1.py @@ -0,0 +1,157 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.constituent_sense_resolution_v1 import ( + AMBIGUOUS, + EXACT, + RESOLVED, + coverage_decision, + lesk_tokens, + overlap_score, + resolve_constituent_sense, + resolver_policy, + row_resolution_status, + structural_synset_ids, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "constituent_sense_resolution_v1.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def test_module_has_no_phrase_exception_embedding_or_label_input(): + source = MODULE.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] + assert "sentence_transformers" not in source + assert "SentenceTransformer" not in source + names = set(inspect.signature(resolve_constituent_sense).parameters) + assert "operator_bucket" not in names + assert "residual_score" not in names + assert "gloss" not in names + policy = resolver_policy() + assert policy["state_at_freeze"] == "SPEC_FROZEN" + assert policy["encoded_at_freeze"] is False + assert "residual_scores" in policy["forbidden_inputs"] + assert "operator_labels" in policy["forbidden_inputs"] + assert policy["extended_lesk"]["minimum_margin"] == 1 + assert policy["extended_lesk"]["stemming"] is False + + +def test_structural_pointer_outranks_a_higher_lesk_score(): + pointers = [ + ("+", 1, "noun:1", "towel"), + ("!", 1, "noun:2", "other"), + ("+", 0, "noun:9", "towel"), + ("@", 1, "noun:8", "towel"), + ] + structural = structural_synset_ids(pointers, "towel", {}) + assert structural == ["noun:1"] + decision = resolve_constituent_sense( + structural, + ["noun:1", "noun:2"], + [("noun:2", 99), ("noun:1", 0)], + ) + assert decision["resolution_status"] == EXACT + assert decision["resolution_method"] == "STRUCTURAL_EXACT" + assert decision["selected_synset"] == "noun:1" + assert decision["top_score"] is None + conflict = structural_synset_ids( + [("+", 1, "noun:1", "shot"), ("<", 1, "noun:2", "shot")], + "shots", + {"shots": {"shot"}, "shot": {"shots"}}, + ) + blocked = resolve_constituent_sense(conflict, ["noun:1", "noun:2"], [("noun:1", 5), ("noun:2", 1)]) + assert blocked["resolution_status"] == AMBIGUOUS + assert blocked["selected_synset"] is None + assert blocked["resolution_method"] == "STRUCTURAL_EXACT" + + +def test_unique_lemma_is_exact_and_absence_stays_unresolved(): + one = resolve_constituent_sense([], ["noun:4"], None) + assert one["resolution_status"] == EXACT + assert one["resolution_method"] == "UNIQUE_LEMMA" + assert one["selected_synset"] == "noun:4" + missing = resolve_constituent_sense([], [], [("noun:1", 3)]) + assert missing["resolution_status"] == "UNRESOLVED" + assert missing["resolution_method"] == "NONE" + assert missing["selected_synset"] is None + + +def test_extended_lesk_scores_squares_and_ties_abstain(): + assert lesk_tokens("The edible, swollen root!") == ["edible", "swollen", "root"] + assert overlap_score(["edible", "swollen", "root"], ["swollen", "root"]) == 4 + assert overlap_score(["edible", "root"], ["edible", "root"]) == overlap_score( + ["edible", "root"], ["edible", "root"] + ) + resolved = resolve_constituent_sense( + [], + ["noun:1", "noun:2"], + [("noun:2", 1), ("noun:1", 4)], + ) + assert resolved["resolution_status"] == RESOLVED + assert resolved["resolution_method"] == "EXTENDED_LESK_V1" + assert resolved["selected_synset"] == "noun:1" + assert resolved["top_score"] == 4 + assert resolved["second_score"] == 1 + assert resolved["margin"] == 3 + tie = resolve_constituent_sense([], ["noun:2", "noun:1"], [("noun:1", 4), ("noun:2", 4)]) + assert tie["resolution_status"] == AMBIGUOUS + assert tie["selected_synset"] is None + assert tie["primary_evidence_code"] == "extended_lesk_tie" + assert tie["margin"] == 0 + zeros = resolve_constituent_sense([], ["noun:1", "noun:2"], [("noun:1", 0), ("noun:2", 0)]) + assert zeros["resolution_status"] == AMBIGUOUS + assert zeros["selected_synset"] is None + + +def test_row_readiness_keeps_the_two_content_rule(): + assert row_resolution_status(2, [EXACT, RESOLVED]) == "RESIDUAL_READY" + assert row_resolution_status(2, [EXACT, AMBIGUOUS]) == "UNKNOWN" + assert row_resolution_status(1, [EXACT]) == "UNKNOWN" + assert row_resolution_status(0, []) == "UNKNOWN" + + +def test_coverage_rule_is_fixed_before_any_label_join(): + insufficient = coverage_decision( + baseline_ambiguous=10, + baseline_ambiguous_not_resolved=6, + residual_ready_high=1, + residual_ready_secondary=1, + ) + assert insufficient["coverage_finding"] == "CONSTITUENT_WSD_COVERAGE_INSUFFICIENT" + assert insufficient["candidate_status"] == "COVERAGE_INSUFFICIENT" + assert insufficient["next_transition_authorized"] is False + assert insufficient["necessary_condition_met"] is True + half = coverage_decision( + baseline_ambiguous=10, + baseline_ambiguous_not_resolved=5, + residual_ready_high=1, + residual_ready_secondary=1, + ) + assert half["coverage_finding"] is None + assert half["candidate_status"] == "COVERAGE_NECESSARY_CONDITION_MET" + unmet = coverage_decision( + baseline_ambiguous=10, + baseline_ambiguous_not_resolved=5, + residual_ready_high=0, + residual_ready_secondary=4, + ) + assert unmet["candidate_status"] == "NECESSARY_CONDITION_UNMET" + assert unmet["necessary_condition_met"] is False diff --git a/tests/shadow/test_hyperlexical_screen_eval.py b/tests/shadow/test_hyperlexical_screen_eval.py new file mode 100644 index 00000000..dfb4ce55 --- /dev/null +++ b/tests/shadow/test_hyperlexical_screen_eval.py @@ -0,0 +1,219 @@ +import json +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.holdout_guard import normalized_text_sha256 +from hyperlexical.screen_eval import ( + ScreenEvalError, + file_sha256, + freeze_labels, + materialize, + score, +) + +FROZEN_AT = "2020-01-01T00:00:00Z" +LABELED_AT = "2020-01-02T00:00:00Z" + + +def _frozen_row(text, predicted, rule, pos="verb"): + tokens = text.split(" ") + return { + "text": text, + "source_pos": pos, + "tokens": tokens, + "predicted": predicted, + "rule": rule, + "gloss": f"gloss for {text}", + } + + +def _write_jsonl(path, rows): + path.write_text("".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8") + + +def _layout(tmp_path): + frozen = tmp_path / "frozen.jsonl" + rows = [ + _frozen_row("alpha beta", "HIGH_VALUE", "conventionalized_semantic_shift"), + _frozen_row("gamma delta", "SECONDARY", "transparent_or_moderate", pos="adj"), + _frozen_row("epsilon zeta", "REJECT", "proper_person_name", pos="noun"), + ] + _write_jsonl(frozen, rows) + dev = tmp_path / "development.txt" + dev.write_text("other phrase\n", encoding="utf-8") + val = tmp_path / "validation.txt" + val.write_text("another phrase\n", encoding="utf-8") + return frozen, dev, val + + +def test_materialize_hides_prediction_and_keeps_source_hash(tmp_path): + frozen, dev, val = _layout(tmp_path) + digest = file_sha256(frozen) + receipt = materialize( + frozen, + tmp_path / "eval", + expected_sha256=digest, + development=dev, + validation_development=val, + frozen_at=FROZEN_AT, + evaluation_id="HLX-EVAL-TEST", + ) + assert file_sha256(frozen) == digest + assert receipt["state"] == "PREDICTIONS_FROZEN" + assert receipt["held_out_precision"] == "NOT_COMPUTABLE" + assert receipt["select_authorized"] is False + assert receipt["admitted"] == 0 + review = [ + json.loads(line) + for line in (tmp_path / "eval" / "operator" / "heldout-001.review.jsonl").read_text().splitlines() + ] + assert len(review) == 3 + for row in review: + assert "predicted" not in row + assert "rule" not in row + assert "bucket" not in row + assert row["row_id"] == normalized_text_sha256(row["surface"]) + assert row["gloss"].startswith("gloss for ") + report = score(tmp_path / "eval") + assert report["status"] == "NOT_COMPUTABLE" + assert report["confusion_matrix"] == "NOT_COMPUTABLE" + assert not (tmp_path / "eval" / "reports" / "heldout-001.confusion.csv").exists() + + +def test_development_overlap_refuses(tmp_path): + frozen, dev, val = _layout(tmp_path) + dev.write_text("alpha beta\n", encoding="utf-8") + with pytest.raises(ScreenEvalError, match="development"): + materialize( + frozen, + tmp_path / "eval", + expected_sha256=file_sha256(frozen), + development=dev, + validation_development=val, + frozen_at=FROZEN_AT, + ) + + +def test_score_joins_on_row_id_and_does_not_authorize(tmp_path): + frozen, dev, val = _layout(tmp_path) + out = tmp_path / "eval" + materialize( + frozen, + out, + expected_sha256=file_sha256(frozen), + development=dev, + validation_development=val, + frozen_at=FROZEN_AT, + evaluation_id="HLX-EVAL-TEST", + ) + predictions = [ + json.loads(line) + for line in (out / "predictions" / "heldout-001.predictions.jsonl").read_text().splitlines() + ] + by_surface = { + json.loads(line)["surface"]: json.loads(line)["row_id"] + for line in (out / "samples" / "heldout-001.jsonl").read_text().splitlines() + } + labels = [ + { + "evaluation_id": "HLX-EVAL-TEST", + "row_id": by_surface["alpha beta"], + "operator_bucket": "HIGH", + "reason_code": "STRONG_IDIOM", + "note": None, + "labeled_at": LABELED_AT, + }, + { + "evaluation_id": "HLX-EVAL-TEST", + "row_id": by_surface["gamma delta"], + "operator_bucket": "REJECT", + "reason_code": "PRODUCTIVE_NUMBER", + "note": None, + "labeled_at": LABELED_AT, + }, + { + "evaluation_id": "HLX-EVAL-TEST", + "row_id": by_surface["epsilon zeta"], + "operator_bucket": "UNRESOLVED", + "reason_code": "INSUFFICIENT_SIGNAL", + "note": None, + "labeled_at": LABELED_AT, + }, + ] + label_path = tmp_path / "labels.jsonl" + _write_jsonl(label_path, labels) + freeze_labels(out, label_path) + # Tampering with the source order must not matter: score reads the frozen label file. + report = score(out) + assert report["status"] == "SCORED" + assert report["select_authorized"] is False + assert report["revision_eligible"] is False + assert report["unresolved_count"] == 1 + assert report["per_bucket"]["HIGH"]["precision"] == 1 + assert report["per_bucket"]["SECONDARY"]["precision"] == 0 + assert report["error_classes"]["false_secondary"] == 1 + errors = [ + json.loads(line) + for line in (out / "reports" / "heldout-001.errors.jsonl").read_text().splitlines() + ] + assert errors[0]["surface"] == "gamma delta" + assert errors[0]["operator_reason"] == "PRODUCTIVE_NUMBER" + assert errors[0]["error_class"] == "false_secondary" + assert {row["row_id"] for row in predictions} == set(by_surface.values()) + + +def test_label_before_freeze_refuses(tmp_path): + frozen, dev, val = _layout(tmp_path) + out = tmp_path / "eval" + materialize( + frozen, + out, + expected_sha256=file_sha256(frozen), + development=dev, + validation_development=val, + frozen_at=FROZEN_AT, + evaluation_id="HLX-EVAL-TEST", + ) + row_id = json.loads((out / "samples" / "heldout-001.jsonl").read_text().splitlines()[0])["row_id"] + # Cover all three identities with an early timestamp on the first. + ids = [ + json.loads(line)["row_id"] + for line in (out / "samples" / "heldout-001.jsonl").read_text().splitlines() + ] + labels = [] + for index, identity in enumerate(ids): + labels.append({ + "evaluation_id": "HLX-EVAL-TEST", + "row_id": identity, + "operator_bucket": "HIGH", + "reason_code": "STRONG_IDIOM", + "note": None, + "labeled_at": "2019-12-31T00:00:00Z" if index == 0 else LABELED_AT, + }) + path = tmp_path / "early.jsonl" + _write_jsonl(path, labels) + with pytest.raises(ScreenEvalError, match="timestamp"): + freeze_labels(out, path) + assert row_id in ids + + +def test_changed_prediction_hash_refuses_score(tmp_path): + frozen, dev, val = _layout(tmp_path) + out = tmp_path / "eval" + materialize( + frozen, + out, + expected_sha256=file_sha256(frozen), + development=dev, + validation_development=val, + frozen_at=FROZEN_AT, + ) + pred = out / "predictions" / "heldout-001.predictions.jsonl" + pred.write_text(pred.read_text() + "\n", encoding="utf-8") + with pytest.raises(ScreenEvalError, match="prediction_artifact_sha256"): + score(out) diff --git a/tests/shadow/test_km_candidate_evaluation.py b/tests/shadow/test_km_candidate_evaluation.py new file mode 100644 index 00000000..6d5ac839 --- /dev/null +++ b/tests/shadow/test_km_candidate_evaluation.py @@ -0,0 +1,107 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.km_candidate_evaluation import ( + AMBIGUOUS_MULTIPLE_SYNSETS, + EXACT_UNIQUE_RECONSTRUCTION, + NO_PWN3_MATCH, + UNKNOWN, + align_item, + join_synset, + lookup_key, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "km_candidate_evaluation.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def test_module_has_no_phrase_exception_or_probe_surface(): + assert rule_surface_violations(MODULE.read_text(encoding="utf-8"), PROBES) == [] + + +def test_join_does_not_accept_surface_gloss_or_operator_label(): + names = set(inspect.signature(join_synset).parameters) + assert names == {"synset", "exact", "ambiguous"} + assert "gloss" not in inspect.signature(align_item).parameters + + +def test_lookup_key_folds_case_and_apostrophe_and_keeps_hyphen(): + assert lookup_key("Prince Albert") == "prince_albert" + assert lookup_key("fool\u2019s paradise") == "fool's_paradise" + assert lookup_key("parenthesis-free notation") == "parenthesis-free_notation" + assert lookup_key("parenthesis-free notation") != lookup_key("parenthesis free notation") + + +def test_unique_reconstruction_maps_labels_and_polysemy_does_not(): + one = align_item("NONCOMPOSITIONAL", None, ["noun:1"]) + assert one["alignment_status"] == EXACT_UNIQUE_RECONSTRUCTION + assert one["semantic_noncompositional"] == "YES" + assert one["primary_evidence_code"] == "km_exact_noncompositional" + assert one["aligned_pwn30_synset"] == "noun:1" + other = align_item("COMPOSITIONAL", None, ["noun:2"]) + assert other["semantic_noncompositional"] == "NO" + assert other["primary_evidence_code"] == "km_exact_compositional" + many = align_item("NONCOMPOSITIONAL", None, ["noun:3", "noun:4"]) + assert many["alignment_status"] == AMBIGUOUS_MULTIPLE_SYNSETS + assert many["semantic_noncompositional"] == UNKNOWN + assert many["aligned_pwn30_synset"] is None + assert many["primary_evidence_code"] == "km_ambiguous_synset" + missing = align_item("NONCOMPOSITIONAL", None, []) + assert missing["alignment_status"] == NO_PWN3_MATCH + assert missing["semantic_noncompositional"] == UNKNOWN + assert missing["primary_evidence_code"] == "km_no_match" + + +def test_a_source_identifier_must_name_one_synset(): + found = align_item("COMPOSITIONAL", "noun:9", ["noun:9"]) + assert found["alignment_status"] == "EXACT_SOURCE_ID" + assert found["semantic_noncompositional"] == "NO" + conflict = align_item("NONCOMPOSITIONAL", "noun:8", ["noun:9"]) + assert conflict["alignment_status"] == "VERSION_CONFLICT" + assert conflict["semantic_noncompositional"] == UNKNOWN + assert conflict["primary_evidence_code"] == "km_version_conflict" + tied = align_item("NONCOMPOSITIONAL", "noun:3", ["noun:3", "noun:4"]) + assert tied["alignment_status"] == AMBIGUOUS_MULTIPLE_SYNSETS + assert tied["semantic_noncompositional"] == UNKNOWN + + +def test_hyperlex_join_is_synset_identity_only(): + exact = { + "noun:1": { + "alignment_status": EXACT_UNIQUE_RECONSTRUCTION, + "semantic_noncompositional": "YES", + "source_item_id": "KM-1", + "aligned_pwn30_synset": "noun:1", + } + } + ambiguous = {"noun:2": ["KM-2"]} + hit = join_synset("noun:1", exact, ambiguous) + assert hit["semantic_noncompositional"] == "YES" + assert hit["candidate_source_item_id"] == "KM-1" + other_sense = join_synset("noun:9", exact, ambiguous) + assert other_sense["semantic_noncompositional"] == UNKNOWN + assert other_sense["primary_evidence_code"] == "km_no_match" + assert other_sense["candidate_aligned_synset"] is None + polyseme = join_synset("noun:2", exact, ambiguous) + assert polyseme["alignment_status"] == AMBIGUOUS_MULTIPLE_SYNSETS + assert polyseme["semantic_noncompositional"] == UNKNOWN + assert polyseme["candidate_aligned_synset"] is None + assert polyseme["primary_evidence_code"] == "km_ambiguous_synset" diff --git a/tests/shadow/test_magpie_candidate_evaluation.py b/tests/shadow/test_magpie_candidate_evaluation.py new file mode 100644 index 00000000..ac91218e --- /dev/null +++ b/tests/shadow/test_magpie_candidate_evaluation.py @@ -0,0 +1,233 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.magpie_candidate_evaluation import ( + ALIGNED_IDIOMATIC, + ALIGNED_LITERAL, + AMBIGUOUS, + CONFLICT, + EXACT, + MIXED, + NONE, + NORMALIZED, + UNKNOWN, + build_index, + evaluate_row, + exact_key, + normalized_key, + semantic_noncompositional, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "magpie_candidate_evaluation.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) +VERSION = "test-version" +ARTIFACT = "b" * 64 + + +def _instance(identifier, idiom, label="i", variant_type="identical", confidence=1, **extra): + row = { + "confidence": confidence, + "id": identifier, + "idiom": idiom, + "label": label, + "variant_type": variant_type, + } + row.update(extra) + return row + + +def _row(surface, synset, instances): + return evaluate_row(surface, synset, build_index(instances), VERSION, ARTIFACT) + + +def test_module_has_no_phrase_exception_or_probe_surface(): + assert rule_surface_violations(MODULE.read_text(encoding="utf-8"), PROBES) == [] + + +def test_evaluator_does_not_accept_operator_labels_or_gloss(): + names = set(inspect.signature(evaluate_row).parameters) + assert "operator_bucket" not in names + assert "gloss" not in names + assert "operator_reason" not in names + + +def test_exact_match_is_case_and_separator_only(): + found = _row( + "At The End", + "noun:1", + [_instance(1, "at the end", "i"), _instance(2, "at the end", "l")], + ) + assert found["surface_match"] == EXACT + assert found["matched_magpie_expression"] == "at the end" + assert found["idiomatic_instance_count"] == 1 + assert found["literal_instance_count"] == 1 + assert found["sense_alignment"] == UNKNOWN + assert found["semantic_noncompositional"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + assert found["source_instance_ids"] == [1, 2] + + +def test_punctuation_difference_is_normalized_and_not_yes(): + found = _row("end of the day", "noun:1", [_instance(4, "end of the day.")]) + assert exact_key("end of the day") != exact_key("end of the day.") + assert found["surface_match"] == NORMALIZED + assert found["semantic_noncompositional"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + + +def test_curly_apostrophe_is_normalized_and_pronouns_are_not_rewritten(): + curly = "flip one\u2019s lid" + found = _row("flip one's lid", "noun:1", [_instance(5, curly, "i")]) + assert found["surface_match"] == NORMALIZED + assert found["semantic_noncompositional"] == UNKNOWN + other = _row("flip his lid", "noun:1", [_instance(6, "flip one's lid", "i")]) + assert other["surface_match"] == NONE + assert other["primary_evidence_code"] == "magpie_no_match" + assert normalized_key("someone's chair") != normalized_key("one's chair") + + +def test_inflection_variant_metadata_does_not_invent_a_match(): + found = _row( + "threw in the towel", + "verb:1", + [_instance(7, "throw in the towel", "i", variant_type="inflection")], + ) + assert found["surface_match"] == NONE + assert found["semantic_noncompositional"] == UNKNOWN + + +def test_colliding_normalized_types_are_ambiguous_and_unattributed(): + found = _row( + "a b", + "noun:1", + [_instance(8, "a-b"), _instance(9, "a b.")], + ) + assert found["surface_match"] == AMBIGUOUS + assert found["matched_magpie_expression"] is None + assert found["literal_instance_count"] is None + assert found["source_instance_ids"] == [] + assert found["ambiguous_candidates"] == ["a b.", "a-b"] + assert found["sense_alignment"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + + +def test_unique_exact_wins_over_another_normalized_type(): + found = _row( + "a b", + "noun:1", + [_instance(10, "a b", "l"), _instance(11, "a-b", "i")], + ) + assert found["surface_match"] == EXACT + assert found["matched_magpie_expression"] == "a b" + assert found["literal_instance_count"] == 1 + assert found["idiomatic_instance_count"] == 0 + + +def test_all_idiomatic_labels_without_a_sense_identifier_stay_unknown(): + found = _row( + "bear fruit", + "verb:9", + [_instance(12, "bear fruit", "i"), _instance(13, "bear fruit", "i")], + ) + assert found["idiomatic_instance_count"] == 2 + assert found["literal_instance_count"] == 0 + assert found["sense_alignment"] == UNKNOWN + assert found["alignment_basis"] == "no_bound_sense_identifier" + assert found["semantic_noncompositional"] == UNKNOWN + + +def test_gloss_text_cannot_change_alignment(): + instances = [_instance(14, "bear fruit", "i")] + index = build_index(instances) + first = evaluate_row("bear fruit", "verb:9", index, VERSION, ARTIFACT) + second = evaluate_row("bear fruit", "verb:9", index, VERSION, ARTIFACT) + assert first == second + assert "gloss" not in first + + +def test_equal_sense_identifier_can_align_without_using_labels_as_rules(): + instances = [ + _instance(15, "bear fruit", "i", synset="verb:9"), + _instance(16, "bear fruit", "i", synset="verb:9"), + ] + found = _row("bear fruit", "verb:9", instances) + assert found["sense_alignment"] == ALIGNED_IDIOMATIC + assert found["semantic_noncompositional"] == "YES" + assert found["primary_evidence_code"] == "magpie_aligned_idiomatic" + literal = _row( + "bear fruit", + "verb:9", + [_instance(17, "bear fruit", "l", synset="verb:9")], + ) + assert literal["sense_alignment"] == ALIGNED_LITERAL + assert literal["semantic_noncompositional"] == "NO" + mixed = _row( + "bear fruit", + "verb:9", + [ + _instance(18, "bear fruit", "i", synset="verb:9"), + _instance(19, "bear fruit", "l", synset="verb:9"), + ], + ) + assert mixed["sense_alignment"] == MIXED + assert mixed["semantic_noncompositional"] == UNKNOWN + assert mixed["primary_evidence_code"] == "magpie_mixed_usage" + + +def test_a_different_sense_identifier_does_not_transfer(): + found = _row( + "bear fruit", + "verb:9", + [_instance(20, "bear fruit", "i", synset="verb:8")], + ) + assert found["sense_alignment"] == UNKNOWN + assert found["alignment_basis"] == "sense_identifier_mismatch" + assert found["semantic_noncompositional"] == UNKNOWN + conflict = _row( + "bear fruit", + "verb:9", + [ + _instance(21, "bear fruit", "i", synset="verb:9"), + _instance(22, "bear fruit", "i", synset="verb:8"), + ], + ) + assert conflict["sense_alignment"] == CONFLICT + assert conflict["semantic_noncompositional"] == UNKNOWN + assert conflict["primary_evidence_code"] == "magpie_sense_conflict" + + +def test_other_and_unclear_labels_stay_unresolved(): + found = _row( + "bear fruit", + "verb:9", + [ + _instance(23, "bear fruit", "o"), + _instance(24, "bear fruit", "?"), + _instance(25, "bear fruit", "f"), + ], + ) + assert found["unresolved_instance_count"] == 3 + assert found["idiomatic_instance_count"] == 0 + assert semantic_noncompositional(UNKNOWN) == UNKNOWN + assert semantic_noncompositional(ALIGNED_IDIOMATIC) == "YES" + assert semantic_noncompositional(ALIGNED_LITERAL) == "NO" + assert semantic_noncompositional(MIXED) == UNKNOWN + assert semantic_noncompositional(CONFLICT) == UNKNOWN diff --git a/tests/shadow/test_model_based_wsd_candidate_v1.py b/tests/shadow/test_model_based_wsd_candidate_v1.py new file mode 100644 index 00000000..811c96e9 --- /dev/null +++ b/tests/shadow/test_model_based_wsd_candidate_v1.py @@ -0,0 +1,243 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.model_based_wsd_candidate_v1 import ( + candidate_gloss_text, + candidate_policy, + coverage_gate, + format_probability, + gloss_lemma, + overlay_status, + project_row_status, + quoted_context, + resolve_model_scores, + summarize_confidence, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "model_based_wsd_candidate_v1.py" +REPLAY = ROOT / "scripts" / "shadow" / "hyperlexical" / "model_based_wsd_candidate_v1_replay.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) +KEYS = { + "noun:1": ["alpha%1:00:00::"], + "noun:2": ["beta%1:00:00::"], +} + + +def test_module_freezes_one_model_and_hides_labels_from_the_decision(): + source = MODULE.read_text(encoding="utf-8") + replay = REPLAY.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] + assert "sentence_transformers" not in source + assert "MiniLM" not in source + assert "sentence_transformers" not in replay + assert "MiniLM" not in replay + names = set(inspect.signature(resolve_model_scores).parameters) + assert "operator_bucket" not in names + assert "residual_score" not in names + policy = candidate_policy() + assert policy["state_at_freeze"] == "SPEC_FROZEN" + assert policy["applied_to_hyperlex_at_freeze"] is False + assert policy["selected_source"] == "none" + assert policy["runtime_integration"] is False + assert policy["residual_replay_authorized"] is False + assert policy["model_name"] == "kanishka/GlossBERT" + assert policy["model_revision"] == "0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a" + assert policy["positive_class_index"] == 1 + assert policy["license"] == "MIT" + assert "operator_labels" in policy["forbidden_inputs"] + assert "residual_embeddings" in policy["forbidden_inputs"] + assert "residual_scores" in policy["forbidden_inputs"] + assert policy["readiness_gate"]["baseline_high_ready"] == 4 + assert policy["readiness_gate"]["baseline_secondary_ready"] == 4 + assert policy["readiness_gate"]["baseline_total_ready"] == 20 + assert policy["readiness_gate"]["comparison"] == "strictly_greater" + assert policy["no_gold_constituent_senses"] is True + assert policy["pwn_mapping"]["cross_version_map"] is False + assert policy["pwn_mapping"]["manual_migration"] is False + + +def test_context_quotes_only_the_indexed_content_token(): + assert quoted_context("alpha in beta", 0, "alpha") == '"alpha" in beta' + assert quoted_context("alpha in beta", 1, "beta") == 'alpha in "beta"' + assert gloss_lemma(["bank", "other"], "banks", {"banks": {"bank"}, "bank": {"banks"}}) == "bank" + assert candidate_gloss_text("bank", "sloping land") == "bank: sloping land" + assert format_probability(0.5) == "0.500000" + + +def test_abstention_ties_and_invalid_senses_do_not_select(): + resolved = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.800000", "noun:2": "0.200000"}, + ) + assert resolved["model_resolution_status"] == "RESOLVED" + assert resolved["selected_synset"] == "noun:1" + assert resolved["selected_sense_key"] == "alpha%1:00:00::" + assert resolved["primary_evidence_code"] == "model_margin" + boundary = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.500000", "noun:2": "0.400000"}, + ) + assert boundary["model_resolution_status"] == "RESOLVED" + assert boundary["model_margin"] == "0.100000" + near = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.550000", "noun:2": "0.500000"}, + ) + assert near["model_resolution_status"] == "AMBIGUOUS" + assert near["selected_synset"] is None + assert near["primary_evidence_code"] == "model_abstention" + low = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.400000", "noun:2": "0.100000"}, + ) + assert low["model_resolution_status"] == "AMBIGUOUS" + assert low["selected_synset"] is None + tie = resolve_model_scores( + ["noun:2", "noun:1"], + KEYS, + {"noun:1": "0.800000", "noun:2": "0.800000"}, + ) + assert tie["model_resolution_status"] == "AMBIGUOUS" + assert tie["selected_synset"] is None + assert tie["primary_evidence_code"] == "model_score_tie" + outsider = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.900000", "noun:9": "0.100000"}, + ) + assert outsider["model_resolution_status"] == "INVALID" + assert outsider["selected_synset"] is None + assert outsider["primary_evidence_code"] == "invalid_model_output" + many_keys = resolve_model_scores( + ["noun:1", "noun:2"], + {"noun:1": ["alpha%1:00:00::", "alpha%1:00:01::"], "noun:2": ["beta%1:00:00::"]}, + {"noun:1": "0.900000", "noun:2": "0.100000"}, + ) + assert many_keys["model_resolution_status"] == "AMBIGUOUS" + assert many_keys["selected_sense_key"] is None + assert many_keys["selected_synset"] is None + assert many_keys["primary_evidence_code"] == "pwn30_sense_key_not_unique" + flooded = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.990000", "noun:2": "0.010000"}, + overflow=True, + ) + assert flooded["model_resolution_status"] == "AMBIGUOUS" + assert flooded["selected_synset"] is None + assert flooded["primary_evidence_code"] == "context_overflow" + failed = resolve_model_scores(["noun:1", "noun:2"], KEYS, None, error="forward failed") + assert failed["model_resolution_status"] == "ERROR" + assert failed["selected_synset"] is None + + +def test_tier3_cannot_override_an_earlier_tier_and_readiness_needs_two_constituents(): + assert overlay_status("EXACT", "STRUCTURAL_EXACT", None) == "EXACT" + assert overlay_status("RESOLVED", "EXTENDED_LESK_V1", None) == "LESK_RESOLVED" + assert overlay_status("UNRESOLVED", "NONE", None) == "UNRESOLVED" + assert overlay_status("AMBIGUOUS", "EXTENDED_LESK_V1", "RESOLVED") == "MODEL_RESOLVED" + assert overlay_status("AMBIGUOUS", "EXTENDED_LESK_V1", "AMBIGUOUS") == "AMBIGUOUS" + assert project_row_status(["EXACT", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert project_row_status(["LESK_RESOLVED", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert project_row_status(["EXACT", "AMBIGUOUS"]) == "UNKNOWN" + assert project_row_status(["EXACT"]) == "UNKNOWN" + try: + overlay_status("EXACT", "STRUCTURAL_EXACT", "RESOLVED") + except RuntimeError as exc: + assert "overrode" in str(exc) + else: + raise AssertionError("exact override was accepted") + + +def test_coverage_gate_is_strict_and_equality_is_not_promising(): + promising = coverage_gate( + high_ready=5, + secondary_ready=5, + total_ready=21, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert promising["candidate_status"] == "CANDIDATE_PROMISING" + assert promising["next_transition_authorized"] is False + baseline = coverage_gate( + high_ready=4, + secondary_ready=4, + total_ready=20, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert baseline["candidate_status"] == "CANDIDATE_INSUFFICIENT" + one_sided = coverage_gate( + high_ready=5, + secondary_ready=4, + total_ready=21, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert one_sided["candidate_status"] == "CANDIDATE_INSUFFICIENT" + rejected = coverage_gate( + high_ready=10, + secondary_ready=10, + total_ready=40, + invalid_output_count=1, + error_count=0, + determinism="IDENTICAL", + ) + assert rejected["candidate_status"] == "CANDIDATE_REJECTED" + drifted = coverage_gate( + high_ready=10, + secondary_ready=10, + total_ready=40, + invalid_output_count=0, + error_count=0, + determinism="MISMATCH", + ) + assert drifted["candidate_status"] == "NOT_DETERMINISTIC" + summary = summarize_confidence( + [ + { + "candidate_pwn30_synsets": ["noun:1", "noun:2"], + "model_confidence": "0.750000", + "model_margin": "0.300000", + "model_resolution_status": "RESOLVED", + }, + { + "candidate_pwn30_synsets": ["noun:1", "noun:2", "noun:3"], + "model_confidence": "0.400000", + "model_margin": "0.050000", + "model_resolution_status": "AMBIGUOUS", + }, + ] + ) + assert summary["resolved"] == 1 + assert summary["abstained"] == 1 + assert summary["confidence_bins"]["0.70_to_0.80"] == 1 + assert summary["confidence_bins"]["below_0.50"] == 1 + assert summary["margin_bins"]["0.25_to_0.50"] == 1 + assert summary["by_candidate_count"]["2"]["resolved"] == 1 + assert "operator_bucket" not in summary diff --git a/tests/shadow/test_residual_model_resolved_replay_v1.py b/tests/shadow/test_residual_model_resolved_replay_v1.py new file mode 100644 index 00000000..cd6e9a07 --- /dev/null +++ b/tests/shadow/test_residual_model_resolved_replay_v1.py @@ -0,0 +1,233 @@ +import inspect +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.residual_model_resolved_replay_v1 import ( + AMBIGUOUS, + BOOTSTRAP_RESAMPLES, + BOOTSTRAP_SEED, + EXACT, + INVERTED, + NOT_COMPUTABLE, + NO_SEPARATION, + RESOLVED, + SUPPORTED, + UNRESOLVED, + abstention_reason, + analysis_plan, + bootstrap_intervals, + direction_result, + extreme_driven, + full_distribution, + integrate_constituent, + pair_comparison, + projection_token, + replay_decision, + row_projection, +) +from hyperlexical.semantic_compositionality_residual import percentile +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "residual_model_resolved_replay_v1.py" +REPLAY = ROOT / "scripts" / "shadow" / "hyperlexical" / "residual_model_resolved_replay_v1_replay.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _resolver(**overrides): + row = { + "candidate_synsets": ["noun:00000001", "noun:00000002"], + "constituent_index": 0, + "constituent_pos": "noun", + "constituent_surface": "alpha", + "parent_row_id": "row-1", + "parent_surface": "alpha beta", + "parent_synset": "noun:00000009", + "primary_evidence_code": "extended_lesk_tie", + "resolution_method": "EXTENDED_LESK_V1", + "resolution_status": AMBIGUOUS, + "selected_synset": None, + } + row.update(overrides) + return row + + +def _model(**overrides): + row = { + "model_resolution_status": RESOLVED, + "primary_evidence_code": "model_margin", + "prior_resolution_status": AMBIGUOUS, + "selected_sense_key": "alpha%1:00:00::", + "selected_synset": "noun:00000001", + } + row.update(overrides) + return row + + +def test_module_hides_labels_and_does_not_encode(): + source = MODULE.read_text(encoding="utf-8") + replay = REPLAY.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] + assert "sentence_transformers" not in source + assert "torch" not in source + assert "operator_bucket" not in source + assert "semantic_noncompositional = YES" not in source + assert "semantic_noncompositional = NO" not in source + assert "operator_bucket" not in inspect.signature(integrate_constituent).parameters + plan = analysis_plan() + assert plan["bootstrap_seed"] == BOOTSTRAP_SEED == 0 + assert plan["bootstrap_resamples"] == BOOTSTRAP_RESAMPLES == 10000 + assert plan["emits_yes_no"] is False + assert plan["semantic_noncompositionality_threshold"] is None + assert plan["threshold_eligible"] is False + assert plan["json_schema_document"] is None + assert replay.index("write_json(RECEIPT_PATH") < replay.index("buckets = load_operator_buckets()") + assert "semantic_noncompositional = YES" not in replay + assert "semantic_noncompositional = NO" not in replay + + +def test_integration_copies_frozen_tiers_and_refuses_overrides(): + exact = integrate_constituent( + _resolver( + primary_evidence_code="structural_exact", + resolution_method="STRUCTURAL_EXACT", + resolution_status=EXACT, + selected_synset="noun:00000003", + ), + None, + ) + assert exact["resolution_tier"] == "TIER1_STRUCTURAL" + assert exact["resolution_status"] == EXACT + assert exact["selected_pwn30_synset"] == "noun:00000003" + assert exact["selected_sense_key_if_available"] is None + with pytest.raises(RuntimeError): + integrate_constituent( + _resolver(resolution_method="UNIQUE_LEMMA", resolution_status=EXACT, selected_synset="noun:1"), + _model(), + ) + with pytest.raises(RuntimeError): + integrate_constituent( + _resolver(resolution_method="EXTENDED_LESK_V1", resolution_status=RESOLVED, selected_synset="noun:1"), + _model(), + ) + copied = integrate_constituent(_resolver(), _model()) + assert copied["resolution_tier"] == "TIER3_GLOSSBERT" + assert copied["resolution_status"] == RESOLVED + assert copied["selected_pwn30_synset"] == "noun:00000001" + assert copied["resolution_provenance"] == "model_margin" + with pytest.raises(RuntimeError): + integrate_constituent(_resolver(), _model(selected_synset="noun:99999999")) + abstained = integrate_constituent( + _resolver(), + _model(model_resolution_status=AMBIGUOUS, selected_sense_key=None, selected_synset=None), + ) + assert abstained["resolution_status"] == AMBIGUOUS + assert abstained["selected_pwn30_synset"] is None + unresolved = integrate_constituent( + _resolver( + primary_evidence_code="no_candidate", + resolution_method="NONE", + resolution_status=UNRESOLVED, + selected_synset=None, + candidate_synsets=[], + ), + None, + ) + assert unresolved["resolution_tier"] == UNRESOLVED + assert projection_token("TIER2_EXTENDED_LESK", RESOLVED) == "LESK_RESOLVED" + assert projection_token("TIER3_GLOSSBERT", RESOLVED) == "MODEL_RESOLVED" + assert row_projection(["EXACT"]) == "UNKNOWN" + assert row_projection(["EXACT", "LESK_RESOLVED", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert abstention_reason(["EXACT"]) == "fewer_than_two_content_constituents" + assert abstention_reason([AMBIGUOUS, UNRESOLVED]) == "ambiguous_content_constituent" + assert abstention_reason([EXACT, RESOLVED]) is None + + +def test_decision_gate_and_direction_are_preregistered(): + high = ["0.9000000000", "0.5000000000", "0.2000000000"] + secondary = ["0.4000000000", "0.4000000000", "0.1000000000"] + comparison = pair_comparison(high, secondary) + assert comparison["status"] == "DESCRIPTIVE" + assert direction_result(comparison) == SUPPORTED + assert Decimal_gt(comparison["median_difference"]) + assert Decimal_gt(comparison["rank_biserial"]) + assert extreme_driven(high, secondary) is True + promising = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + ) + assert promising["candidate_status"] == "CANDIDATE_PROMISING" + assert promising["threshold_eligible"] is False + assert promising["next_transition_authorized"] is False + concentrated = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=True, + extremes=False, + pos_split=False, + ) + assert concentrated["candidate_status"] == "CANDIDATE_INSUFFICIENT" + assert replay_decision( + readiness_reproduced=False, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + )["candidate_status"] == NOT_COMPUTABLE + assert replay_decision( + readiness_reproduced=True, + determinism="DIFFERENT", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + )["candidate_status"] == "NOT_DETERMINISTIC" + inverted = pair_comparison(secondary, high) + assert direction_result(inverted) == INVERTED + flat = pair_comparison(["0.2000000000", "0.2000000000"], ["0.2000000000"]) + assert direction_result(flat) == NO_SEPARATION + assert pair_comparison([], ["0.1"])["status"] == NOT_COMPUTABLE + + +def Decimal_gt(text: str) -> bool: + from decimal import Decimal + + return Decimal(text) > 0 + + +def test_distribution_matches_residual_percentile_and_bootstrap_is_seeded(): + scores = ["0.1000000000", "0.2000000000", "0.4000000000", "0.8000000000"] + report = full_distribution(scores) + assert report["p25"] == percentile(sorted(scores, key=lambda item: __import__("decimal").Decimal(item)), 25) + assert report["count"] == 4 + assert report["std"] is not None + first = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=0, resamples=30) + repeat = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=0, resamples=30) + other = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=1, resamples=30) + assert first == repeat + assert first != other + assert first["seed"] == 0 + assert first["resamples"] == 30 diff --git a/tests/shadow/test_semantic_compositionality_residual.py b/tests/shadow/test_semantic_compositionality_residual.py new file mode 100644 index 00000000..367315f8 --- /dev/null +++ b/tests/shadow/test_semantic_compositionality_residual.py @@ -0,0 +1,251 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.semantic_compositionality_residual import ( + AMBIGUOUS, + EXACT, + UNIQUE, + UNRESOLVED, + candidate_policy, + distribution, + evaluation_status, + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + percentile, + representation_text, + residual_score, + resolve_constituent, + resolved_synset, + score_record, + select_lemma, + vector_hash, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "semantic_compositionality_residual.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _extraction(surface: str = "alpha beta") -> dict: + return extract_constituents(surface) + + +def test_module_has_no_phrase_exception_or_probe_surface(): + assert rule_surface_violations(MODULE.read_text(encoding="utf-8"), PROBES) == [] + + +def test_resolution_does_not_accept_gloss_or_operator_labels(): + blocked = ( + exact_synset_ids, + lexical_synset_ids, + resolve_constituent, + resolved_synset, + score_record, + residual_score, + ) + for function in blocked: + names = set(inspect.signature(function).parameters) + assert "gloss" not in names + assert "operator_bucket" not in names + assert "operator_label" not in names + policy = candidate_policy() + assert "gloss_similarity" in policy["constituent_sense_resolution"]["forbidden"] + assert policy["semantic_noncompositionality_threshold"] is None + assert policy["emits_yes_no"] is False + assert not hasattr(sys.modules["hyperlexical.semantic_compositionality_residual"], "gloss_similarity") + + +def test_structural_tokens_leave_content_words(): + towel = extract_constituents("throw in the towel") + assert towel["surface_tokens"] == ["throw", "in", "the", "towel"] + assert towel["content_constituents"] == ["throw", "towel"] + assert towel["ignored_structural_tokens"] == ["in", "the"] + assert towel["constituent_extraction_status"] == "EXTRACTED" + shots = extract_constituents("call the shots") + assert shots["content_constituents"] == ["call", "shots"] + assert shots["ignored_structural_tokens"] == ["the"] + luck = extract_constituents("as luck would have it") + assert luck["content_constituents"] == ["luck"] + assert luck["constituent_extraction_status"] == "UNKNOWN" + folded = extract_constituents("Throw In The Towel") + assert folded["content_constituents"] == ["Throw", "Towel"] + assert folded["ignored_structural_tokens"] == ["In", "The"] + hyphen = extract_constituents("well-known person") + assert hyphen["content_constituents"] == ["well-known", "person"] + owned = extract_constituents("one\u2019s own goal") + assert owned["ignored_structural_tokens"] == ["one\u2019s"] + assert owned["content_constituents"] == ["own", "goal"] + + +def test_exact_pointer_beats_polysemy_and_zero_target_does_not_count(): + pointers = [ + ("+", 1, "noun:1", "towel"), + ("+", 0, "noun:9", "towel"), + ("@", 1, "noun:8", "towel"), + ("\\", 2, "noun:1", "towels"), + ] + exact = exact_synset_ids( + pointers, + "towel", + {"towel": {"towels"}, "towels": {"towel"}}, + ) + assert exact == ["noun:1", "noun:1"] + assert resolve_constituent(exact, ["noun:1", "noun:2", "verb:3"]) == EXACT + assert resolved_synset(exact, ["noun:2"]) == "noun:1" + many = exact_synset_ids( + [("+", 1, "noun:1", "shot"), ("+", 1, "noun:2", "shot")], + "shots", + {"shots": {"shot"}}, + ) + assert resolve_constituent(many, ["noun:1"]) == AMBIGUOUS + assert resolved_synset(many, ["noun:1"]) is None + + +def test_lexical_resolution_abstains_when_several_synsets_match(): + index = {"towel": ["noun:4"], "shot": ["noun:5", "noun:6"]} + one = lexical_synset_ids(index, "towel", {}) + assert resolve_constituent([], one) == UNIQUE + assert resolved_synset([], one) == "noun:4" + many = lexical_synset_ids(index, "shots", {"shots": {"shot"}}) + assert resolve_constituent([], many) == AMBIGUOUS + assert resolve_constituent([], []) == UNRESOLVED + assert select_lemma(["towel", "bath_towel"], "towels", {"towels": {"towel"}}) == "towel" + + +def test_residual_is_continuous_and_deterministic(): + same = residual_score([1.0, 0.0], [[1.0, 0.0], [1.0, 0.0]]) + assert same is not None + assert same[0] == "0.0000000000" + orthogonal = residual_score([1.0, 0.0], [[0.0, 1.0], [0.0, 1.0]]) + assert orthogonal is not None + assert orthogonal[0] == "1.0000000000" + assert residual_score([0.0, 0.0], [[1.0, 0.0], [0.0, 1.0]]) is None + assert vector_hash([1.0, 0.0]) == vector_hash([1.0, 0.0]) + assert vector_hash([1.0, 0.0]) != vector_hash([0.0, 1.0]) + assert len(vector_hash([1.0])) == 64 + text = representation_text("bath_towel", "noun", " a towel ") + assert text == "bath towel (noun): a towel" + + +def test_ambiguous_constituent_is_unknown_and_is_not_encoded(): + extraction = _extraction("call the shots") + record = score_record( + row_id="row", + surface="call the shots", + pos="verb", + synset="verb:00000001", + extraction=extraction, + resolutions=["UNIQUE", "AMBIGUOUS"], + resolved_synsets=["verb:2", None], + resolved_lemmas=["call", None], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + ) + assert record["score_status"] == "UNKNOWN" + assert record["primary_abstention_reason"] == "ambiguous_content_constituent" + assert record["residual_score"] is None + assert record["whole_vector_hash"] is None + assert record["constituent_resolution_status"] == ["UNIQUE", "AMBIGUOUS"] + short = score_record( + row_id="row", + surface="as luck would have it", + pos="noun", + synset="noun:00000002", + extraction=extract_constituents("as luck would have it"), + resolutions=[], + resolved_synsets=[], + resolved_lemmas=[], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + ) + assert short["primary_abstention_reason"] == "fewer_than_two_content_constituents" + assert short["resolved_constituent_synsets"] == [] + + +def test_scored_row_keeps_a_residual_and_status_ignores_agreement(): + extraction = _extraction() + record = score_record( + row_id="row", + surface="alpha beta", + pos="noun", + synset="noun:00000003", + extraction=extraction, + resolutions=["EXACT", "UNIQUE"], + resolved_synsets=["noun:1", "noun:2"], + resolved_lemmas=["alpha", "beta"], + whole_representation="alpha beta (noun): a whole", + constituent_representations=["alpha (noun): one", "beta (noun): two"], + whole_vector=[1.0, 0.0], + constituent_vectors=[[1.0, 0.0], [1.0, 0.0]], + candidate_spec_sha256="spec", + ) + assert record["score_status"] == "SCORED" + assert record["residual_score"] == "0.0000000000" + assert record["primary_abstention_reason"] is None + assert record["composition_operator"] == "normalized_mean_v1" + assert record["whole_vector_hash"] + assert len(record["constituent_vector_hashes"]) == 2 + overflow = score_record( + row_id="row", + surface="alpha beta", + pos="noun", + synset="noun:00000003", + extraction=extraction, + resolutions=["EXACT", "UNIQUE"], + resolved_synsets=["noun:1", "noun:2"], + resolved_lemmas=["alpha", "beta"], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + sequence_overflow=True, + ) + assert overflow["primary_abstention_reason"] == "representation_exceeds_max_sequence_length" + assert evaluation_status(0) == ( + "CANDIDATE_INSUFFICIENT", + "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + ) + assert evaluation_status(3) == ( + "CANDIDATE_DISTRIBUTION_FROZEN", + "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + ) + + +def test_percentile_uses_preregistered_linear_interpolation(): + values = ["0.0000000000", "1.0000000000", "2.0000000000", "3.0000000000"] + assert percentile(values, 25) == "0.7500000000" + assert percentile(values, 50) == "1.5000000000" + assert percentile(values, 75) == "2.2500000000" + empty = distribution([]) + assert empty["status"] == "NOT_COMPUTABLE" + assert empty["median"] is None + filled = distribution(values) + assert filled["status"] == "DESCRIPTIVE" + assert filled["min"] == "0.0000000000" + assert filled["max"] == "3.0000000000" + assert filled["mean"] == "1.5000000000" diff --git a/tests/shadow/test_unbind_screen_v4.py b/tests/shadow/test_unbind_screen_v4.py new file mode 100644 index 00000000..f408094e --- /dev/null +++ b/tests/shadow/test_unbind_screen_v4.py @@ -0,0 +1,236 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v3 import screen +from hyperlexical.unbind_screen_v4 import ( + EmptyLexicon, + ScreenV4Error, + apply_v4, + assess, + measurement_allowed, + normalize_lexical, + rule_surface_violations, +) + +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", +) +V3_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v3.py" +V4_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v4.py" + + +class MapLex: + def __init__(self, nouns, adjectives=()): + self._nouns = nouns + self._adjectives = set(adjectives) + + def noun_lex(self, lemma): + return self._nouns.get(lemma) + + def has_adjective(self, lemma): + return lemma in self._adjectives + + +class BoomLex: + def noun_lex(self, lemma): + raise AssertionError(lemma) + + def has_adjective(self, lemma): + raise AssertionError(lemma) + + +def test_normalization_folds_hyphen_and_keeps_apostrophe(): + assert normalize_lexical("Seventy-Five") == normalize_lexical("seventy five") + assert normalize_lexical("one-hundred") == normalize_lexical("one hundred") + assert "'" in normalize_lexical("one's birthday") + + +def test_v3_number_grammar_does_not_fold_a_hyphen(): + bucket, _rule, _phase = screen("forty-two fifty", "adj", ["forty-two", "fifty"], "a count") + assert bucket == "SECONDARY" + + +def test_v3_spaced_number_still_rejects(): + bucket, rule, _phase = screen("forty two", "adj", ["forty", "two"], "a count") + assert bucket == "REJECT" + assert rule == "productive_numeric_expression" + + +def test_v4_folds_hyphenated_numbers_only_from_secondary(): + moved = apply_v4("SECONDARY", "forty-two fifty", "a count", "adj", EmptyLexicon()) + assert moved["v4_bucket"] == "REJECT" + assert moved["primary_evidence"] == "productive_number" + assert moved["inspected"] is True + held = apply_v4("HIGH_VALUE", "forty-two fifty", "a count", "adj", BoomLex()) + assert held["v4_bucket"] == "HIGH" + assert held["primary_evidence"] is None + assert held["inspected"] is False + rejected = apply_v4("REJECT", "four times", "by a factor of four", "adv", BoomLex()) + assert rejected["v4_bucket"] == "REJECT" + assert rejected["inspected"] is False + + +def test_multiplier_is_a_productive_number(): + moved = apply_v4("SECONDARY", "four times", "by a factor of four", "adv", EmptyLexicon()) + assert moved["v4_bucket"] == "REJECT" + assert moved["primary_evidence"] == "productive_number" + assert moved["supporting_evidence"] == [] + + +def test_comparative_particle_stays_secondary_and_shifted_particle_promotes(): + stayed = apply_v4("SECONDARY", "better off", "in a more fortunate condition", "adj", EmptyLexicon()) + assert stayed["v4_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + promoted = apply_v4("SECONDARY", "zorp up", "having a strong distaste", "adj", EmptyLexicon()) + assert promoted["v4_bucket"] == "HIGH" + assert promoted["primary_evidence"] == "noncompositional_phrasal_binding" + + +def test_literal_particle_with_stem_overlap_stays(): + stayed = apply_v4("SECONDARY", "flare out", "become flared and widen", "verb", EmptyLexicon()) + assert stayed["v4_bucket"] == "SECONDARY" + + +def test_patch_b_frames_are_classes(): + cases = [ + ("strike the ceiling", "get very angry", "verb", "nonliteral_semantic_shift"), + ("like a flash", "without delay", "adv", "conventionalized_idiom"), + ("sure as sunrise", "absolutely certain", "adj", "conventionalized_idiom"), + ("for all practical senses", "in every practical way", "adv", "conventionalized_idiom"), + ("give it a whirl", "whirl", "verb", "conventionalized_idiom"), + ("spin on a coin", "have a small turning radius", "verb", "nonliteral_semantic_shift"), + ("in the civic gaze", "of great interest to the civic world", "adj", "nonliteral_semantic_shift"), + ("stone cold", "without heat", "adj", "fixed_lexicalized_expression"), + ("cooked for finished", "destroyed", "adj", "fixed_lexicalized_expression"), + ("blip a zorp", "show nothing", "verb", "nonliteral_semantic_shift"), + ] + for surface, gloss, pos, evidence in cases: + moved = apply_v4("SECONDARY", surface, gloss, pos, EmptyLexicon()) + assert moved["v4_bucket"] == "HIGH", surface + assert moved["primary_evidence"] == evidence, surface + assert moved["primary_evidence"] not in moved["supporting_evidence"] + + +def test_patch_a_uses_gloss_lexfile_not_the_surface_string(): + person = apply_v4( + "SECONDARY", + "alice brooke carter", + "Scottish physiologist and sibling", + "noun", + MapLex({"scottish": "noun.person", "physiologist": "noun.person"}, {"scottish"}), + ) + assert person["primary_evidence"] == "multi_token_person_name" + role = apply_v4( + "SECONDARY", + "sneak thief", + "a thief who steals", + "noun", + MapLex({"thief": "noun.person"}), + ) + assert role["v4_bucket"] == "SECONDARY" + species = apply_v4( + "SECONDARY", + "ribbon lemur", + "Indian macaque with a tuft", + "noun", + MapLex({"indian": "noun.person", "macaque": "noun.animal"}, {"indian"}), + ) + assert species["primary_evidence"] == "species_or_common_name_referent" + organization = apply_v4( + "SECONDARY", + "warden of the seal", + "a small gang of fighters", + "noun", + MapLex({"small": "noun.cognition", "gang": "noun.group"}, {"small"}), + ) + assert organization["primary_evidence"] == "organization_from_gloss" + medical = apply_v4( + "SECONDARY", + "separation of the choroid", + "visual impairment resulting from the retina becoming separated", + "noun", + MapLex( + { + "impairment": "noun.event", + "retina": "noun.body", + "choroid": "noun.body", + "eye": "noun.body", + }, + {"visual"}, + ), + ) + assert medical["primary_evidence"] == "medical_technical_expression" + + +def test_scorer_source_has_no_probe_surface_or_phrase_literal(): + for path in (V3_PATH, V4_PATH): + source = path.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + constants = [ + node.value + for node in ast.walk(tree) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + ] + for phrase in PROBE_SURFACES: + assert phrase not in constants + assert phrase not in source + + +def test_gate_passes_only_lawful_secondary_moves(): + rows = [ + _replay("alpha beta", "HIGH", "HIGH", "HIGH"), + _replay("gamma delta", "REJECT", "REJECT", "REJECT"), + _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "conventionalized_idiom"), + _replay("eta theta", "SECONDARY", "REJECT", "REJECT", "productive_number"), + _replay("iota kappa", "SECONDARY", "SECONDARY", "SECONDARY"), + ] + report = assess(rows, phrase_specific_rule_fired=False, expected_rows=5) + assert report["regression"] == "REGRESSION_VERIFIED" + assert report["failures"] == [] + assert measurement_allowed(report) is True + broken = list(rows) + broken[0] = _replay("alpha beta", "HIGH", "SECONDARY", "HIGH") + failed = assess(broken, phrase_specific_rule_fired=False, expected_rows=5) + assert failed["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(failed) is False + wrong_patch = list(rows) + wrong_patch[2] = _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "productive_number") + assert assess(wrong_patch, phrase_specific_rule_fired=False, expected_rows=5)["regression"] == "REGRESSION_FAILED" + phrase = assess(rows, phrase_specific_rule_fired=True, expected_rows=5) + assert phrase["regression"] == "REGRESSION_FAILED" + assert phrase["phrase_specific_rule_fired"] is True + + +def test_unknown_bucket_is_refused(): + with pytest.raises(ScreenV4Error): + apply_v4("MAYBE", "alpha beta", "gloss", "noun", EmptyLexicon()) + + +def _replay(surface, v3, v4, operator, primary=None): + return { + "surface": surface, + "v3_bucket": v3, + "v4_bucket": v4, + "operator_bucket": operator, + "primary_evidence": primary, + "supporting_evidence": [], + } diff --git a/tests/shadow/test_unbind_screen_v5.py b/tests/shadow/test_unbind_screen_v5.py new file mode 100644 index 00000000..aae38a84 --- /dev/null +++ b/tests/shadow/test_unbind_screen_v5.py @@ -0,0 +1,266 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v5 import ( + EmptyLexicon, + Entry, + Sense, + apply_v5, + assess, + measurement_allowed, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", +) +V5_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v5.py" + + +class BoomLex: + def entry(self, surface): + raise AssertionError(surface) + + def senses(self, token): + raise AssertionError(token) + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def test_outer_buckets_pass_through_without_inspection(): + high = apply_v5("HIGH_VALUE", "alpha beta", "gloss", "noun", BoomLex()) + assert high["v5_bucket"] == "HIGH" + assert high["v4_bucket"] == "HIGH" + assert high["primary_evidence"] is None + assert high["inspected"] is False + rejected = apply_v5("REJECT", "gamma delta", "gloss", "noun", BoomLex()) + assert rejected["v5_bucket"] == "REJECT" + assert rejected["inspected"] is False + assert rejected["supporting_evidence"] == [] + + +def test_proper_name_and_pertainym_designate(): + named = MapLex( + {"north example laboratory": _entry("noun", "North_Example_Laboratory", "noun.artifact", ("North_Example_Laboratory",))}, + {}, + ) + moved = apply_v5("SECONDARY", "north example laboratory", "a workplace", "noun", named) + assert moved["v5_bucket"] == "REJECT" + assert moved["primary_evidence"] == "referential_terminological_dominance" + pert = MapLex( + {"sample bore": _entry("adj", "sample_bore", "adj.pert", ("sample_bore",))}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "bore": (_sense("bore", "noun", "noun.attribute", "a grade of excellence"),)}, + ) + designated = apply_v5("SECONDARY", "sample bore", "of or relating to a measured width", "adj", pert) + assert designated["v5_bucket"] == "REJECT" + assert designated["supporting_evidence"] == [] + + +def test_exocentric_life_form_rejects_and_endocentric_life_form_stays(): + exo = MapLex( + {"sample bloom": _entry("noun", "sample_bloom", "noun.plant", ("sample_bloom",), (("flower",),))}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "bloom": (_sense("bloom", "noun", "noun.plant", "a flower of a plant"),)}, + ) + # constituent gloss repeats flower, so this row is recoverable and must not use that overlap + # to avoid designation. Designation is decided from the synset, before recoverability. + assert apply_v5("SECONDARY", "sample bloom", "a perennial herb", "noun", exo)["v5_bucket"] == "REJECT" + endo = MapLex( + {"sample tree": _entry("noun", "sample_tree", "noun.plant", ("sample_tree",), (("tree",),))}, + {"sample": (_sense("sample", "adj", "adj.all", "illustrative"),), + "tree": (_sense("tree", "noun", "noun.plant", "a tall woody plant"),)}, + ) + stayed = apply_v5("SECONDARY", "sample tree", "a tall woody plant of the sample kind", "noun", endo) + assert stayed["v5_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + + +def test_compound_category_rejects_unless_an_ordinary_synonym_is_present(): + term = MapLex( + {"phi index": _entry( + "noun", "phi_index", "noun.cognition", ("phi_index",), (("nonparametric_statistic",),) + )}, + {"phi": (_sense("phi", "noun", "noun.communication", "a letter of an alphabet"),), + "index": (_sense("index", "noun", "noun.relation", "a numerical scale"),)}, + ) + assert apply_v5("SECONDARY", "phi index", "a statistic of agreement", "noun", term)["primary_evidence"] == "referential_terminological_dominance" + goods = MapLex( + {"sample goods": _entry( + "noun", "sample_goods", "noun.artifact", ("sample_goods", "haberdashery"), (("soft_goods",),) + )}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "goods": (_sense("goods", "noun", "noun.artifact", "articles of commerce"),)}, + ) + stayed = apply_v5("SECONDARY", "sample goods", "articles of commerce", "noun", goods) + assert stayed["v5_bucket"] == "SECONDARY" + + +def test_adverb_idiom_promotes_and_recoverable_adverb_stays(): + idiom = MapLex( + {"blip out": _entry("adv", "blip_out", "adv.all", ("blip_out", "brusquely"))}, + {"blip": (_sense("blip", "noun", "noun.event", "a small mark on a screen"),), + "out": (_sense("out", "adv", "adv.all", "away from the inside"),)}, + ) + moved = apply_v5("SECONDARY", "blip out", "in a blunt manner", "adv", idiom) + assert moved["v5_bucket"] == "HIGH" + assert moved["primary_evidence"] == "lexicalized_noncompositional" + plain = MapLex( + {"as common": _entry("adv", "as_common", "adv.all", ("as_common", "commonly"))}, + {"common": (_sense("common", "adj", "adj.all", "occurring in the common manner"),)}, + ) + stayed = apply_v5("SECONDARY", "as common", "in the common manner", "adv", plain) + assert stayed["v5_bucket"] == "SECONDARY" + + +def test_orthography_body_and_verb_shift_are_noncompositional(): + ortho = MapLex( + {"to a q": _entry("adv", "to_a_q", "adv.all", ("to_a_q", "to_the_letter"))}, + {}, + ) + assert apply_v5("SECONDARY", "to a q", "in every respect", "adv", ortho)["v5_bucket"] == "HIGH" + body = MapLex( + {"slip of the digit": _entry("noun", "slip_of_the_digit", "noun.communication", ("slip_of_the_digit",))}, + {"slip": (_sense("slip", "noun", "noun.act", "a socially awkward act"),), + "digit": (_sense("digit", "noun", "noun.body", "a finger or toe"),)}, + ) + assert apply_v5("SECONDARY", "slip of the digit", "an accidental mistake in counting", "noun", body)["v5_bucket"] == "HIGH" + shifted = MapLex( + {"dwindle low": _entry("verb", "dwindle_low", "verb.consumption", ("dwindle_low",))}, + {"dwindle": (_sense("dwindle", "verb", "verb.motion", "move fast on foot"),), + "low": (_sense("low", "adj", "adj.all", "not high"),)}, + ) + assert apply_v5("SECONDARY", "dwindle low", "to be spent or finished", "verb", shifted)["v5_bucket"] == "HIGH" + same = MapLex( + {"nudge at": _entry("verb", "nudge_at", "verb.contact", ("nudge_at", "prod"))}, + {"nudge": (_sense("nudge", "verb", "verb.contact", "poke or thrust abruptly"),)}, + ) + assert apply_v5("SECONDARY", "nudge at", "to push against gently", "verb", same)["v5_bucket"] == "SECONDARY" + + +def test_metalinguistic_reduplication_and_missing_lemma_stay(): + discourse = MapLex( + {"by the quip": _entry("adv", "by_the_quip", "adv.all", ("by_the_quip", "incidentally"))}, + {"quip": (_sense("quip", "noun", "noun.communication", "a witty remark"),)}, + ) + assert apply_v5("SECONDARY", "by the quip", "introducing a different topic", "adv", discourse)["v5_bucket"] == "SECONDARY" + repeated = MapLex( + {"plink by plink": _entry("adv", "plink_by_plink", "adv.all", ("plink_by_plink", "gradually"))}, + {"plink": (_sense("plink", "noun", "noun.event", "a short sound"),)}, + ) + assert apply_v5("SECONDARY", "plink by plink", "in a gradual manner", "adv", repeated)["v5_bucket"] == "SECONDARY" + loan = MapLex( + {"al zorp": _entry("adj", "al_zorp", "adj.all", ("al_zorp",))}, + {"al": (_sense("al", "noun", "noun.substance", "a metallic element"),)}, + ) + assert apply_v5("SECONDARY", "al zorp", "of pasta cooked firm", "adj", loan)["v5_bucket"] == "SECONDARY" + + +def test_unknown_surface_stays_secondary(): + stayed = apply_v5("SECONDARY", "brand new coinage", "a fresh phrase", "noun", EmptyLexicon()) + assert stayed["v5_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + assert stayed["inspected"] is True + + +def test_scorer_source_has_no_probe_surface_or_phrase_literal(): + source = V5_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + constants = [ + node.value + for node in ast.walk(tree) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + ] + for phrase in PROBE_SURFACES: + assert phrase not in source + assert phrase not in constants + + +def test_gate_passes_only_lawful_secondary_moves(): + rows = [ + _replay("alpha beta", "HIGH", "HIGH", "HIGH"), + _replay("gamma delta", "REJECT", "REJECT", "REJECT"), + _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "lexicalized_noncompositional"), + _replay("eta theta", "SECONDARY", "REJECT", "REJECT", "referential_terminological_dominance"), + _replay("iota kappa", "SECONDARY", "SECONDARY", "SECONDARY"), + ] + report = assess(rows, phrase_specific_rule_fired=False, expected_rows=5) + assert report["regression"] == "REGRESSION_VERIFIED" + assert report["failures"] == [] + assert report["operator_conflict_on_move"] == 0 + assert measurement_allowed(report) is True + broken = list(rows) + broken[0] = _replay("alpha beta", "HIGH", "SECONDARY", "HIGH") + failed = assess(broken, phrase_specific_rule_fired=False, expected_rows=5) + assert failed["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(failed) is False + conflict = list(rows) + conflict[2] = _replay("epsilon zeta", "SECONDARY", "HIGH", "SECONDARY", "lexicalized_noncompositional") + assert assess(conflict, phrase_specific_rule_fired=False, expected_rows=5)["operator_conflict_on_move"] == 1 + phrase = assess(rows, phrase_specific_rule_fired=True, expected_rows=5) + assert phrase["regression"] == "REGRESSION_FAILED" + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v5("MAYBE", "alpha beta", "gloss", "noun", EmptyLexicon()) + + +def _replay(surface, prior, nxt, operator, primary=None): + return { + "surface": surface, + "v4_bucket": prior, + "v5_bucket": nxt, + "operator_bucket": operator, + "primary_evidence": primary, + "supporting_evidence": [], + } diff --git a/tests/shadow/test_unbind_screen_v6.py b/tests/shadow/test_unbind_screen_v6.py new file mode 100644 index 00000000..0c88c6df --- /dev/null +++ b/tests/shadow/test_unbind_screen_v6.py @@ -0,0 +1,237 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_screen_v5 import Entry, Sense +from hyperlexical.unbind_screen_v6 import ( + COMPOSITIONAL_EVIDENCE, + NONREFERENTIAL_EVIDENCE, + apply_v6, + assess, + measurement_allowed, +) + +PROBE_SURFACES = ( + "keep out", + "on the job", + "hit the roof", + "pop the question", + "all of a sudden", + ".22 caliber", +) +V6_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v6.py" + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def _compositional_lex(): + return MapLex( + {"fully shut": _entry("verb", "Fully_Shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + + +def test_compositional_high_falls_to_secondary_and_stops(): + decision = apply_v6("HIGH", "fully shut", "completely prevent entering", "verb", _compositional_lex()) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == COMPOSITIONAL_EVIDENCE + assert decision["supporting_evidence"] == [] + + +def test_idiomatic_mapping_blocks_post_hoc_rationalization(): + lexicon = MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut", "combust"))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + decision = apply_v6("HIGH", "fully shut", "completely prevent entering", "verb", lexicon) + assert decision["v6_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_single_constituent_gloss_is_not_ordinary_syntax(): + lexicon = MapLex( + {"quite sudden": _entry("adv", "quite_sudden", "adv.all", ("quite_sudden",))}, + { + "quite": (_sense("quite", "adv", "adv.all", "to a degree"),), + "sudden": (_sense("sudden", "adj", "adj.all", "happening without warning"),), + }, + ) + decision = apply_v6("HIGH", "quite sudden", "without warning", "adv", lexicon) + assert decision["v6_bucket"] == "HIGH" + + +def test_state_pertainym_falls_to_secondary(): + lexicon = MapLex( + {"on duty": _entry("adj", "on_duty", "adj.pert", ("on_duty",))}, + {"duty": (_sense("duty", "noun", "noun.act", "work that is a paid activity"),)}, + ) + decision = apply_v6("REJECT", "on duty", "actively engaged in paid work", "adj", lexicon) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == NONREFERENTIAL_EVIDENCE + + +def test_relational_pertainym_stays_a_designation(): + lexicon = MapLex( + {"gun bore": _entry("adj", "gun_bore", "adj.pert", ("gun_bore",))}, + {"bore": (_sense("bore", "noun", "noun.attribute", "a degree of excellence"),)}, + ) + decision = apply_v6( + "REJECT", + "gun bore", + "of or relating to the bore of a gun", + "adj", + lexicon, + ) + assert decision["v6_bucket"] == "REJECT" + assert decision["primary_evidence"] is None + + +def test_sentence_like_naming_gloss_stays_reject(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",), (("terrorist_group",),))}, + { + "alpha": (_sense("alpha", "noun", "noun.communication", "the first letter"),), + "force": (_sense("force", "noun", "noun.group", "a group of people"),), + }, + ) + decision = apply_v6( + "REJECT", + "alpha force", + "a violent group that seeks a separate state for its members", + "noun", + lexicon, + ) + assert decision["v6_bucket"] == "REJECT" + + +def test_provisional_secondary_can_still_reject(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + {"force": (_sense("force", "noun", "noun.group", "a group of people"),)}, + ) + decision = apply_v6("SECONDARY", "alpha force", "a named group", "noun", lexicon) + assert decision["v6_bucket"] == "REJECT" + assert decision["primary_evidence"] == "referential_terminological_dominance" + + +def test_recoverable_secondary_is_not_promoted(): + decision = apply_v6( + "SECONDARY", + "fully shut", + "completely prevent entering", + "verb", + MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ), + ) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] is None + + +def test_previously_correct_is_not_bucket_immutability(): + held = assess( + [ + { + "surface": "still right", + "v5_bucket": "HIGH", + "v6_bucket": "HIGH", + "operator_bucket": "HIGH", + "primary_evidence": None, + "supporting_evidence": [], + }, + { + "surface": "corrected", + "v5_bucket": "REJECT", + "v6_bucket": "SECONDARY", + "operator_bucket": "SECONDARY", + "primary_evidence": NONREFERENTIAL_EVIDENCE, + "supporting_evidence": [], + }, + ], + phrase_specific_rule_fired=False, + expected_rows=2, + ) + assert held["previously_correct_lost"] == 0 + assert held["failures"] == [] + broken = assess( + [ + { + "surface": "was right", + "v5_bucket": "HIGH", + "v6_bucket": "SECONDARY", + "operator_bucket": "HIGH", + "primary_evidence": COMPOSITIONAL_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert broken["previously_correct_lost"] == 1 + assert broken["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(broken) is False + + +def test_direct_outer_swap_fails_the_gate(): + report = assess( + [ + { + "surface": "swapped", + "v5_bucket": "HIGH", + "v6_bucket": "REJECT", + "operator_bucket": "REJECT", + "primary_evidence": "referential_terminological_dominance", + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert report["direct_swaps"] == 1 + assert "HIGH and REJECT swapped directly" in report["failures"] + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v6("MAYBE", "fully shut", "gloss", "verb", _compositional_lex()) + + +def test_probe_surfaces_are_not_rules(): + source = V6_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + assert any(isinstance(node, ast.FunctionDef) and node.name == "apply_v6" for node in ast.walk(tree)) diff --git a/tests/shadow/test_unbind_screen_v7.py b/tests/shadow/test_unbind_screen_v7.py new file mode 100644 index 00000000..23a3ab00 --- /dev/null +++ b/tests/shadow/test_unbind_screen_v7.py @@ -0,0 +1,267 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_screen_v5 import Entry, Sense +from hyperlexical.unbind_screen_v6 import COMPOSITIONAL_EVIDENCE, apply_v6 +from hyperlexical.unbind_screen_v7 import ( + ORDINARY_EVIDENCE, + apply_v7, + assess, + ordinary_compositional_derivation, +) + +PROBE_SURFACES = ( + "keep out", + "to a lesser extent", + "to the letter", + "with child", + "dressed to the nines", + "union jack", + "atomic number 98", + "law of definite proportions", +) +V7_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v7.py" + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def _comparative_lex(): + return MapLex( + {"to a greater degree": _entry("adv", "to_a_greater_degree", "adv.all", ("to_a_greater_degree",))}, + { + "greater": (_sense("greater", "adj", "adj.all", "of greater size"),), + "degree": (_sense("degree", "noun", "noun.attribute", "a position on a scale"),), + }, + ) + + +def _syntactic_lex(lemmas=("fully_shut",)): + return MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", lemmas)}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + + +def _phrasal_lex(): + return MapLex( + {"move out": _entry("verb", "move_out", "verb.motion", ("move_out",))}, + { + "move": (_sense("move", "verb", "verb.motion", "change position"),), + "out": (_sense("out", "adv", "adv.all", "away outside"),), + }, + ) + + +def test_ordinary_comparative_composition_may_fire(): + decision = apply_v7( + "HIGH", + "to a greater degree", + "used to form the comparative of some adjectives and adverbs", + "adv", + _comparative_lex(), + ) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + assert decision["supporting_evidence"] == [] + assert ordinary_compositional_derivation( + "to a greater degree", + "used to form the comparative of some adjectives and adverbs", + _comparative_lex(), + ) + + +def test_ordinary_syntactic_composition_may_fire(): + decision = apply_v7("HIGH", "fully shut", "completely prevent entering", "verb", _syntactic_lex()) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + + +def test_ordinary_phrasal_composition_may_fire(): + decision = apply_v7("HIGH", "move out", "change position away outside", "verb", _phrasal_lex()) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + + +def test_conventionalized_idiom_must_not_fire(): + decision = apply_v7( + "HIGH", + "fully shut", + "completely prevent entering", + "verb", + _syntactic_lex(("fully_shut", "combust")), + ) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_post_hoc_metaphor_must_not_fire(): + decision = apply_v7( + "HIGH", + "fully shut", + "metaphorically completely prevent entering", + "verb", + _syntactic_lex(), + ) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_gloss_resemblance_must_not_fire(): + lexicon = MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely done"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering now"),), + }, + ) + assert apply_v6("HIGH", "fully shut", "completely prevent", "verb", lexicon)["v6_bucket"] == "SECONDARY" + assert apply_v6("HIGH", "fully shut", "completely prevent", "verb", lexicon)["primary_evidence"] == COMPOSITIONAL_EVIDENCE + decision = apply_v7("HIGH", "fully shut", "completely prevent", "verb", lexicon) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + assert decision["primary_evidence"] != COMPOSITIONAL_EVIDENCE + + +def test_named_phrase_is_irrelevant_to_the_high_challenge(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + { + "alpha": (_sense("alpha", "noun", "noun.communication", "the first letter"),), + "force": (_sense("force", "noun", "noun.group", "a group of people"),), + }, + ) + assert ordinary_compositional_derivation("alpha force", "a named group", lexicon) is False + high = apply_v7("HIGH", "alpha force", "a named group", "noun", lexicon) + assert high["v7_bucket"] == "HIGH" + assert high["primary_evidence"] is None + rejected = apply_v7("REJECT", "alpha force", "a named group", "noun", lexicon) + inherited = apply_v6("REJECT", "alpha force", "a named group", "noun", lexicon) + assert rejected["v7_bucket"] == inherited["v6_bucket"] == "REJECT" + assert rejected["primary_evidence"] != ORDINARY_EVIDENCE + + +def test_demotion_stops_at_secondary(): + lexicon = _comparative_lex() + gloss = "used to form the comparative of some adjectives and adverbs" + demoted = apply_v7("HIGH", "to a greater degree", gloss, "adv", lexicon) + assert demoted["v7_bucket"] == "SECONDARY" + stopped = apply_v7("SECONDARY", "to a greater degree", gloss, "adv", lexicon) + assert stopped["v7_bucket"] == "SECONDARY" + assert stopped["primary_evidence"] is None + + +def test_reject_behavior_matches_v6_and_secondary_is_not_reopened(): + pertainym = MapLex( + {"on duty": _entry("adj", "on_duty", "adj.pert", ("on_duty",))}, + {"duty": (_sense("duty", "noun", "noun.act", "work that is a paid activity"),)}, + ) + rejected = apply_v7("REJECT", "on duty", "actively engaged in paid work", "adj", pertainym) + inherited = apply_v6("REJECT", "on duty", "actively engaged in paid work", "adj", pertainym) + assert rejected["v7_bucket"] == inherited["v6_bucket"] == "SECONDARY" + assert rejected["primary_evidence"] == inherited["primary_evidence"] == "nonreferential_lexical_use" + naming = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + {"force": (_sense("force", "noun", "noun.group", "a group of people"),)}, + ) + assert apply_v6("SECONDARY", "alpha force", "a named group", "noun", naming)["v6_bucket"] == "REJECT" + held = apply_v7("SECONDARY", "alpha force", "a named group", "noun", naming) + assert held["v7_bucket"] == "SECONDARY" + assert held["primary_evidence"] is None + + +def test_previously_correct_is_not_bucket_immutability(): + allowed = assess( + [ + { + "surface": "may retreat", + "v6_bucket": "HIGH", + "v7_bucket": "SECONDARY", + "operator_bucket": "SECONDARY", + "primary_evidence": ORDINARY_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert allowed["previously_correct_lost"] == 0 + assert allowed["correct_high_lost"] == 0 + assert allowed["regression"] == "REGRESSION_VERIFIED" + assert allowed["measurement_eligible"] is False + broken = assess( + [ + { + "surface": "was right", + "v6_bucket": "HIGH", + "v7_bucket": "SECONDARY", + "operator_bucket": "HIGH", + "primary_evidence": ORDINARY_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert broken["previously_correct_lost"] == 1 + assert broken["correct_high_lost"] == 1 + assert broken["regression"] == "REGRESSION_FAILED" + + +def test_direct_outer_swap_fails_the_gate(): + report = assess( + [ + { + "surface": "swapped", + "v6_bucket": "HIGH", + "v7_bucket": "REJECT", + "operator_bucket": "REJECT", + "primary_evidence": "referential_terminological_dominance", + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert report["direct_swaps"] == 1 + assert "HIGH and REJECT swapped directly" in report["failures"] + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v7("MAYBE", "move out", "gloss", "verb", _phrasal_lex()) + + +def test_probe_surfaces_are_not_rules(): + source = V7_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + assert "compositional_recoverability" not in source + tree = ast.parse(source) + assert any(isinstance(node, ast.FunctionDef) and node.name == "apply_v7" for node in ast.walk(tree)) diff --git a/tests/shadow/test_unbind_sense_screen_v1.py b/tests/shadow/test_unbind_sense_screen_v1.py new file mode 100644 index 00000000..dd1e1fb9 --- /dev/null +++ b/tests/shadow/test_unbind_sense_screen_v1.py @@ -0,0 +1,268 @@ +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_sense_screen_v1 import Pointer, Synset, classify, parse_data_line + +SCORER = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_sense_screen_v1.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _synset(ss_type, lemmas, pointers=()): + return Synset("00000000", ss_type, tuple(lemmas), tuple(pointers)) + + +def _pointer(symbol, source=0, target=0, offset="00000011", pos="n"): + return Pointer(symbol, offset, pos, source, target) + + +def _class(surface, gloss, synset, exceptions=None, targets=None): + return classify(surface, gloss, synset, exceptions or {}, targets or {}) + + +def test_instance_hypernym_is_referential(): + found = _class("alpha beta", "a particular named thing", _synset("n", ("alpha_beta",), (_pointer("@i"),))) + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + assert found["primary_evidence_code"] == "referential_designation" + assert found["evidence_source"] == "synset.instance_hypernym" + assert found["confidence_status"] == "DETERMINATE" + + +def test_lifespan_is_referential_and_instance_wins_when_both_fire(): + life = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",))) + assert life["evidence_source"] == "synset.gloss.lifespan" + both = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",), (_pointer("@i"),))) + assert both["sense_class"] == "REFERENTIAL" + assert both["evidence_source"] == "synset.instance_hypernym" + + +def test_year_range_without_parentheses_is_not_a_lifespan(): + found = _class("alpha beta", "a span from 1870-1949", _synset("n", ("alpha_beta",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_geographic_allusion_and_capitalization_are_not_referential(): + gloss = "a sudden turning point in a life (similar to a story on the road from one city to another)" + found = _class( + "path to example", + gloss, + _synset("n", ("Path_to_Example",), (_pointer("@", source=0),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["confidence_status"] == "INSUFFICIENT" + + +def test_technical_gloss_without_a_record_signal_stays_ambiguous(): + found = _class("sample statute", "a law about a measurement of a species", _synset("n", ("sample_statute",))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_unrelated_single_word_colemma_is_noncompositional(): + found = _class("alpha beta", "a stored predicate", _synset("v", ("alpha_beta", "exclude"))) + assert found["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL" + assert found["bucket"] == "HIGH" + assert found["evidence_source"] == "synset.lemmas.unrelated_single_word" + + +def test_constituent_and_its_exception_form_are_not_unrelated(): + bare = _class("alpha beta", "a predicate", _synset("v", ("alpha_beta", "alpha"))) + assert bare["sense_class"] == "AMBIGUOUS" + inflected = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "alphas")), + {"alphas": {"alpha"}, "alpha": {"alphas"}}, + ) + assert inflected["sense_class"] == "AMBIGUOUS" + + +def test_lexical_pointer_without_a_constituent_relation_is_ambiguous(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta",), (_pointer("!", source=1, target=1),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["evidence_source"] == "none" + + +def test_semantic_pointer_is_not_a_lexical_unit(): + found = _class( + "alpha beta", + "a predicate", + _synset("n", ("alpha_beta",), (_pointer("!", source=0, target=0),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_pointer_on_the_other_lemma_does_not_count(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "gamma_delta"), (_pointer("+", source=2, target=1),)), + targets={("n", "00000011"): ("alpha",)}, + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_derivation_back_to_a_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("v", ("alpha_beta",), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["evidence_source"] == "synset.lexical_pointer.derivation_or_pertainym_to_constituent" + + +def test_pertainym_to_an_inflected_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("r", ("alpha_beta",), (_pointer("\\", source=1, target=1, pos="a"),)), + exceptions={"betas": {"beta"}, "beta": {"betas"}}, + targets={("a", "00000011"): ("betas",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["confidence_status"] == "DETERMINATE" + + +def test_unrelated_equivalent_overrides_a_constituent_pointer(): + found = _class( + "alpha beta", + "a stored predicate", + _synset("v", ("alpha_beta", "exclude"), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL" + + +def test_one_token_alternation_is_ordinary_composition(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["evidence_source"] == "synset.lemmas.productive_alternation" + + +def test_alternation_that_excludes_the_surface_does_not_fire(): + found = _class( + "alpha beta gamma", + "a predicate", + _synset("n", ("alpha_beta_gamma", "as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_comparative_and_superlative_formulas_are_ordinary_composition(): + comparative = _class("alpha beta", "used to form the comparative of some words", _synset("r", ("alpha_beta",))) + superlative = _class("alpha beta", "used to form the superlative of some words", _synset("r", ("alpha_beta",))) + buried = _class("alpha beta", "a phrase used to form the comparative later", _synset("r", ("alpha_beta",))) + assert comparative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert comparative["evidence_source"] == "synset.gloss.grammatical_operator" + assert superlative["evidence_source"] == "synset.gloss.grammatical_operator" + assert buried["sense_class"] == "AMBIGUOUS" + + +def test_membership_alone_is_not_secondary(): + found = _class("alpha beta", "a listed phrase", _synset("n", ("alpha_beta",), (_pointer("@"),))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_adjective_without_a_unit_signal_is_ambiguous(): + found = _class("alpha beta", "a listed modifier", _synset("a", ("alpha_beta",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_referential_yes_conflicts_with_a_productive_frame(): + found = _class( + "as wide as needed", + "a particular (1870-1949)", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed"), (_pointer("@i"),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_unit_yes_and_no_conflict(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed", "exclude")), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_hyphen_and_underscore_count_as_the_same_lemma(): + found = _class( + "full of the moon", + "a listed name", + _synset("n", ("full-of-the-moon",)), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_high_ambiguity_is_left_in_place(): + rows = [ + _class(f"item {index}", "a listed phrase", _synset("n", (f"item_{index}",))) + for index in range(5) + ] + assert [row["sense_class"] for row in rows] == ["AMBIGUOUS"] * 5 + assert [row["bucket"] for row in rows] == ["QUARANTINE"] * 5 + + +def test_data_line_parser_keeps_hex_word_numbers_and_stops_before_frames(): + line = "00013172 29 v 01 bungle 0 003 @ 00010435 v 0000 + 00074790 n 0104 + 09879744 n 0101 01 + 02 00 | spoil by behaving clumsily" + synset, gloss = parse_data_line(line) + assert gloss == "spoil by behaving clumsily" + assert synset.lemmas == ("bungle",) + assert len(synset.pointers) == 3 + assert synset.pointers[1].symbol == "+" + assert synset.pointers[1].source == 1 + assert synset.pointers[1].target == 4 + wide = ( + "00002137 03 n 02 abstraction 0 abstract_entity 0 010 " + "@ 00001740 n 0000 + 00692329 v 0101 ~ 00023100 n 0000 ~ 00024264 n 0000 " + "~ 00031264 n 0000 ~ 00031921 n 0000 ~ 00033020 n 0000 ~ 00033615 n 0000 " + "~ 05810143 n 0000 ~ 07999699 n 0000 | a general concept" + ) + parsed_wide, _gloss = parse_data_line(wide) + assert len(parsed_wide.pointers) == 10 + assert parsed_wide.pointers[1].symbol == "+" + assert parsed_wide.pointers[1].source == 1 + assert parsed_wide.pointers[1].target == 1 + lifespan = "00000001 11 n 01 alpha_beta 0 001 @ 00000002 n 0000 | a made example (1900-1910); extra" + parsed, first = parse_data_line(lifespan) + found = _class("alpha beta", first, parsed) + assert found["sense_class"] == "REFERENTIAL" + assert found["evidence_source"] == "synset.gloss.lifespan" + + +def test_probe_surfaces_are_not_rules(): + source = SCORER.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] diff --git a/tests/shadow/test_unbind_sense_screen_v2.py b/tests/shadow/test_unbind_sense_screen_v2.py new file mode 100644 index 00000000..816e8a3c --- /dev/null +++ b/tests/shadow/test_unbind_sense_screen_v2.py @@ -0,0 +1,304 @@ +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_sense_screen_v2 import Pointer, Synset, classify, parse_data_line + +SCORER = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_sense_screen_v2.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _synset(ss_type, lemmas, pointers=()): + return Synset("00000000", ss_type, tuple(lemmas), tuple(pointers)) + + +def _pointer(symbol, source=0, target=0, offset="00000011", pos="n"): + return Pointer(symbol, offset, pos, source, target) + + +def _class(surface, gloss, synset, exceptions=None, targets=None): + return classify(surface, gloss, synset, exceptions or {}, targets or {}) + + +def test_instance_hypernym_is_referential_and_wins_over_a_colemma(): + found = _class( + "alpha beta", + "a particular named thing", + _synset("n", ("alpha_beta", "exclude"), (_pointer("@i"),)), + ) + assert found["referential_state"] == "YES" + assert found["lexicalized_state"] == "YES" + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + assert found["primary_evidence_code"] == "referential_designation" + assert found["evidence_sources"][0] == "synset.instance_hypernym" + assert "whole_expression_lexicalization" in found["supporting_evidence_codes"] + assert found["confidence_status"] == "DETERMINATE" + + +def test_lifespan_is_referential_until_an_instance_pointer_is_also_present(): + life = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",))) + assert life["evidence_sources"][0] == "synset.gloss.lifespan" + both = _class( + "alpha beta", + "a person (1870-1949)", + _synset("n", ("alpha_beta",), (_pointer("@i"),)), + ) + assert both["evidence_sources"][0] == "synset.instance_hypernym" + + +def test_year_range_without_parentheses_is_not_referential(): + found = _class("alpha beta", "a span from 1870-1949", _synset("n", ("alpha_beta",))) + assert found["referential_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + + +def test_allusion_and_capitalization_are_not_referential(): + gloss = "a sudden turning point in a life (similar to a story on the road from one city to another)" + found = _class("path to example", gloss, _synset("n", ("Path_to_Example",), (_pointer("@", source=0),))) + assert found["referential_state"] == "UNKNOWN" + assert found["lexicalized_state"] == "UNKNOWN" + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_technical_gloss_stays_ambiguous(): + found = _class("sample statute", "a law about a measurement of a species", _synset("n", ("sample_statute",))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_colemma_is_lexicalized_yes_and_not_high(): + found = _class("alpha beta", "a stored predicate", _synset("v", ("alpha_beta", "exclude"))) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "UNKNOWN" + assert found["compositional_state"] != "NO" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["primary_evidence_code"] == "insufficient_record_evidence" + assert found["supporting_evidence_codes"] == ["whole_expression_lexicalization"] + assert found["evidence_sources"] == ["synset.lemmas.unrelated_single_word"] + assert found["confidence_status"] == "INSUFFICIENT" + + +def test_constituent_and_its_inflection_are_not_an_unrelated_colemma(): + bare = _class("alpha beta", "a predicate", _synset("v", ("alpha_beta", "alpha"))) + inflected = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "alphas")), + {"alphas": {"alpha"}, "alpha": {"alphas"}}, + ) + assert bare["lexicalized_state"] == "UNKNOWN" + assert inflected["lexicalized_state"] == "UNKNOWN" + + +def test_lexical_pointer_without_a_constituent_relation_stays_ambiguous(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta",), (_pointer("!", source=1, target=1),)), + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_semantic_pointer_and_a_pointer_on_another_lemma_do_not_lexicalize(): + semantic = _class( + "alpha beta", + "a predicate", + _synset("n", ("alpha_beta",), (_pointer("!", source=0, target=0),)), + ) + other = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "gamma_delta"), (_pointer("+", source=2, target=1),)), + targets={("n", "00000011"): ("alpha",)}, + ) + assert semantic["lexicalized_state"] == "UNKNOWN" + assert other["lexicalized_state"] == "UNKNOWN" + assert other["compositional_state"] == "UNKNOWN" + + +def test_constituent_derivation_is_lexicalized_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("v", ("alpha_beta",), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["primary_evidence_code"] == "compositional_semantic_relation" + assert found["supporting_evidence_codes"] == ["whole_expression_lexicalization"] + assert found["evidence_sources"][0] == "synset.lexical_pointer.derivation_or_pertainym_to_constituent" + + +def test_pertainym_to_an_inflected_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("r", ("alpha_beta",), (_pointer("\\", source=1, target=1, pos="a"),)), + exceptions={"betas": {"beta"}, "beta": {"betas"}}, + targets={("a", "00000011"): ("betas",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["referential_state"] == "NO" + + +def test_colemma_plus_constituent_is_secondary_not_high(): + found = _class( + "alpha beta", + "a stored predicate", + _synset("v", ("alpha_beta", "exclude"), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + + +def test_colemma_plus_alternation_is_lexicalized_compositional(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed", "exclude")), + ) + assert found["referential_state"] == "NO" + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["primary_evidence_code"] == "productive_grammatical_frame" + assert "whole_expression_lexicalization" in found["supporting_evidence_codes"] + + +def test_alternation_alone_is_compositional_yes_and_ambiguous(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["referential_state"] == "NO" + assert found["lexicalized_state"] == "UNKNOWN" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["supporting_evidence_codes"] == ["productive_grammatical_frame"] + + +def test_alternation_that_excludes_the_surface_does_not_fire(): + found = _class( + "alpha beta gamma", + "a predicate", + _synset("n", ("alpha_beta_gamma", "as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + + +def test_operator_gloss_is_ordinary_composition(): + comparative = _class("alpha beta", "used to form the comparative of some words", _synset("r", ("alpha_beta",))) + superlative = _class("alpha beta", "used to form the superlative of some words", _synset("r", ("alpha_beta",))) + buried = _class("alpha beta", "a phrase used to form the comparative later", _synset("r", ("alpha_beta",))) + assert comparative["lexicalized_state"] == "NO" + assert comparative["compositional_state"] == "YES" + assert comparative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert comparative["bucket"] == "SECONDARY" + assert comparative["primary_evidence_code"] == "productive_grammatical_frame" + assert comparative["evidence_sources"][0] == "synset.gloss.grammatical_operator" + assert superlative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert buried["sense_class"] == "AMBIGUOUS" + assert buried["lexicalized_state"] == "UNKNOWN" + + +def test_operator_gloss_plus_colemma_conflicts_and_is_not_high(): + found = _class( + "alpha beta", + "used to form the comparative of some words", + _synset("r", ("alpha_beta", "exclude")), + ) + assert found["lexicalized_state"] == "CONFLICT" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["primary_evidence_code"] == "conflicting_record_evidence" + assert found["confidence_status"] == "CONTRADICTORY" + + +def test_membership_alone_is_not_secondary_or_high(): + noun = _class("alpha beta", "a listed phrase", _synset("n", ("alpha_beta",), (_pointer("@"),))) + adjective = _class("alpha beta", "a listed modifier", _synset("a", ("alpha_beta",))) + assert noun["referential_state"] == "UNKNOWN" + assert noun["sense_class"] == "AMBIGUOUS" + assert noun["evidence_sources"] == ["none"] + assert adjective["referential_state"] == "NO" + assert adjective["sense_class"] == "AMBIGUOUS" + + +def test_instance_pointer_on_an_alternation_stays_referential(): + found = _class( + "as wide as needed", + "a particular (1870-1949)", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed"), (_pointer("@i"),)), + ) + assert found["referential_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + + +def test_hyphen_and_underscore_count_as_the_same_lemma(): + found = _class("full of the moon", "a listed name", _synset("n", ("full-of-the-moon",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_colemma_rows_are_not_repaired_into_high(): + rows = [ + _class(f"item {index}", "a listed phrase", _synset("n", (f"item_{index}", "exclude"))) + for index in range(5) + ] + assert [row["sense_class"] for row in rows] == ["AMBIGUOUS"] * 5 + assert [row["bucket"] for row in rows] == ["QUARANTINE"] * 5 + assert [row["compositional_state"] for row in rows] == ["UNKNOWN"] * 5 + + +def test_decimal_pointer_count_stops_before_frames(): + line = "00013172 29 v 01 bungle 0 003 @ 00010435 v 0000 + 00074790 n 0104 + 09879744 n 0101 01 + 02 00 | spoil by behaving clumsily" + synset, gloss = parse_data_line(line) + assert len(synset.pointers) == 3 + wide = ( + "00002137 03 n 02 abstraction 0 abstract_entity 0 010 " + "@ 00001740 n 0000 + 00692329 v 0101 ~ 00023100 n 0000 ~ 00024264 n 0000 " + "~ 00031264 n 0000 ~ 00031921 n 0000 ~ 00033020 n 0000 ~ 00033615 n 0000 " + "~ 05810143 n 0000 ~ 07999699 n 0000 | a general concept" + ) + parsed, _gloss = parse_data_line(wide) + assert len(parsed.pointers) == 10 + found = _class("alpha beta", gloss, synset) + assert found["compositional_state"] == "UNKNOWN" + + +def test_probe_surfaces_are_not_rules(): + source = SCORER.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == []