diff --git a/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py new file mode 100644 index 00000000..4dd12090 --- /dev/null +++ b/scripts/shadow/hyperlexical/constituent_sense_resolution_v1.py @@ -0,0 +1,376 @@ +"""Deterministic constituent-sense resolution for a Hyperlex multiword row. + +Structural WordNet evidence outranks contextual overlap. Extended Lesk uses +gloss text only. Operator labels, residual scores, and embeddings are not inputs. +""" + +from __future__ import annotations + +from hyperlexical.km_candidate_evaluation import lookup_key +from hyperlexical.semantic_compositionality_residual import ( + STRUCTURAL_TOKENS, + neighbor_keys, +) + +RULE_VERSION = "RUNE.CONSTITUENT_SENSE_RESOLUTION.v1" +EXTENDED_LESK = "EXTENDED_LESK_V1" +STRUCTURAL_EXACT = "STRUCTURAL_EXACT" +UNIQUE_LEMMA = "UNIQUE_LEMMA" +METHOD_NONE = "NONE" +EXACT = "EXACT" +RESOLVED = "RESOLVED" +AMBIGUOUS = "AMBIGUOUS" +UNRESOLVED = "UNRESOLVED" +RESIDUAL_READY = "RESIDUAL_READY" +ROW_UNKNOWN = "UNKNOWN" +MINIMUM_MARGIN = 1 +RELATION_DEPTH = 1 +LEXICAL_POINTER_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +RELATION_EXPANSION_SYMBOLS = frozenset({"@", "+", "\\", "&", "^", "="}) +READY_STATUSES = frozenset({EXACT, RESOLVED}) + +_RESEARCH_QUESTION = ( + "Can Hyperlex deterministically resolve enough content-constituent WordNet " + "senses to construct a semantic compositional baseline, without using " + "operator labels or residual scores as evidence?" +) +_LESK_SCORE_RULE = ( + "Greedy longest contiguous token overlap. Each match adds the square of its " + "length. Matched tokens are removed. Search restarts until no token matches." +) +_TIE_RULE = ( + "Integer scores. If the top score minus the second score is below the minimum " + "margin, the constituent is AMBIGUOUS. Equal scores are a tie. No sense is chosen." +) +_STOPWORD_RULE = ( + "Tokens whose lookup form is in the residual v1 structural class are removed, " + "as are tokens shorter than two characters. No stemmer and no lemmatizer." +) +_CONTEXT_RULE = ( + "Context is the parent frozen gloss plus the first gloss clause of every other " + "content constituent in the row that structural evidence already resolved. " + "Lesk output is not context." +) +_CANDIDATE_TEXT_RULE = ( + "A candidate contributes its full PWN 3.0 gloss, including quoted examples, " + "then the first gloss clause of each depth-1 related synset." +) +_CIRCULARITY_RULE = ( + "The parent gloss is local context. It is not an embedding target. Parent sense " + "is not constituent sense." +) +_NORMALIZATION_RULE = ( + "casefold, apostrophe fold, split on characters outside letters, digits, apostrophe, and hyphen" +) +_TARGET_ZERO = "does not name a lemma" +_TIER_CONFLICT = "more than one structural target is AMBIGUOUS and blocks Lesk" +_TIER_STRUCTURAL = ( + "one lexical pointer whose target word number is nonzero and whose lemma matches " + "the constituent or one exception hop" +) +_TIER_UNIQUE = "no structural target and one lemma synset across noun, verb, adj, and adv" +_TIER_UNRESOLVED = "no structural target and no lemma synset" +_READY_REQUIRES = "at_least_two_content_constituents_and_each_is_EXACT_or_RESOLVED" +_EMPTY_ROW = ( + "Every content constituent is resolved. A row with fewer than two content " + "constituents stays UNKNOWN. A row with none emits one abstention record." +) + + +def resolver_policy() -> dict: + """Return the preregistered resolver. The dict has no development counts.""" + return { + "applied_at_freeze": False, + "candidate_text": _CANDIDATE_TEXT_RULE, + "circularity": _CIRCULARITY_RULE, + "context": _CONTEXT_RULE, + "empty_row": _EMPTY_ROW, + "encoded_at_freeze": False, + "extended_lesk": { + "examples_included_for_candidate": True, + "examples_included_for_related_synsets": False, + "method": EXTENDED_LESK, + "minimum_margin": MINIMUM_MARGIN, + "minimum_token_length": 2, + "normalization": _NORMALIZATION_RULE, + "relation_depth": RELATION_DEPTH, + "relation_symbols": sorted(RELATION_EXPANSION_SYMBOLS), + "related_gloss": "first_clause", + "score": _LESK_SCORE_RULE, + "stemming": False, + "stopwords": _STOPWORD_RULE, + "structural_tokens": sorted(STRUCTURAL_TOKENS), + "tie": _TIE_RULE, + "zero_overlap_is_a_tie": True, + }, + "forbidden_inputs": [ + "embedding_similarity", + "language_model_judge", + "manual_selection", + "operator_labels", + "phrase_exceptions", + "residual_scores", + "unbind_bucket", + ], + "measurement_eligible": False, + "research_question": _RESEARCH_QUESTION, + "row_rule": { + "ready": RESIDUAL_READY, + "ready_requires": _READY_REQUIRES, + "unknown": ROW_UNKNOWN, + }, + "rule": RULE_VERSION, + "state_at_freeze": "SPEC_FROZEN", + "status_rule": { + "baseline_ambiguous_not_resolved_gt_half": { + "finding": "CONSTITUENT_WSD_COVERAGE_INSUFFICIENT", + "next_legal_transition": "MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION", + "status": "COVERAGE_INSUFFICIENT", + }, + "else_if_residual_ready_high_and_secondary_positive": { + "finding": None, + "next_legal_transition": "RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION", + "status": "COVERAGE_NECESSARY_CONDITION_MET", + }, + "else": { + "finding": None, + "next_legal_transition": "CONSTITUENT_SENSE_RESOLUTION_REVISION_AUTHORIZATION", + "status": "NECESSARY_CONDITION_UNMET", + }, + }, + "structural_pointers": sorted(LEXICAL_POINTER_SYMBOLS), + "target_word_number_zero": _TARGET_ZERO, + "tier1": { + "conflict": _TIER_CONFLICT, + "structural": _TIER_STRUCTURAL, + "unique_lemma": _TIER_UNIQUE, + "unresolved": _TIER_UNRESOLVED, + }, + } + + +def lesk_tokens(text: str) -> list[str]: + """Normalize gloss text into overlap tokens.""" + folded = text.casefold().replace("\u2019", "'").replace("\u2018", "'").replace("`", "'") + tokens = [] + current = [] + for character in folded: + if character.isascii() and (character.isalnum() or character in "'-"): + current.append(character) + continue + if current: + tokens.append("".join(current).strip("'-")) + current = [] + if current: + tokens.append("".join(current).strip("'-")) + kept = [] + for token in tokens: + if len(token) < 2 or lookup_key(token) in STRUCTURAL_TOKENS: + continue + kept.append(token) + return kept + + +def longest_overlap(context: list[str], candidate: list[str]) -> tuple[int, int, int] | None: + """Return the leftmost longest contiguous match as context start, candidate start, length.""" + best_length = 0 + best = None + for context_start, token in enumerate(context): + for candidate_start, other in enumerate(candidate): + if other != token: + continue + length = 0 + while ( + context_start + length < len(context) + and candidate_start + length < len(candidate) + and context[context_start + length] == candidate[candidate_start + length] + ): + length += 1 + if length > best_length: + best_length = length + best = (context_start, candidate_start, length) + return best + + +def overlap_score(context: list[str], candidate: list[str]) -> int: + """Score one candidate. The same token cannot support two matches.""" + left = list(context) + right = list(candidate) + score = 0 + while left and right: + found = longest_overlap(left, right) + if found is None: + break + context_start, candidate_start, length = found + score += length * length + del left[context_start : context_start + length] + del right[candidate_start : candidate_start + length] + return score + + +def structural_synset_ids( + pointers: list[tuple[str, int, str, str]], + constituent: str, + exceptions: dict[str, set[str]], +) -> list[str]: + """Collect lexical-pointer targets that name this constituent.""" + keys = neighbor_keys(constituent, exceptions) + found = [] + for symbol, target_word, synset_id, lemma in pointers: + if symbol not in LEXICAL_POINTER_SYMBOLS or target_word <= 0 or not lemma: + continue + if lookup_key(lemma) not in keys: + continue + found.append(synset_id) + return found + + +def constituent_pos(selected_synset: str | None, candidate_synsets: list[str]) -> str | None: + if selected_synset: + return selected_synset.split(":", 1)[0] + poses = {synset_id.split(":", 1)[0] for synset_id in candidate_synsets} + if len(poses) == 1: + return next(iter(poses)) + return None + + +def resolve_constituent_sense( + structural_ids: list[str], + lexical_ids: list[str], + lesk_scores: list[tuple[str, int]] | None, +) -> dict: + """Resolve one constituent. Structural evidence blocks contextual overlap.""" + structural = list(dict.fromkeys(structural_ids)) + lexical = list(dict.fromkeys(lexical_ids)) + if len(structural) == 1: + return _decision( + EXACT, + STRUCTURAL_EXACT, + structural[0], + None, + None, + None, + "structural_exact_pointer", + [f"pointer_target:{structural[0]}"], + ) + if len(structural) > 1: + return _decision( + AMBIGUOUS, + STRUCTURAL_EXACT, + None, + None, + None, + None, + "structural_pointer_conflict", + [f"pointer_target:{synset_id}" for synset_id in structural], + ) + if len(lexical) == 1: + return _decision( + EXACT, + UNIQUE_LEMMA, + lexical[0], + None, + None, + None, + "unique_lemma_synset", + ["lexical_candidates:1"], + ) + if not lexical: + return _decision( + UNRESOLVED, + METHOD_NONE, + None, + None, + None, + None, + "no_candidate_synset", + [], + ) + if lesk_scores is None: + raise RuntimeError("ambiguous constituent has no Lesk scores") + scored = {synset_id: score for synset_id, score in lesk_scores} + if set(scored) != set(lexical) or len(lesk_scores) != len(scored): + raise RuntimeError("Lesk scores do not match the candidate synsets") + if any(not isinstance(score, int) for score in scored.values()): + raise RuntimeError("Lesk scores must be integers") + ordered = sorted(scored.items(), key=lambda item: (-item[1], item[0])) + top_id, top_score = ordered[0] + _second_id, second_score = ordered[1] + margin = top_score - second_score + if margin < MINIMUM_MARGIN: + code = "extended_lesk_tie" if margin == 0 else "extended_lesk_margin" + return _decision( + AMBIGUOUS, + EXTENDED_LESK, + None, + top_score, + second_score, + margin, + code, + [f"lesk_candidates:{len(ordered)}"], + ) + return _decision( + RESOLVED, + EXTENDED_LESK, + top_id, + top_score, + second_score, + margin, + "extended_lesk_margin", + [f"lesk_candidates:{len(ordered)}"], + ) + + +def _decision(status, method, selected, top_score, second_score, margin, code, support) -> dict: + return { + "margin": margin, + "primary_evidence_code": code, + "resolution_method": method, + "resolution_status": status, + "second_score": second_score, + "selected_synset": selected, + "supporting_evidence": list(support), + "top_score": top_score, + } + + +def row_resolution_status(content_count: int, statuses: list[str]) -> str: + if content_count < 2 or len(statuses) != content_count: + return ROW_UNKNOWN + if all(status in READY_STATUSES for status in statuses): + return RESIDUAL_READY + return ROW_UNKNOWN + + +def coverage_decision( + *, + baseline_ambiguous: int, + baseline_ambiguous_not_resolved: int, + residual_ready_high: int, + residual_ready_secondary: int, +) -> dict: + """Apply the preregistered coverage rule. The counts are not a threshold search.""" + if min(baseline_ambiguous, baseline_ambiguous_not_resolved, residual_ready_high, residual_ready_secondary) < 0: + raise RuntimeError("negative coverage count") + insufficient = baseline_ambiguous > 0 and baseline_ambiguous_not_resolved * 2 > baseline_ambiguous + necessary = residual_ready_high > 0 and residual_ready_secondary > 0 + if insufficient: + status = "COVERAGE_INSUFFICIENT" + finding = "CONSTITUENT_WSD_COVERAGE_INSUFFICIENT" + transition = "MODEL_BASED_WSD_CANDIDATE_EVALUATION_AUTHORIZATION" + elif necessary: + status = "COVERAGE_NECESSARY_CONDITION_MET" + finding = None + transition = "RESIDUAL_REPLAY_WITH_RESOLVED_SENSES_AUTHORIZATION" + else: + status = "NECESSARY_CONDITION_UNMET" + finding = None + transition = "CONSTITUENT_SENSE_RESOLUTION_REVISION_AUTHORIZATION" + return { + "candidate_status": status, + "coverage_finding": finding, + "necessary_condition_met": necessary, + "next_legal_transition": transition, + "next_transition_authorized": False, + "state": "DEVELOPMENT_ANALYZED", + } diff --git a/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py b/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py new file mode 100644 index 00000000..fa66b52a --- /dev/null +++ b/scripts/shadow/hyperlexical/magpie_candidate_evaluation.py @@ -0,0 +1,363 @@ +"""Development-only matcher for the MAGPIE candidate evaluation. + +Surface identity is orthographic. A surface hit does not assign +semantic noncompositionality. Sense alignment requires identifier +equality with the supplied synset. This module does not read operator +labels, gloss text, or a model judgment. +""" + +from __future__ import annotations + +from collections import Counter +from collections.abc import Iterable +from decimal import Decimal + +EXACT = "EXACT" +NORMALIZED = "NORMALIZED" +VARIANT = "VARIANT" +NONE = "NONE" +AMBIGUOUS = "AMBIGUOUS" +SURFACE_MATCHES = frozenset({EXACT, NORMALIZED, VARIANT, NONE, AMBIGUOUS}) + +ALIGNED_IDIOMATIC = "ALIGNED_IDIOMATIC" +ALIGNED_LITERAL = "ALIGNED_LITERAL" +MIXED = "MIXED" +CONFLICT = "CONFLICT" +UNKNOWN = "UNKNOWN" +ALIGNMENTS = frozenset({ALIGNED_IDIOMATIC, ALIGNED_LITERAL, MIXED, CONFLICT, UNKNOWN}) + +YES = "YES" +NO = "NO" + +CODE_ALIGNED_IDIOMATIC = "magpie_aligned_idiomatic" +CODE_ALIGNED_LITERAL = "magpie_aligned_literal" +CODE_MIXED = "magpie_mixed_usage" +CODE_SURFACE = "magpie_surface_only" +CODE_CONFLICT = "magpie_sense_conflict" +CODE_NONE = "magpie_no_match" + +_SENSE_FIELDS = frozenset({"synset", "synset_offset", "sense_key", "wordnet_offset"}) +_APOSTROPHES = str.maketrans( + { + "\u2019": "'", + "\u2018": "'", + "\u02bc": "'", + "\u2032": "'", + "`": "'", + "\u00b4": "'", + } +) +_PUNCTUATION = str.maketrans( + { + "-": " ", + "\u2010": " ", + "\u2011": " ", + "\u2013": " ", + "\u2014": " ", + ".": " ", + ",": " ", + ";": " ", + ":": " ", + "!": " ", + "?": " ", + '"': " ", + "(": " ", + ")": " ", + "[": " ", + "]": " ", + "{": " ", + "}": " ", + "/": " ", + "\\": " ", + } +) + + +def exact_key(text: str) -> str: + """Casefold and collapse separators. Punctuation stays in the key.""" + folded = text.casefold().replace("_", " ") + return " ".join(folded.split()) + + +def normalized_key(text: str) -> str: + """Exact key plus apostrophe folding and punctuation removal. + + Apostrophes are kept. Possessive pronouns are not rewritten, and no + stem or inflection is restored. + """ + folded = exact_key(text).translate(_APOSTROPHES).translate(_PUNCTUATION) + return " ".join(folded.split()) + + +def _confidence_key(value: object) -> str: + if isinstance(value, bool) or not isinstance(value, (int, float, Decimal)): + raise ValueError("confidence is not numeric") + if isinstance(value, Decimal): + return format(value, "f") + if isinstance(value, int): + return str(value) + return format(Decimal(str(value)), "f") + + +def _sense_token(instance: dict) -> str | None: + found = [ + field + for field in sorted(_SENSE_FIELDS) + if field in instance and instance[field] not in (None, "") + ] + if not found: + return None + if len(found) == 1: + return str(instance[found[0]]) + return "|".join(f"{field}={instance[field]}" for field in found) + + +def _label_bucket(label: object) -> str: + if label == "i": + return "idiomatic" + if label == "l": + return "literal" + return "unresolved" + + +class TypeUsage: + def __init__(self, idiom: str) -> None: + self.idiom = idiom + self.literal_instance_count = 0 + self.idiomatic_instance_count = 0 + self.unresolved_instance_count = 0 + self.source_instance_ids: list[int] = [] + self.variant_types: Counter[str] = Counter() + self.confidence_histogram: Counter[str] = Counter() + self.judgment_counts: list[int] = [] + self.sense_ids: list[str] = [] + self.instances_with_sense_id = 0 + self.instance_count = 0 + + def add(self, instance: dict) -> None: + identifier = instance["id"] + if isinstance(identifier, bool) or not isinstance(identifier, int): + raise ValueError("instance id is not an int") + bucket = _label_bucket(instance["label"]) + if bucket == "idiomatic": + self.idiomatic_instance_count += 1 + elif bucket == "literal": + self.literal_instance_count += 1 + else: + self.unresolved_instance_count += 1 + self.source_instance_ids.append(identifier) + variant = instance["variant_type"] + if not isinstance(variant, str) or not variant: + raise ValueError("variant_type is missing") + self.variant_types[variant] += 1 + self.confidence_histogram[_confidence_key(instance["confidence"])] += 1 + if "judgment_count" in instance and instance["judgment_count"] is not None: + judgment = instance["judgment_count"] + if isinstance(judgment, bool) or not isinstance(judgment, int): + raise ValueError("judgment_count is not an int") + self.judgment_counts.append(judgment) + token = _sense_token(instance) + self.instance_count += 1 + if token is not None: + self.instances_with_sense_id += 1 + self.sense_ids.append(token) + + def finish(self) -> None: + self.source_instance_ids.sort() + if len(self.source_instance_ids) != len(set(self.source_instance_ids)): + raise ValueError("duplicate instance id inside one expression") + + +class CorpusIndex: + def __init__(self) -> None: + self.types: dict[str, TypeUsage] = {} + self.by_exact: dict[str, list[str]] = {} + self.by_normalized: dict[str, list[str]] = {} + self.field_names: set[str] = set() + self.instance_count = 0 + self.records_bound_sense = False + + @property + def type_count(self) -> int: + return len(self.types) + + +def build_index(instances: Iterable[dict]) -> CorpusIndex: + index = CorpusIndex() + seen_ids: set[int] = set() + for instance in instances: + index.field_names.update(instance.keys()) + idiom = instance["idiom"] + if not isinstance(idiom, str) or not idiom.strip(): + raise ValueError("idiom string is empty") + identifier = instance["id"] + if identifier in seen_ids: + raise ValueError("duplicate instance id") + seen_ids.add(identifier) + usage = index.types.get(idiom) + if usage is None: + usage = TypeUsage(idiom) + index.types[idiom] = usage + usage.add(instance) + index.instance_count += 1 + if index.field_names & _SENSE_FIELDS: + index.records_bound_sense = True + exact_groups: dict[str, list[str]] = {} + normal_groups: dict[str, list[str]] = {} + for idiom, usage in index.types.items(): + usage.finish() + exact_groups.setdefault(exact_key(idiom), []).append(idiom) + key = normalized_key(idiom) + if key: + normal_groups.setdefault(key, []).append(idiom) + for key, idioms in exact_groups.items(): + index.by_exact[key] = sorted(idioms) + for key, idioms in normal_groups.items(): + index.by_normalized[key] = sorted(idioms) + return index + + +def _confidence_summary(usage: TypeUsage | None) -> dict | None: + if usage is None: + return None + summary = { + "histogram": dict(sorted(usage.confidence_histogram.items())), + "instance_count": usage.instance_count, + } + if usage.judgment_counts: + summary["judgment_count_maximum"] = max(usage.judgment_counts) + summary["judgment_count_minimum"] = min(usage.judgment_counts) + return summary + + +def match_surface(surface: str, index: CorpusIndex) -> dict: + """Return one surface relation. Variant metadata is not a second idiom.""" + exact_hits = index.by_exact.get(exact_key(surface), []) + if len(exact_hits) == 1: + relation = EXACT + idiom = exact_hits[0] + candidates: list[str] = [] + elif len(exact_hits) > 1: + relation = AMBIGUOUS + idiom = None + candidates = list(exact_hits) + else: + normal_hits = index.by_normalized.get(normalized_key(surface), []) + if len(normal_hits) == 1: + relation = NORMALIZED + idiom = normal_hits[0] + candidates = [] + elif len(normal_hits) > 1: + relation = AMBIGUOUS + idiom = None + candidates = list(normal_hits) + else: + relation = NONE + idiom = None + candidates = [] + usage = index.types[idiom] if idiom is not None else None + attributed = usage is not None + return { + "ambiguous_candidates": candidates, + "annotation_confidence": _confidence_summary(usage), + "attributed": attributed, + "idiomatic_instance_count": usage.idiomatic_instance_count if usage else None, + "literal_instance_count": usage.literal_instance_count if usage else None, + "matched_magpie_expression": idiom, + "source_instance_ids": list(usage.source_instance_ids) if usage else [], + "surface_match": relation, + "unresolved_instance_count": usage.unresolved_instance_count if usage else None, + "usage": usage, + "variant_types": dict(sorted(usage.variant_types.items())) if usage else {}, + } + + +def align_sense(match: dict, supplied_synset: str, index: CorpusIndex) -> tuple[str, str]: + """Align only by equality of a source sense identifier to the synset. + + Mixed literal and idiomatic counts do not by themselves create MIXED. + The supplied gloss is not an argument. + """ + relation = match["surface_match"] + if relation == NONE: + return UNKNOWN, "no_surface_match" + if relation == AMBIGUOUS: + return UNKNOWN, "ambiguous_surface" + if relation == VARIANT: + return UNKNOWN, "variant_without_sense_identifier" + usage: TypeUsage = match["usage"] + if not index.records_bound_sense or usage.instances_with_sense_id == 0: + return UNKNOWN, "no_bound_sense_identifier" + if usage.instances_with_sense_id != usage.instance_count: + return UNKNOWN, "partial_sense_identifier" + identifiers = set(usage.sense_ids) + if supplied_synset in identifiers and len(identifiers) > 1: + return CONFLICT, "sense_identifier_conflict" + if identifiers != {supplied_synset}: + return UNKNOWN, "sense_identifier_mismatch" + literal = usage.literal_instance_count + idiomatic = usage.idiomatic_instance_count + unresolved = usage.unresolved_instance_count + if idiomatic > 0 and literal == 0 and unresolved == 0: + return ALIGNED_IDIOMATIC, "aligned_idiomatic" + if literal > 0 and idiomatic == 0 and unresolved == 0: + return ALIGNED_LITERAL, "aligned_literal" + if idiomatic > 0 and literal > 0: + return MIXED, "mixed_usage_on_aligned_sense" + return UNKNOWN, "unresolved_labels_on_aligned_sense" + + +def evidence_code(surface_match: str, alignment: str) -> str: + if alignment == ALIGNED_IDIOMATIC: + return CODE_ALIGNED_IDIOMATIC + if alignment == ALIGNED_LITERAL: + return CODE_ALIGNED_LITERAL + if alignment == MIXED: + return CODE_MIXED + if alignment == CONFLICT: + return CODE_CONFLICT + if surface_match == NONE: + return CODE_NONE + return CODE_SURFACE + + +def semantic_noncompositional(alignment: str) -> str: + if alignment == ALIGNED_IDIOMATIC: + return YES + if alignment == ALIGNED_LITERAL: + return NO + return UNKNOWN + + +def evaluate_row( + surface: str, + synset: str, + index: CorpusIndex, + source_version: str, + source_artifact_hash: str, +) -> dict: + """One development row. Operator labels and glosses are not inputs.""" + match = match_surface(surface, index) + alignment, basis = align_sense(match, synset, index) + code = evidence_code(match["surface_match"], alignment) + semantic = semantic_noncompositional(alignment) + if match["surface_match"] != NONE and semantic == YES and alignment != ALIGNED_IDIOMATIC: + raise RuntimeError("surface match assigned YES without aligned idiomatic evidence") + return { + "alignment_basis": basis, + "ambiguous_candidates": match["ambiguous_candidates"], + "annotation_confidence": match["annotation_confidence"], + "idiomatic_instance_count": match["idiomatic_instance_count"], + "literal_instance_count": match["literal_instance_count"], + "matched_magpie_expression": match["matched_magpie_expression"], + "primary_evidence_code": code, + "semantic_noncompositional": semantic, + "sense_alignment": alignment, + "source_artifact_hash": source_artifact_hash, + "source_instance_ids": match["source_instance_ids"], + "source_name": "MAGPIE", + "source_version": source_version, + "surface_match": match["surface_match"], + "unresolved_instance_count": match["unresolved_instance_count"], + "variant_types": match["variant_types"], + } diff --git a/scripts/shadow/hyperlexical/unbind_screen_v3.py b/scripts/shadow/hyperlexical/unbind_screen_v3.py new file mode 100644 index 00000000..3c2ab94d --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v3.py @@ -0,0 +1,204 @@ +"""Frozen RUNE.UNBIND_SCREEN.v3. + +The bucket function is the scored screen. v4 may call it to obtain a bucket. +v4 must not rewrite this decision procedure. +""" + +from __future__ import annotations + +import re +from functools import lru_cache +from pathlib import Path + +FILES = { + "noun": ("index.noun", "data.noun"), + "verb": ("index.verb", "data.verb"), + "adj": ("index.adj", "data.adj"), + "adv": ("index.adv", "data.adv"), +} +STOP = { + "a", "an", "the", "of", "to", "in", "on", "for", "and", "or", "with", "from", + "by", "at", "as", "into", "over", "that", "this", "it", "be", "is", "are", + "was", "were", "been", "being", "not", "no", "than", "then", "if", "but", + "its", "who", "which", "when", "where", "what", "how", "about", "such", + "all", "every", "other", "one", +} +GEO_HEADS = frozenset({ + "gulf", "bay", "cape", "lake", "sea", "mount", "strait", "ocean", "island", + "river", "port", "peninsula", "isthmus", "archipelago", "republic", "kingdom", +}) +INST_HEADS = frozenset({ + "department", "ministry", "bureau", "agency", "university", "committee", + "commission", "court", "office", "board", "council", "senate", "congress", + "parliament", "institute", "administration", "authority", +}) +PLACES = frozenset({ + "alabama", "alaska", "arizona", "arkansas", "california", "colorado", + "connecticut", "delaware", "florida", "georgia", "hawaii", "idaho", + "illinois", "indiana", "iowa", "kansas", "kentucky", "louisiana", "maine", + "maryland", "massachusetts", "michigan", "minnesota", "mississippi", + "missouri", "montana", "nebraska", "nevada", "hampshire", "jersey", + "mexico", "york", "carolina", "dakota", "ohio", "oklahoma", "oregon", + "pennsylvania", "rhode", "tennessee", "texas", "utah", "vermont", + "virginia", "washington", "wisconsin", "wyoming", "america", "canada", + "brazil", "argentina", "chile", "peru", "france", "germany", "italy", + "spain", "portugal", "ireland", "scotland", "england", "britain", "europe", + "africa", "asia", "australia", "india", "china", "japan", "korea", "egypt", + "greece", "russia", "poland", +}) +PARTICLES = frozenset({ + "out", "up", "off", "away", "down", "back", "over", "through", "apart", + "aside", "along", "in", "on", +}) +POSSESSORS = frozenset({"my", "your", "his", "her", "our", "their", "one's"}) +DEGREE = frozenset({"too", "very", "more", "most", "so", "quite", "rather", "extremely"}) +NUMBERS = frozenset({ + "zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", + "ten", "eleven", "twelve", "thirteen", "fourteen", "fifteen", "sixteen", + "seventeen", "eighteen", "nineteen", "twenty", "thirty", "forty", "fifty", + "sixty", "seventy", "eighty", "ninety", "hundred", "thousand", "million", "billion", +}) +TITLE_NOUNS = frozenset({ + "law", "laws", "theory", "rules", "principle", "theorem", "equation", + "constant", "effect", "doctrine", +}) +NAME_PARTICLES = frozenset({"de", "von", "van", "di", "da", "del", "du", "des"}) +TITLE_HEADS = frozenset({"master", "bachelor", "doctor"}) +PROCEDURE_TAILS = frozenset({"surgery", "procedure", "operation"}) +TAXON_SUFFIX = ("idae", "aceae", "inae", "iformes", "oidea") +LATIN_SPECIES = re.compile(r"(ensis|oides|aceae|aris)$") +INITIALISM = re.compile(r"[a-z]\.$") +POSSESSIVE = re.compile(r"[a-z]+'s$") +NAME_TOKEN = re.compile(r"[a-z]+$") +YEAR = re.compile(r"\b(?:1[0-9]{3}|20[0-9]{2})\b") + +def stems(text): + out = [] + for word in re.findall(r"[a-z']+", text.lower()): + word = word.strip("'") + if len(word) < 3 or word in STOP: + continue + out.append(word[:4]) + return out + +def load_index(path): + found = {} + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + parts = line.split() + if "_" not in parts[0]: + continue + synset_cnt = int(parts[2]) + p_cnt = int(parts[3]) + rest = parts[4 + p_cnt:] + offsets = rest[2:2 + synset_cnt] + if offsets: + found[parts[0]] = offsets[0] + return found + +def load_glosses(path, wanted): + glosses = {} + want = set(wanted) + if not want: + return glosses + for line in path.open(encoding="utf-8", errors="replace"): + if not line or line[0] == " ": + continue + offset = line.split(" ", 1)[0] + if offset not in want: + continue + raw = line.split("|", 1)[1].strip() + glosses[offset] = raw.split(";", 1)[0].strip() + if len(glosses) == len(want): + break + return glosses + +@lru_cache(maxsize=4) +def load_indexes(root: str): + base = Path(root) + return {pos: load_index(base / pair[0]) for pos, pair in FILES.items()} + + +def gloss_for(surface, pos, root): + """First-sense gloss clause. The gloss is screening evidence, not a target.""" + indexes = load_indexes(str(root)) + lemma = surface.replace(" ", "_") + table = indexes.get(pos) or {} + offset = table.get(lemma) + if not offset: + for alt, idx in indexes.items(): + if lemma in idx: + pos, offset = alt, idx[lemma] + break + if not offset: + return pos, "" + data = load_glosses(Path(root) / FILES[pos][1], {offset}) + return pos, data.get(offset, "") + +def screen(surface, source_pos, tokens, gloss): + gloss_l = gloss.lower() + if not (2 <= len(tokens) <= 6) or len(surface) > 80 or " ".join(tokens) != surface: + return "REJECT", "surface_constraint", "exclude" + if any(INITIALISM.fullmatch(t) for t in tokens) or YEAR.search(gloss): + return "REJECT", "proper_person_name", "exclude" + if "organization" in gloss_l: + return "REJECT", "named_organization", "exclude" + if len(tokens) >= 2 and tokens[1] == "of" and tokens[0] in GEO_HEADS: + return "REJECT", "primarily_referential_expression", "exclude" + if len(tokens) >= 2 and tokens[1] == "of" and tokens[0] in INST_HEADS: + return "REJECT", "institutional_title", "exclude" + if "saint" in tokens or "st." in tokens: + return "REJECT", "titled_work_or_designation", "exclude" + if any(POSSESSIVE.fullmatch(t) for t in tokens) and any(t in TITLE_NOUNS for t in tokens): + return "REJECT", "titled_work_or_designation", "exclude" + if tokens and tokens[0] in TITLE_HEADS and tokens[1:2] == ["of"]: + return "REJECT", "degree_or_credential_name", "exclude" + if ( + len(tokens) >= 5 + and any(t in NAME_PARTICLES for t in tokens) + and all(NAME_TOKEN.fullmatch(t) for t in tokens) + ): + return "REJECT", "proper_person_name", "exclude" + if any(t.endswith(TAXON_SUFFIX) for t in tokens) or ( + tokens and tokens[0] in {"family", "genus", "species", "order", "phylum", "tribe"} + ) or "family of" in gloss_l: + return "REJECT", "taxonomy_or_species_label", "exclude" + if ( + source_pos == "noun" + and len(tokens) == 2 + and all(re.fullmatch(r"[a-z]{4,}", t) for t in tokens) + and LATIN_SPECIES.search(tokens[1]) + ): + return "REJECT", "taxonomy_or_species_label", "exclude" + if any(t in PLACES for t in tokens) and len(tokens) <= 3: + return "REJECT", "primarily_referential_expression", "exclude" + if tokens and all(t in NUMBERS for t in tokens): + return "REJECT", "productive_numeric_expression", "exclude" + if "per" in tokens or "unit for measuring" in gloss_l: + return "REJECT", "technical_measurement_expression", "exclude" + if len(tokens) >= 3 and tokens[-1] in PROCEDURE_TAILS: + return "REJECT", "named_technical_procedure", "exclude" + if source_pos == "adj" and len(tokens) == 2 and tokens[0] in DEGREE: + return "REJECT", "unconstrained_free_composition", "exclude" + if len(tokens) == 2 and tokens[0] in {"much", "more", "less"} and tokens[1] == "as": + return "REJECT", "unconstrained_free_composition", "exclude" + gstem = set(stems(gloss)) + sstem = stems(surface) + overlap = 0 if not sstem else len([w for w in sstem if w in gstem]) / len(sstem) + if source_pos == "verb" and len(tokens) == 2 and tokens[1] in PARTICLES and tokens[0][:4] not in gstem: + return "HIGH_VALUE", "strong_phrasal_binding", "score" + for i, tok in enumerate(tokens): + if tok in POSSESSORS: + rest = [w for w in tokens[i + 1:] if len(w) >= 3 and w not in STOP] + if rest and rest[-1][:4] not in gstem: + return "HIGH_VALUE", "lexicalized_variable_slot", "score" + if len(tokens) == 3 and tokens[1] == "and": + sides = [t[:4] for t in (tokens[0], tokens[2]) if len(t) >= 3] + if sides and all(s not in gstem for s in sides): + return "HIGH_VALUE", "fixed_nonliteral_expression", "score" + if len(tokens) == 4 and tokens[1] == "as" and tokens[2] == "a" and tokens[3][:4] not in gstem: + return "HIGH_VALUE", "fixed_nonliteral_expression", "score" + if source_pos in {"verb", "adj", "adv"} and len(tokens) >= 4 and overlap == 0: + return "HIGH_VALUE", "conventionalized_semantic_shift", "score" + return "SECONDARY", "transparent_or_moderate", "score" diff --git a/scripts/shadow/hyperlexical/unbind_screen_v5.py b/scripts/shadow/hyperlexical/unbind_screen_v5.py new file mode 100644 index 00000000..06790b04 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v5.py @@ -0,0 +1,427 @@ +"""Secondary-only wrapper over a frozen v4 bucket. + +HIGH and REJECT pass through uninspected. A SECONDARY row is inspected once. +Referential or terminological dominance moves it to REJECT. A conventionalized +surface whose gloss is not recoverable from the ordinary first senses of its +constituents moves it to HIGH. A lexicalized but compositional surface stays +SECONDARY. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path + +from hyperlexical.unbind_screen_v3 import STOP +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v5" +REJECT_EVIDENCE = "referential_terminological_dominance" +HIGH_EVIDENCE = "lexicalized_noncompositional" +_FUNCTION = set(STOP) | {"one's", "jr", "jr's"} +_META = frozenset({"used", "introducing"}) +_LIFE = frozenset({"noun.plant", "noun.animal"}) +_WORD = re.compile(r"[A-Za-z']+") +_TOKEN = re.compile(r"[a-z0-9']+") +_POS = ("noun", "verb", "adj", "adv") + + +@dataclass(frozen=True) +class Sense: + token: str + pos: str + lex: str + gloss: str + + +@dataclass(frozen=True) +class Entry: + pos: str + lemma: str + lex: str + lemmas: tuple[str, ...] + hypernyms: tuple[tuple[str, ...], ...] + + +class EmptyLexicon: + def entry(self, surface: str) -> Entry | None: + return None + + def senses(self, token: str) -> tuple[Sense, ...]: + return () + + +class WordNetLexicon: + """First-sense synsets. Gloss evidence, not a phrase list.""" + + def __init__(self, root: str | Path): + base = Path(root) + lexnames = _lexnames(base / "lexnames") + self._data = {pos: _load_data(base / f"data.{pos}") for pos in _POS} + self._index = {pos: _load_index(base / f"index.{pos}") for pos in _POS} + self._lexnames = lexnames + self._entries: dict[str, tuple[str, str, str]] = {} + for pos in _POS: + for key, offsets in self._index[pos].items(): + if not offsets: + continue + folded = key.casefold().replace("-", "_") + self._entries.setdefault(folded, (pos, key, offsets[0])) + + def entry(self, surface: str) -> Entry | None: + folded = surface.casefold().replace(" ", "_").replace("-", "_") + found = self._entries.get(folded) + if found is None: + return None + pos, key, offset = found + built = self._entry(pos, offset) + return Entry(built.pos, key, built.lex, built.lemmas, built.hypernyms) + + def senses(self, token: str) -> tuple[Sense, ...]: + found = [] + for pos in _POS: + offsets = self._index[pos].get(token) + if not offsets: + continue + entry = self._entry(pos, offsets[0]) + gloss = self._gloss(pos, offsets[0]) + found.append(Sense(token, pos, entry.lex, gloss)) + return tuple(found) + + def _entry(self, pos: str, offset: str) -> Entry: + lemmas, lex, hypers = _parse(self._data[pos][offset], self._lexnames, self._data["noun"]) + return Entry(pos, lemmas[0] if lemmas else "", lex, tuple(lemmas), tuple(tuple(item) for item in hypers)) + + def _gloss(self, pos: str, offset: str) -> str: + line = self._data[pos][offset] + return line.split("|", 1)[1].split(";", 1)[0].strip() + + +def apply_v5(v4_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v5 bucket. HIGH and REJECT are not inspected.""" + del source_pos + bucket = canonical_bucket(v4_bucket) + normalized = normalize_lexical(surface) + if bucket != "SECONDARY": + return _decision(bucket, bucket, None, normalized, inspected=False) + evidence = _inspect(surface or "", gloss or "", lexicon) + if evidence == REJECT_EVIDENCE: + return _decision(bucket, "REJECT", evidence, normalized, inspected=True) + if evidence == HIGH_EVIDENCE: + return _decision(bucket, "HIGH", evidence, normalized, inspected=True) + return _decision(bucket, "SECONDARY", None, normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 141) -> dict: + """Replay gate. This is not a precision score.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = {"secondary_to_high": 0, "secondary_to_reject": 0, "secondary_unchanged": 0} + held = {"high": 0, "reject": 0} + conflicts = 0 + for row in rows: + prior = canonical_bucket(str(row["v4_bucket"])) + nxt = canonical_bucket(str(row["v5_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == "HIGH": + held["high"] += 1 + if nxt != "HIGH": + failures.append("HIGH row changed bucket") + if operator == "HIGH" and nxt != "HIGH": + failures.append("previously correct HIGH is no longer correct") + elif prior == "REJECT": + held["reject"] += 1 + if nxt != "REJECT": + failures.append("REJECT row changed bucket") + if operator == "REJECT" and nxt != "REJECT": + failures.append("previously correct REJECT is no longer correct") + elif prior == "SECONDARY": + if nxt == "SECONDARY": + moves["secondary_unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged SECONDARY carries transition evidence") + elif nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + else: + failures.append("SECONDARY moved outside HIGH and REJECT") + else: + failures.append("prior bucket is outside HIGH, REJECT, and SECONDARY") + if prior != nxt and operator != nxt: + conflicts += 1 + failures.append("operator conflict on a move") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + verified = not failures + return { + "schema": "hyperlex.unbind_screen_v5_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "high_unchanged": held["high"], + "reject_unchanged": held["reject"], + "moves": moves, + "operator_conflict_on_move": conflicts, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + and report.get("operator_conflict_on_move") == 0 + ) + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v4_bucket": prior, + "v5_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _inspect(surface: str, gloss: str, lexicon) -> str | None: + found = lexicon.entry(surface) + if found is None: + return None + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + ordinary, missing, senses = _ordinary(content, lexicon) + if _designates(found, stems, content): + return REJECT_EVIDENCE + if _noncompositional(found, gloss, content, stems, ordinary, missing, senses): + return HIGH_EVIDENCE + return None + + +def _designates(found: Entry, stems: set[str], content: list[str]) -> bool: + if any(char.isupper() for char in found.lemma): + return True + if found.lex == "adj.pert": + return True + if found.lex in _LIFE and _exocentric_life(found.hypernyms, stems): + return True + if _exocentric_category(found.hypernyms, stems) and not _ordinary_synonym(found.lemmas, content): + return True + return False + + +def _noncompositional(found: Entry, gloss, content, stems, ordinary, missing, senses) -> bool: + if _blocked(gloss, ordinary, missing, content): + return False + if _orthographic(content) or _body_clash(found, gloss, content, senses): + return True + if _verb_shift(found, content, senses): + return True + return ( + found.pos == "adv" + and _unrelated_paraphrase(found.lemmas, content, stems) + and not _morphological(found.lemmas, content, stems) + ) + + +def _blocked(gloss: str, ordinary: set[str], missing: list[str], content: list[str]) -> bool: + words = _WORD.findall(gloss or "") + if words and words[0].casefold() in _META: + return True + if missing or len(content) != len(set(content)): + return True + return bool(_stems(gloss) & ordinary) + + +def _ordinary(content: list[str], lexicon): + ordinary: set[str] = set() + missing: list[str] = [] + senses: list[Sense] = [] + for token in content: + if len(token) < 3: + continue + found = tuple(lexicon.senses(token)) + if not found: + missing.append(token) + continue + for sense in found: + ordinary |= _stems(sense.gloss) + stemmed = _stem(token) + if stemmed: + ordinary.add(stemmed) + senses.append(sense) + return ordinary, missing, senses + + +def _exocentric_life(hypernyms, stems: set[str]) -> bool: + if not hypernyms: + return False + for lemmas in hypernyms: + for lemma in lemmas: + if any(_stem(part) in stems for part in re.split(r"[_-]", lemma)): + return False + return True + + +def _exocentric_category(hypernyms, stems: set[str]) -> bool: + for lemmas in hypernyms: + for lemma in lemmas: + if "_" not in lemma: + continue + parts = [part for part in (_stem(item) for item in re.split(r"[_-]", lemma)) if part] + if parts and not any(part in stems for part in parts): + return True + return False + + +def _ordinary_synonym(lemmas, content: list[str]) -> bool: + owned = set(content) + for lemma in _single_words(lemmas): + if _initialism(lemma): + continue + low = re.sub(r"[^a-z]", "", lemma.casefold()) + if low and low not in owned: + return True + return False + + +def _body_clash(found: Entry, gloss: str, content: list[str], senses: list[Sense]) -> bool: + if found.lex == "noun.body": + return False + if not any(sense.lex == "noun.body" for sense in senses): + return False + gloss_l = (gloss or "").casefold() + return not any(token in gloss_l for token in content) + + +def _verb_shift(found: Entry, content: list[str], senses: list[Sense]) -> bool: + if found.pos != "verb" or not content or found.lex.endswith(".all"): + return False + head = content[0] + head_lex = next((sense.lex for sense in senses if sense.token == head and sense.pos == "verb"), None) + if not head_lex or head_lex.endswith(".all") or head_lex == found.lex: + return False + return True + + +def _unrelated_paraphrase(lemmas, content, stems) -> bool: + return any(not _related(lemma, content, stems) for lemma in _single_words(lemmas)) + + +def _morphological(lemmas, content, stems) -> bool: + return any(_related(lemma, content, stems) for lemma in _single_words(lemmas)) + + +def _related(lemma: str, content: list[str], stems: set[str]) -> bool: + low = lemma.casefold() + stemmed = _stem(low) + if stemmed and stemmed in stems: + return True + return any(len(token) >= 4 and (token in low or low in token) for token in content) + + +def _orthographic(content: list[str]) -> bool: + return any(len(token) == 1 and token.isalpha() for token in content) + + +def _single_words(lemmas) -> list[str]: + return [lemma for lemma in lemmas if "_" not in lemma and "-" not in lemma and "(" not in lemma] + + +def _initialism(lemma: str) -> bool: + letters = re.sub(r"[^A-Za-z]", "", lemma) + return bool(letters) and letters.isupper() + + +def _content(surface: str) -> list[str]: + text = surface.casefold().replace("-", " ") + return [token for token in _TOKEN.findall(text) if token not in _FUNCTION] + + +def _stem(word: str) -> str: + token = word.casefold().strip("'") + if len(token) < 3 or token in _FUNCTION: + return "" + return token[:4] + + +def _stems(text: str) -> set[str]: + return {item for item in (_stem(word) for word in _WORD.findall(text or "")) if item} + + +def _lexnames(path: Path) -> dict[int, str]: + names = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line.strip(): + continue + number, name, *_rest = line.split() + names[int(number)] = name + return names + + +def _load_index(path: Path) -> dict[str, list[str]]: + found = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line or line[0] == " ": + continue + parts = line.split() + synset_cnt = int(parts[2]) + pointer_cnt = int(parts[3]) + rest = parts[4 + pointer_cnt:] + found[parts[0]] = rest[2:2 + synset_cnt] + return found + + +def _load_data(path: Path) -> dict[str, str]: + found = {} + for line in path.read_text(encoding="utf-8", errors="replace").splitlines(): + if not line or line[0] == " ": + continue + found[line.split(" ", 1)[0]] = line + return found + + +def _parse(line: str, lexnames: dict[int, str], noun_data: dict[str, str]): + parts = line.split() + lex = lexnames[int(parts[1])] + count = int(parts[3], 16) + lemmas = [] + index = 4 + for _ in range(count): + lemmas.append(parts[index]) + index += 2 + pointer_cnt = int(parts[index]) + index += 1 + hypers = [] + for _ in range(pointer_cnt): + symbol, target, target_pos = parts[index], parts[index + 1], parts[index + 2] + index += 4 + if symbol in {"@", "@i"} and target_pos == "n" and target in noun_data: + hyper, _lex, _nested = _parse(noun_data[target], lexnames, noun_data) + hypers.append(hyper) + return lemmas, lex, hypers diff --git a/scripts/shadow/hyperlexical/unbind_screen_v6.py b/scripts/shadow/hyperlexical/unbind_screen_v6.py new file mode 100644 index 00000000..5d908e91 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v6.py @@ -0,0 +1,262 @@ +"""Challengeable outer buckets over a frozen v5 decision. + +A frozen v5 bucket is provisional. High may fall to secondary when the gloss +is recoverable from ordinary constituent senses and ordinary syntax. Reject +may fall to secondary when the surface is a lexical state or relation rather +than a designation. A demotion stops for that application. A provisional +secondary row may still move by the two v5 evidences. High and reject do not +swap. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical +from hyperlexical.unbind_screen_v5 import ( + HIGH_EVIDENCE, + REJECT_EVIDENCE, + Entry, + _body_clash, + _content, + _exocentric_category, + _exocentric_life, + _initialism, + _ordinary, + _ordinary_synonym, + _orthographic, + _related, + _single_words, + _stem, + _stems, + _verb_shift, + _LIFE, +) + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v6" +COMPOSITIONAL_EVIDENCE = "compositional_recoverability" +NONREFERENTIAL_EVIDENCE = "nonreferential_lexical_use" +_RELATIONAL = re.compile(r"^(of or relating to|relating to|related to)\b", re.IGNORECASE) + + +def apply_v6(v5_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v6 bucket. A demoted row is not reconsidered.""" + del source_pos + bucket = canonical_bucket(v5_bucket) + normalized = normalize_lexical(surface) + if bucket == "HIGH": + if _compositional(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", COMPOSITIONAL_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "HIGH", None, normalized, inspected=True) + if bucket == "REJECT": + if _nonreferential(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", NONREFERENTIAL_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "REJECT", None, normalized, inspected=True) + if bucket != "SECONDARY": + raise ValueError(f"provisional bucket {bucket} is outside HIGH, REJECT, and SECONDARY") + evidence = _secondary_evidence(surface or "", gloss or "", lexicon) + if evidence == REJECT_EVIDENCE: + return _decision(bucket, "REJECT", evidence, normalized, inspected=True) + if evidence == HIGH_EVIDENCE: + return _decision(bucket, "HIGH", evidence, normalized, inspected=True) + return _decision(bucket, "SECONDARY", None, normalized, inspected=True) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 169) -> dict: + """Replay gate. Previously correct rows must stay correct. Buckets may move.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = { + "high_to_secondary": 0, + "reject_to_secondary": 0, + "secondary_to_high": 0, + "secondary_to_reject": 0, + "unchanged": 0, + } + previously_correct = 0 + previously_correct_lost = 0 + direct_swaps = 0 + for row in rows: + prior = canonical_bucket(str(row["v5_bucket"])) + nxt = canonical_bucket(str(row["v6_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == operator: + previously_correct += 1 + if nxt != operator: + previously_correct_lost += 1 + failures.append("previously correct row is no longer correct") + if prior == nxt: + moves["unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged row carries transition evidence") + elif prior == "HIGH" and nxt == "SECONDARY": + moves["high_to_secondary"] += 1 + if primary != COMPOSITIONAL_EVIDENCE or supporting: + failures.append("HIGH to SECONDARY lacks compositional_recoverability") + if nxt != operator: + failures.append("outer reversal does not land on the operator bucket") + elif prior == "REJECT" and nxt == "SECONDARY": + moves["reject_to_secondary"] += 1 + if primary != NONREFERENTIAL_EVIDENCE or supporting: + failures.append("REJECT to SECONDARY lacks nonreferential_lexical_use") + if nxt != operator: + failures.append("outer reversal does not land on the operator bucket") + elif prior == "SECONDARY" and nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif prior == "SECONDARY" and nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + elif {prior, nxt} == {"HIGH", "REJECT"}: + direct_swaps += 1 + failures.append("HIGH and REJECT swapped directly") + else: + failures.append("row moved outside the four legal transitions") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + assertions = { + "replay_count": len(rows) == expected_rows, + "previously_correct_remain_correct": previously_correct_lost == 0, + "high_to_secondary_evidence": "HIGH to SECONDARY lacks compositional_recoverability" not in failures, + "reject_to_secondary_evidence": "REJECT to SECONDARY lacks nonreferential_lexical_use" not in failures, + "secondary_to_high_evidence": "SECONDARY to HIGH lacks lexicalized_noncompositional" not in failures, + "secondary_to_reject_evidence": "SECONDARY to REJECT lacks referential_terminological_dominance" not in failures, + "no_direct_outer_swap": direct_swaps == 0, + "no_phrase_specific_rules": not phrase_specific_rule_fired, + } + verified = not failures and all(assertions.values()) + return { + "schema": "hyperlex.unbind_screen_v6_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "assertions": assertions, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "previously_correct": previously_correct, + "previously_correct_lost": previously_correct_lost, + "direct_swaps": direct_swaps, + "moves": moves, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": verified, + "select_authorized": False, + "revision_eligible": False, + } + + +def measurement_allowed(report: dict) -> bool: + return bool( + report.get("regression") == "REGRESSION_VERIFIED" + and report.get("phrase_specific_rule_fired") is False + and not report.get("failures") + and report.get("measurement_eligible") is True + and report.get("previously_correct_lost") == 0 + and report.get("direct_swaps") == 0 + and all((report.get("assertions") or {}).values()) + ) + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v5_bucket": prior, + "v6_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _secondary_evidence(surface: str, gloss: str, lexicon) -> str | None: + """The frozen v5 secondary tests. Not applied to a row demoted in this pass.""" + from hyperlexical.unbind_screen_v5 import _inspect + + return _inspect(surface, gloss, lexicon) + + +def _compositional(surface: str, gloss: str, lexicon) -> bool: + """Recoverable from ordinary senses and ordinary syntax, without an idiomatic mapping.""" + found = lexicon.entry(surface) + if found is None: + return False + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + if _idiomatic_mapping(found, content, stems): + return False + ordinary, missing, senses = _ordinary(content, lexicon) + if missing or _orthographic(content) or _body_clash(found, gloss, content, senses): + return False + if _verb_shift(found, content, senses): + return False + gloss_stems = _stems(gloss) + if not gloss_stems or not gloss_stems <= ordinary: + return False + return _contributors(content, gloss_stems, lexicon) >= 2 + + +def _nonreferential(surface: str, gloss: str, lexicon) -> bool: + """A pertainym used as a state or relation, not as a designation.""" + found = lexicon.entry(surface) + if found is None or found.lex != "adj.pert": + return False + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + if _hard_designation(found, stems, content): + return False + text = (gloss or "").strip() + if not text or _RELATIONAL.match(text): + return False + ordinary, missing, _senses = _ordinary(content, lexicon) + if missing: + return False + gloss_stems = _stems(text) + sense_stems = ordinary - stems + return bool(gloss_stems & sense_stems) + + +def _hard_designation(found: Entry, stems: set[str], content: list[str]) -> bool: + if any(char.isupper() for char in found.lemma): + return True + if found.lex in _LIFE and _exocentric_life(found.hypernyms, stems): + return True + if _exocentric_category(found.hypernyms, stems) and not _ordinary_synonym(found.lemmas, content): + return True + return False + + +def _idiomatic_mapping(found: Entry, content: list[str], stems: set[str]) -> bool: + for lemma in _single_words(found.lemmas): + if _initialism(lemma): + continue + if not _related(lemma, content, stems): + return True + return False + + +def _contributors(content: list[str], gloss_stems: set[str], lexicon) -> int: + count = 0 + for token in content: + if len(token) < 3: + continue + token_stems = set() + for sense in lexicon.senses(token): + token_stems |= _stems(sense.gloss) + stemmed = _stem(token) + if stemmed: + token_stems.add(stemmed) + if gloss_stems & token_stems: + count += 1 + return count diff --git a/scripts/shadow/hyperlexical/unbind_screen_v7.py b/scripts/shadow/hyperlexical/unbind_screen_v7.py new file mode 100644 index 00000000..4584ce7c --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_screen_v7.py @@ -0,0 +1,279 @@ +"""One high challenge over a frozen v6 bucket. + +A provisional high may fall to secondary when ordinary compositional derivation +fires. That demotion stops. Reject and secondary buckets stay at the v6 +decision: those transitions were already applied, and a stopped demotion is +not reopened. The v6 high-challenge evidence is not a v7 transition. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_screen_v4 import canonical_bucket, normalize_lexical +from hyperlexical.unbind_screen_v5 import ( + HIGH_EVIDENCE, + REJECT_EVIDENCE, + _FUNCTION, + _WORD, + _content, + _initialism, + _related, + _single_words, + _stem, +) +from hyperlexical.unbind_screen_v6 import NONREFERENTIAL_EVIDENCE, apply_v6 + +RULE_VERSION = "RUNE.UNBIND_SCREEN.v7" +ORDINARY_EVIDENCE = "ordinary_compositional_derivation" +_COMPARATIVE_GLOSS = re.compile(r"^used to form the comparative\b", re.IGNORECASE) +_DEGREE_SENSE = re.compile(r"^(?:of less|of more|of greater|of smaller)\b", re.IGNORECASE) +_METAPHOR = re.compile(r"\bmetaphors?\b|\bmetaphorical(?:ly)?\b", re.IGNORECASE) +_PARTICLE = frozenset({ + "out", "off", "up", "down", "away", "back", "over", "through", "along", +}) + + +def apply_v7(v6_bucket: str, surface: str, gloss: str, source_pos: str, lexicon) -> dict: + """Return the v7 bucket. Only a provisional high is eligible for the new predicate.""" + bucket = canonical_bucket(v6_bucket) + normalized = normalize_lexical(surface) + if bucket == "HIGH": + if ordinary_compositional_derivation(surface or "", gloss or "", lexicon): + return _decision(bucket, "SECONDARY", ORDINARY_EVIDENCE, normalized, inspected=True) + return _decision(bucket, "HIGH", None, normalized, inspected=True) + if bucket == "REJECT": + inherited = apply_v6(bucket, surface, gloss, source_pos, lexicon) + return _decision( + bucket, + inherited["v6_bucket"], + inherited["primary_evidence"], + normalized, + inspected=bool(inherited["inspected"]), + ) + if bucket != "SECONDARY": + raise ValueError(f"provisional bucket {bucket} is outside HIGH, REJECT, and SECONDARY") + # v6 already ran the secondary transitions. Running them again would reopen + # a demotion that v6 stopped, so the frozen secondary decision stands. + return _decision(bucket, "SECONDARY", None, normalized, inspected=False) + + +def ordinary_compositional_derivation(surface: str, gloss: str, lexicon) -> bool: + """True when the recorded sense is ordinary composition, not a stored binding. + + Stem overlap with a constituent gloss is not enough, and a metaphorical + retelling of the gloss is not enough. The lexical record has to present a + comparative, syntactic, or phrasal composition with no unrelated synonym. + """ + found = lexicon.entry(surface) + if found is None or _stored_binding(found, surface) or _METAPHOR.search(gloss or ""): + return False + return ( + _comparative(found, surface, gloss, lexicon) + or _syntactic(surface, gloss, lexicon) + or _phrasal(surface, gloss, lexicon) + ) + + +def assess(rows: list[dict], *, phrase_specific_rule_fired: bool, expected_rows: int = 197) -> dict: + """Regression gate. Previously correct rows must stay correct. Buckets may move.""" + failures: list[str] = [] + surfaces = [str(row["surface"]) for row in rows] + if len(rows) != expected_rows: + failures.append(f"replay rows {len(rows)} != {expected_rows}") + if len(set(surfaces)) != len(surfaces): + failures.append("replay surface is duplicated") + moves = { + "high_to_secondary": 0, + "reject_to_secondary": 0, + "secondary_to_high": 0, + "secondary_to_reject": 0, + "unchanged": 0, + } + previously_correct = 0 + previously_correct_lost = 0 + correct_high = 0 + correct_high_lost = 0 + direct_swaps = 0 + for row in rows: + prior = canonical_bucket(str(row["v6_bucket"])) + nxt = canonical_bucket(str(row["v7_bucket"])) + operator = canonical_bucket(str(row["operator_bucket"])) + primary = row.get("primary_evidence") + supporting = list(row.get("supporting_evidence") or []) + if prior == operator: + previously_correct += 1 + if nxt != operator: + previously_correct_lost += 1 + failures.append("previously correct row is no longer correct") + if prior == "HIGH" and operator == "HIGH": + correct_high += 1 + if nxt != "HIGH": + correct_high_lost += 1 + failures.append("previously correct HIGH was demoted") + if prior == nxt: + moves["unchanged"] += 1 + if primary is not None or supporting: + failures.append("unchanged row carries transition evidence") + elif prior == "HIGH" and nxt == "SECONDARY": + moves["high_to_secondary"] += 1 + if primary != ORDINARY_EVIDENCE or supporting: + failures.append("HIGH to SECONDARY lacks ordinary_compositional_derivation") + elif prior == "REJECT" and nxt == "SECONDARY": + moves["reject_to_secondary"] += 1 + if primary != NONREFERENTIAL_EVIDENCE or supporting: + failures.append("REJECT to SECONDARY lacks nonreferential_lexical_use") + elif prior == "SECONDARY" and nxt == "HIGH": + moves["secondary_to_high"] += 1 + if primary != HIGH_EVIDENCE or supporting: + failures.append("SECONDARY to HIGH lacks lexicalized_noncompositional") + elif prior == "SECONDARY" and nxt == "REJECT": + moves["secondary_to_reject"] += 1 + if primary != REJECT_EVIDENCE or supporting: + failures.append("SECONDARY to REJECT lacks referential_terminological_dominance") + elif {prior, nxt} == {"HIGH", "REJECT"}: + direct_swaps += 1 + failures.append("HIGH and REJECT swapped directly") + else: + failures.append("row moved outside the legal transitions") + if prior != nxt and primary is None: + failures.append("changed row has no primary evidence") + if phrase_specific_rule_fired: + failures.append("phrase-specific rule fired") + assertions = { + "A_replay_197": len(rows) == expected_rows and len(set(surfaces)) == len(surfaces), + "B_previously_correct_remain_correct": previously_correct_lost == 0, + "C_high_demotion_evidence": "HIGH to SECONDARY lacks ordinary_compositional_derivation" not in failures, + "D_correct_high_stays_high": correct_high_lost == 0, + "E_v6_transitions_unchanged": ( + "REJECT to SECONDARY lacks nonreferential_lexical_use" not in failures + and "SECONDARY to HIGH lacks lexicalized_noncompositional" not in failures + and "SECONDARY to REJECT lacks referential_terminological_dominance" not in failures + and moves["secondary_to_high"] == 0 + and moves["secondary_to_reject"] == 0 + and moves["reject_to_secondary"] == 0 + ), + "F_no_direct_outer_swap": direct_swaps == 0, + "G_no_phrase_rules": not phrase_specific_rule_fired, + "H_demotion_stops": moves["high_to_secondary"] == 0 or "HIGH to SECONDARY lacks ordinary_compositional_derivation" not in failures, + "I_measurement_not_drawn": True, + } + verified = not failures and all(assertions.values()) + return { + "schema": "hyperlex.unbind_screen_v7_gate_report.v1", + "rule": RULE_VERSION, + "regression": "REGRESSION_VERIFIED" if verified else "REGRESSION_FAILED", + "state": "REGRESSION_VERIFIED" if verified else "ENCODED", + "failures": failures, + "assertions": assertions, + "expected_rows": expected_rows, + "replay_rows": len(rows), + "unique_surfaces": len(set(surfaces)), + "previously_correct": previously_correct, + "previously_correct_lost": previously_correct_lost, + "correct_high": correct_high, + "correct_high_lost": correct_high_lost, + "direct_swaps": direct_swaps, + "moves": moves, + "phrase_specific_rule_fired": bool(phrase_specific_rule_fired), + "measurement_eligible": False, + "measurement_sample_drawn": False, + "regression_is_not_generalization": True, + "select_authorized": False, + "revision_eligible": False, + } + + +def _decision(prior, nxt, primary, normalized, *, inspected: bool) -> dict: + return { + "rule": RULE_VERSION, + "v6_bucket": prior, + "v7_bucket": nxt, + "primary_evidence": primary, + "supporting_evidence": [], + "normalized": normalized, + "inspected": inspected, + } + + +def _stored_binding(found, surface: str) -> bool: + """An unrelated single-word synonym is a stored conventionalized binding.""" + content = _content(surface) + stems = {item for item in (_stem(token) for token in content) if item} + for lemma in _single_words(found.lemmas): + if _initialism(lemma): + continue + if not _related(lemma, content, stems): + return True + return False + + +def _comparative(found, surface: str, gloss: str, lexicon) -> bool: + """A periphrastic comparative whose record is the degree construction itself.""" + if found.lex != "adv.all" or not _COMPARATIVE_GLOSS.match((gloss or "").strip()): + return False + return any(_degree_adjective(token, lexicon) for token in _content(surface)) + + +def _degree_adjective(token: str, lexicon) -> bool: + if len(token) < 5 or not token.endswith("er"): + return False + for sense in lexicon.senses(token): + if sense.pos != "adj" and sense.lex != "adj.all": + continue + if _DEGREE_SENSE.match((sense.gloss or "").strip()): + return True + return False + + +def _syntactic(surface: str, gloss: str, lexicon) -> bool: + """The gloss is the recorded senses of two open-class constituents, in order.""" + heads = [token for token in _content(surface) if token not in _PARTICLE] + if len(heads) < 2: + return False + return _gloss_is_sense_sum(heads, gloss, lexicon) + + +def _phrasal(surface: str, gloss: str, lexicon) -> bool: + """The gloss is a verb sense plus one particle sense, with nothing left over.""" + content = _content(surface) + particles = [token for token in content if token in _PARTICLE] + heads = [token for token in content if token not in _PARTICLE] + if len(particles) != 1 or len(heads) != 1: + return False + if not any(sense.pos == "verb" for sense in lexicon.senses(heads[0])): + return False + return _gloss_is_sense_sum([heads[0], particles[0]], gloss, lexicon) + + +def _gloss_is_sense_sum(tokens: list[str], gloss: str, lexicon) -> bool: + target = _open_words(gloss) + if len(target) < 2: + return False + choices: list[tuple[tuple[str, ...], ...]] = [] + for token in tokens: + glosses = tuple(_open_words(sense.gloss) for sense in lexicon.senses(token) if _open_words(sense.gloss)) + if not glosses: + return False + choices.append(glosses) + return _any_concatenation(choices, target) + + +def _any_concatenation(choices: list[tuple[tuple[str, ...], ...]], target: list[str]) -> bool: + def walk(index: int, built: list[str]) -> bool: + if index == len(choices): + return built == target and all(built) + for gloss in choices[index]: + if walk(index + 1, built + list(gloss)): + return True + return False + + return walk(0, []) + + +def _open_words(text: str) -> list[str]: + return [ + word.casefold() + for word in _WORD.findall(text or "") + if word.casefold() not in _FUNCTION and len(word) >= 3 + ] diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py new file mode 100644 index 00000000..5f7ce781 --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v1.py @@ -0,0 +1,267 @@ +"""Sense-first unbind screen. + +The class comes from the frozen WordNet record of the supplied synset. +Membership in WordNet is not a class. Absence of a signal is not secondary. +""" + +from __future__ import annotations + +import re +from pathlib import Path + +RULE_VERSION = "RUNE.UNBIND_SENSE_SCREEN.v1" +_LIFESPAN = re.compile(r"\([0-9]{4}-[0-9]{4}\)") +_COMPARATIVE = re.compile(r"^used to form the comparative\b") +_SUPERLATIVE = re.compile(r"^used to form the superlative\b") +_LEXICAL_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +_RELATION_SYMBOLS = frozenset({"+", "\\"}) +_PREDICATE_TYPES = frozenset({"r", "a", "s"}) +_FILES = { + "noun": "data.noun", + "verb": "data.verb", + "adj": "data.adj", + "adv": "data.adv", +} +_EXC = { + "noun": "noun.exc", + "verb": "verb.exc", + "adj": "adj.exc", + "adv": "adv.exc", +} +_CODES = { + "REFERENTIAL": "referential_designation", + "ORDINARY_COMPOSITIONAL": "productive_grammatical_frame", + "LEXICALIZED_NONCOMPOSITIONAL": "noncompositional_semantic_mapping", + "LEXICALIZED_COMPOSITIONAL": "compositional_lexical_unit", + "AMBIGUOUS": "insufficient_record_evidence", +} +_BUCKETS = { + "REFERENTIAL": "REJECT", + "LEXICALIZED_NONCOMPOSITIONAL": "HIGH", + "LEXICALIZED_COMPOSITIONAL": "SECONDARY", + "ORDINARY_COMPOSITIONAL": "SECONDARY", + "AMBIGUOUS": "QUARANTINE", +} + + +class Pointer: + def __init__(self, symbol: str, offset: str, pos: str, source: int, target: int): + self.symbol = symbol + self.offset = offset + self.pos = pos + self.source = source + self.target = target + + +class Synset: + def __init__(self, offset: str, ss_type: str, lemmas: tuple[str, ...], pointers: tuple[Pointer, ...]): + self.offset = offset + self.ss_type = ss_type + self.lemmas = lemmas + self.pointers = pointers + + +def parse_data_line(line: str) -> tuple[Synset, str] | None: + """Return the synset and the first gloss clause. + + The word count and the source/target word numbers are hexadecimal. + The pointer count is a decimal integer. Verb frames follow the pointers + and are not pointers. + """ + if not line or line[0] == " ": + return None + meta, bar, gloss = line.partition("|") + if not bar: + return None + tokens = meta.split() + offset = tokens[0] + ss_type = tokens[2] + word_count = int(tokens[3], 16) + index = 4 + lemmas = [] + for _ in range(word_count): + lemmas.append(tokens[index]) + index += 2 + pointer_count = int(tokens[index], 10) + index += 1 + pointers = [] + for _ in range(pointer_count): + symbol = tokens[index] + target = tokens[index + 1] + pos = tokens[index + 2] + link = tokens[index + 3] + pointers.append(Pointer(symbol, target, pos, int(link[:2], 16), int(link[2:], 16))) + index += 4 + first = gloss.strip().split(";", 1)[0].strip() + return Synset(offset, ss_type, tuple(lemmas), tuple(pointers)), first + + +def load_exceptions(root: str | Path) -> dict[str, set[str]]: + linked: dict[str, set[str]] = {} + base = Path(root) + for name in _EXC.values(): + for line in (base / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parts = line.split() + if len(parts) < 2: + continue + left = parts[0].casefold() + right = parts[1].casefold() + linked.setdefault(left, set()).add(right) + linked.setdefault(right, set()).add(left) + return linked + + +def load_wordnet(root: str | Path) -> tuple[dict[tuple[str, str], Synset], dict[tuple[str, str], str]]: + synsets: dict[tuple[str, str], Synset] = {} + glosses: dict[tuple[str, str], str] = {} + base = Path(root) + for pos, name in _FILES.items(): + for line in (base / name).read_text(encoding="utf-8", errors="replace").splitlines(): + parsed = parse_data_line(line) + if parsed is None: + continue + synset, gloss = parsed + for key in ((pos, synset.offset), (synset.ss_type, synset.offset)): + synsets[key] = synset + glosses[key] = gloss + return synsets, glosses + + +def classify(surface: str, gloss: str, synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> dict: + """Apply the frozen procedure once. The result is one class and one bucket.""" + text = gloss or "" + tokens = _tokens(surface) + ref_yes, ref_source = _referential_yes(text, synset) + ref_no = _referential_no(text, synset, ref_yes) + unrelated = _unrelated(tokens, synset, exceptions) + pointer = _lexical_pointer(tokens, synset) + alternation = _alternation(tokens, synset) + operator = _operator(text) + unit_yes = bool(unrelated) or pointer + unit_no = alternation or operator + constituent = _constituent_relation(tokens, synset, exceptions, targets) + # The constituent signal is defined only when no unrelated co-lemma is present, + # so those two signals do not fire together. + if (ref_yes and unit_no) or (unit_yes and unit_no): + return _emit("AMBIGUOUS", "none", insufficient=True) + _ = ref_no + if ref_yes: + return _emit("REFERENTIAL", ref_source, insufficient=False) + if unit_no: + source = "synset.lemmas.productive_alternation" if alternation else "synset.gloss.grammatical_operator" + return _emit("ORDINARY_COMPOSITIONAL", source, insufficient=False) + if not unit_yes: + return _emit("AMBIGUOUS", "none", insufficient=True) + if unrelated: + return _emit("LEXICALIZED_NONCOMPOSITIONAL", "synset.lemmas.unrelated_single_word", insufficient=False) + if constituent: + return _emit("LEXICALIZED_COMPOSITIONAL", "synset.lexical_pointer.derivation_or_pertainym_to_constituent", insufficient=False) + return _emit("AMBIGUOUS", "none", insufficient=True) + + +def _emit(sense_class: str, source: str, *, insufficient: bool) -> dict: + return { + "rule": RULE_VERSION, + "sense_class": sense_class, + "bucket": _BUCKETS[sense_class], + "primary_evidence_code": _CODES[sense_class], + "evidence_source": source, + "confidence_status": "INSUFFICIENT" if insufficient else "DETERMINATE", + } + + +def _tokens(text: str) -> tuple[str, ...]: + return tuple(_norm(text).split()) + + +def _norm(text: str) -> str: + return " ".join(text.replace("_", " ").replace("-", " ").casefold().split()) + + +def _referential_yes(gloss: str, synset: Synset) -> tuple[bool, str]: + if any(pointer.symbol == "@i" for pointer in synset.pointers): + return True, "synset.instance_hypernym" + if _LIFESPAN.search(gloss): + return True, "synset.gloss.lifespan" + return False, "none" + + +def _referential_no(gloss: str, synset: Synset, ref_yes: bool) -> bool: + if ref_yes: + return False + return synset.ss_type in _PREDICATE_TYPES or _operator(gloss) + + +def _operator(gloss: str) -> bool: + return bool(_COMPARATIVE.match(gloss) or _SUPERLATIVE.match(gloss)) + + +def _surface_index(tokens: tuple[str, ...], synset: Synset) -> int: + wanted = " ".join(tokens) + for index, lemma in enumerate(synset.lemmas, start=1): + if _norm(lemma) == wanted: + return index + return 0 + + +def _unrelated(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]]) -> bool: + owned = set(tokens) + wanted = " ".join(tokens) + for lemma in synset.lemmas: + if _norm(lemma) == wanted: + continue + parts = _tokens(lemma) + if len(parts) != 1: + continue + word = parts[0] + if word in owned or _linked(word, owned, exceptions): + continue + return True + return False + + +def _linked(word: str, tokens: set[str], exceptions: dict[str, set[str]]) -> bool: + related = exceptions.get(word, set()) + for token in tokens: + if token in related or word in exceptions.get(token, set()): + return True + return False + + +def _lexical_pointer(tokens: tuple[str, ...], synset: Synset) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + return any(pointer.source == source and pointer.symbol in _LEXICAL_SYMBOLS for pointer in synset.pointers) + + +def _alternation(tokens: tuple[str, ...], synset: Synset) -> bool: + groups = [_tokens(lemma) for lemma in synset.lemmas if len(_tokens(lemma)) >= 2] + for left in range(len(groups)): + for right in range(left + 1, len(groups)): + one = groups[left] + other = groups[right] + if len(one) != len(other): + continue + if sum(token != sibling for token, sibling in zip(one, other)) != 1: + continue + if one == tokens or other == tokens: + return True + return False + + +def _constituent_relation(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + owned = set(tokens) + for pointer in synset.pointers: + if pointer.source != source or pointer.symbol not in _RELATION_SYMBOLS or pointer.target < 1: + continue + lemmas = targets.get((pointer.pos, pointer.offset), ()) + if pointer.target > len(lemmas): + continue + parts = _tokens(lemmas[pointer.target - 1]) + if len(parts) == 1 and (parts[0] in owned or _linked(parts[0], owned, exceptions)): + return True + return False diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py new file mode 100644 index 00000000..c8dd4ffe --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v1_replay.py @@ -0,0 +1,308 @@ +"""Replay the frozen 225-row development manifest through the encoded sense screen. + +The replay is development evidence. It does not draw a measurement sample, +revise the frozen procedure, or write a sense class back onto the manifest. +""" + +from __future__ import annotations + +import hashlib +import json +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from hyperlexical.screen_eval import _error_class, _metrics +from hyperlexical.unbind_sense_screen_v1 import classify, load_exceptions, load_wordnet + +LEDGER = Path("/home/morpheus/hlx-private/eval-reserve-20260926") +HYPERLEX = Path("/home/morpheus/Hyperlex") +WORDNET = LEDGER / "acquisition/sources/wordnet-3.0/wordnet" +OUT = LEDGER / "operator-review/HLX-EVAL-UNBIND-SENSE-SCREEN-V1-HYPOTHESIS-001" +PREDICTIONS = OUT / "development_replay_predictions.jsonl" +REPORT = OUT / "development_replay_report.json" +TRACKER = OUT / "HYPOTHESIS.json" +EVIDENCE = OUT / "DEVELOPMENT_EVIDENCE.json" +PROCEDURE = OUT / "CLASSIFICATION_PROCEDURE.json" + +EXPECTED = { + OUT / "ACCEPTANCE.json": "cff6af0f05ec5e12fb29ddfd2ec321addc94c73258c31860345f6d49960065b0", + OUT / "HYPOTHESIS.draft.json": "93375446b1f4a1f70c60f747a56b626ae667c8944d0eea54deddb9d57d3d9e38", + EVIDENCE: "0e9b3c1af9dd573bf6e2034640e468e8ab9074e1e76c90cef1f39f68d607bc03", + OUT / "LINEAGE_RETIREMENT.json": "fd5d9ebb94d7a6e6ea69609c4e2125ec9914f6705ae256b780223bbea2e26f6f", + OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json": "529defbc2b56152c3290d5b09f309764128b035906797229dab54857cd249df0", + PROCEDURE: "4d9dad77d8d315e810863101041229c53570ed16970074c86abaecd0cc3012ad", + TRACKER: "f9c4757b6eec558b5e1baf644bcf33c27c949807e7f00cd15df869eb6411de31", + LEDGER / "events.jsonl": "96b74a92d44f1cf9fe152b18e5207176f161ba3bfce528dac38aa4571a742f9c", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v3.py": "179d8dcc112214c70566bd3c9a0397e1ebab9131666b0ca1f2a3817973aaccc6", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v4.py": "f1e86e2f21544655cda6a136885a186b20885d501cb7ea9c75e18b3dd4a42377", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v5.py": "70504574523f2e8fde0fb974e3027205dded2c96213dd997f44475ea6856f948", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v6.py": "59699496c15aaedfbe69a7e49b5c6e62d1e543ce5a1e0e9a0255a98a62036fba", + HYPERLEX / "scripts/shadow/hyperlexical/unbind_screen_v7.py": "73335bde8eec262ebecfedfc0d0ecb0a965da5c6b66e53c16f2aee2f38b061ab", +} +SENSE_CLASSES = ( + "REFERENTIAL", + "LEXICALIZED_NONCOMPOSITIONAL", + "LEXICALIZED_COMPOSITIONAL", + "ORDINARY_COMPOSITIONAL", + "AMBIGUOUS", +) +BUCKETS = ("HIGH", "SECONDARY", "REJECT", "QUARANTINE") +OPERATOR_BUCKETS = BUCKETS + ("UNRESOLVED",) + + +def sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def refuse(message: str) -> None: + raise SystemExit(message) + + +def write_json(path: Path, payload: dict) -> str: + text = json.dumps(payload, indent=2, sort_keys=True, ensure_ascii=True) + "\n" + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def write_jsonl(path: Path, rows: list[dict]) -> str: + text = "".join(json.dumps(row, sort_keys=True, ensure_ascii=True) + "\n" for row in rows) + path.write_text(text, encoding="utf-8") + path.chmod(0o600) + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fraction(numerator: int, denominator: int) -> str | None: + if denominator == 0: + return None + return f"{numerator}/{denominator}" + + +def main() -> None: + for path, expected in EXPECTED.items(): + found = sha256(path) + if found != expected: + refuse(f"hash mismatch {path.name}: {found}") + if PREDICTIONS.exists() or REPORT.exists(): + refuse("development replay artifacts already exist") + evidence = json.loads(EVIDENCE.read_text(encoding="utf-8")) + if evidence.get("sense_classes_assigned") is not False: + refuse("development manifest already records assigned sense classes") + rows = evidence["rows"] + if len(rows) != 225 or evidence.get("row_count") != 225: + refuse("development manifest is not the frozen 225-row inventory") + if len({row["row_id"] for row in rows}) != 225: + refuse("development row ids are not unique") + if any(row.get("sense_class") is not None for row in rows): + refuse("a development row already has a sense class") + tracker = json.loads(TRACKER.read_text(encoding="utf-8")) + if tracker.get("state") != "PROCEDURE_FROZEN" or tracker.get("encoded") is not False: + refuse("tracker is not waiting at PROCEDURE_FROZEN") + procedure = json.loads(PROCEDURE.read_text(encoding="utf-8")) + if procedure.get("encoded") is not False or procedure.get("development_replay_run") is not False: + refuse("frozen procedure artifact is not in its sealed unencoded state") + + synsets, glosses = load_wordnet(WORDNET) + exceptions = load_exceptions(WORDNET) + targets = {key: synset.lemmas for key, synset in synsets.items()} + classified = [] + for row in rows: + key = (row["synset_pos"], row["synset_offset"]) + synset = synsets.get(key) + gloss = glosses.get(key) + if synset is None or gloss is None: + refuse(f"missing synset {row['synset_pos']} {row['synset_offset']}") + if gloss != row["gloss"]: + refuse(f"frozen gloss does not match the data-line first clause for {row['row_id']}") + decision = classify(row["surface"], row["gloss"], synset, exceptions, targets) + if decision["sense_class"] not in SENSE_CLASSES or decision["bucket"] not in BUCKETS: + refuse(f"classifier returned an unknown class for {row['row_id']}") + classified.append((row, decision)) + + predictions = [] + for row, decision in sorted(classified, key=lambda item: item[0]["row_id"]): + predictions.append( + { + "schema": "hyperlex.unbind_sense_screen_v1_development_row.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "application_index": 1, + "row_id": row["row_id"], + "surface": row["surface"], + "pos": row["pos"], + "synset": f"{row['synset_pos']}:{row['synset_offset']}", + "synset_offset": row["synset_offset"], + "synset_pos": row["synset_pos"], + "sense_class": decision["sense_class"], + "bucket": decision["bucket"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_source": decision["evidence_source"], + "confidence_status": decision["confidence_status"], + } + ) + if any("operator_bucket" in row or "historical_unbind_screen" in row for row in predictions): + refuse("prediction rows carry operator or historical fields") + prediction_sha = write_jsonl(PREDICTIONS, predictions) + + by_id = {row["row_id"]: decision for row, decision in classified} + ordered_rows = sorted(rows, key=lambda row: row["row_id"]) + pairs = [] + sense_by_operator = {op: {sense: 0 for sense in SENSE_CLASSES} for op in OPERATOR_BUCKETS} + error_rows = {"false_high": [], "false_reject": [], "false_secondary": [], "false_quarantine": []} + for row in ordered_rows: + decision = by_id[row["row_id"]] + operator = row["operator_bucket"] + pairs.append((row["row_id"], decision["bucket"], operator, decision["primary_evidence_code"])) + sense_by_operator[operator][decision["sense_class"]] += 1 + kind = _error_class(decision["bucket"], operator) + if kind: + error_rows[kind].append( + { + "row_id": row["row_id"], + "surface": row["surface"], + "operator_bucket": operator, + "bucket": decision["bucket"], + "sense_class": decision["sense_class"], + "primary_evidence_code": decision["primary_evidence_code"], + "evidence_source": decision["evidence_source"], + } + ) + metrics = _metrics(pairs) + sense_counts = Counter(row["sense_class"] for row in predictions) + bucket_counts = Counter(row["bucket"] for row in predictions) + evidence_counts = Counter(row["primary_evidence_code"] for row in predictions) + source_counts = Counter(row["evidence_source"] for row in predictions) + confidence_counts = Counter(row["confidence_status"] for row in predictions) + ambiguous = sense_counts["AMBIGUOUS"] + + def bucket_block(name: str) -> dict: + stats = metrics["per_bucket"][name] + support = stats["support"] + predicted = bucket_counts[name] + true_positive = sum(1 for _i, pred, op, _r in pairs if pred == name and op == name) + return { + "precision": stats["precision"], + "precision_fraction": fraction(true_positive, predicted), + "recall": stats["recall"], + "recall_fraction": fraction(true_positive, support), + "operator_support": support, + "predicted": predicted, + } + + when = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + report = { + "schema": "hyperlex.unbind_sense_screen_v1_development_replay.v1", + "rule": "RUNE.UNBIND_SENSE_SCREEN.v1", + "role": "development evidence, not validation", + "not_a_validation_set": True, + "tuning_on_these_rows_is_not_validation": True, + "state": "DEVELOPMENT_ANALYZED", + "state_path": [ + "PROCEDURE_FROZEN", + "ENCODE_AUTHORIZED", + "ENCODED", + "225_ROW_DEVELOPMENT_REPLAY", + "DEVELOPMENT_ANALYZED", + ], + "analyzed_at": when, + "rows": 225, + "application_index": 1, + "applications_per_row": 1, + "prediction_sha256": prediction_sha, + "procedure_sha256": EXPECTED[PROCEDURE], + "procedure_mutated": False, + "acceptance_sha256": EXPECTED[OUT / "ACCEPTANCE.json"], + "acceptance_mutated": False, + "draft_hypothesis_sha256": EXPECTED[OUT / "HYPOTHESIS.draft.json"], + "development_evidence_sha256": EXPECTED[EVIDENCE], + "development_manifest_sense_classes_written": False, + "lexeme_architecture_sha256": EXPECTED[OUT / "LEXEME_STRUCTURE_SCREEN.architecture.json"], + "lexeme_architecture_mutated": False, + "lineage_retirement_sha256": EXPECTED[OUT / "LINEAGE_RETIREMENT.json"], + "events_sha256": EXPECTED[LEDGER / "events.jsonl"], + "sense_class_counts": {name: sense_counts[name] for name in SENSE_CLASSES}, + "bucket_counts": {name: bucket_counts[name] for name in BUCKETS}, + "operator_vs_bucket_confusion": metrics["confusion_matrix"], + "operator_vs_sense_class": sense_by_operator, + "ambiguous_count": ambiguous, + "ambiguous_rate": ambiguous / 225, + "ambiguous_rate_fraction": fraction(ambiguous, 225), + "ambiguous_rate_is_expected_measurement": True, + "high_ambiguity_was_not_repaired": True, + "absence_of_evidence_is_not_secondary": True, + "evidence_code_counts": dict(sorted(evidence_counts.items())), + "evidence_source_counts": dict(sorted(source_counts.items())), + "confidence_status_counts": dict(sorted(confidence_counts.items())), + "HIGH": bucket_block("HIGH"), + "SECONDARY": bucket_block("SECONDARY"), + "REJECT": bucket_block("REJECT"), + "QUARANTINE": bucket_block("QUARANTINE"), + "quarantine_support": metrics["per_bucket"]["QUARANTINE"]["support"], + "false_high": len(error_rows["false_high"]), + "false_reject": len(error_rows["false_reject"]), + "false_secondary": len(error_rows["false_secondary"]), + "false_quarantine": len(error_rows["false_quarantine"]), + "false_secondary_is_descriptive_only": True, + "false_high_rows": error_rows["false_high"], + "false_reject_rows": error_rows["false_reject"], + "outer_bucket_metrics_are_descriptive": True, + "measurement_bar_applied": False, + "measurement_sample_drawn": False, + "measurement_eligible": False, + "revision_eligible": False, + "select_authorized": False, + "admitted": 0, + "settled": 0, + "gold": 0, + "ledger_appended": False, + "next_legal_transition": "MEASUREMENT_AUTHORIZATION", + "next_transition_authorized": False, + } + report_sha = write_json(REPORT, report) + + untouched = json.loads(EVIDENCE.read_text(encoding="utf-8")) + if any(row.get("sense_class") is not None for row in untouched["rows"]): + refuse("replay wrote a sense class onto the development manifest") + if sha256(EVIDENCE) != EXPECTED[EVIDENCE] or sha256(PROCEDURE) != EXPECTED[PROCEDURE]: + refuse("replay mutated a sealed artifact") + + tracker["previous_state"] = "PROCEDURE_FROZEN" + tracker["previous_tracker_sha256"] = EXPECTED[TRACKER] + tracker["state"] = "DEVELOPMENT_ANALYZED" + tracker["encoded"] = True + tracker["encoding_authorized"] = True + tracker["applied"] = True + tracker["development_replay_run"] = True + tracker["rows_classified"] = 225 + tracker["measurement_sample_drawn"] = False + tracker["measurement_eligible"] = False + tracker["revision_eligible"] = False + tracker["next_legal_transition"] = "MEASUREMENT_AUTHORIZATION" + tracker["next_transition_authorized"] = False + tracker["select_authorized"] = False + tracker["authorized"] = False + tracker["prediction_sha256"] = prediction_sha + tracker["report_sha256"] = report_sha + tracker["ambiguous_rate_fraction"] = report["ambiguous_rate_fraction"] + tracker["high_ambiguity_was_not_repaired"] = True + tracker_sha = write_json(TRACKER, tracker) + print(json.dumps({ + "prediction_sha256": prediction_sha, + "report_sha256": report_sha, + "tracker_sha256": tracker_sha, + "sense_class_counts": report["sense_class_counts"], + "bucket_counts": report["bucket_counts"], + "ambiguous_rate_fraction": report["ambiguous_rate_fraction"], + "HIGH": report["HIGH"], + "SECONDARY": report["SECONDARY"], + "REJECT": report["REJECT"], + "QUARANTINE": report["QUARANTINE"], + "false_high": report["false_high"], + "false_reject": report["false_reject"], + "false_secondary": report["false_secondary"], + "evidence_code_counts": report["evidence_code_counts"], + }, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py b/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py new file mode 100644 index 00000000..2a804a5d --- /dev/null +++ b/scripts/shadow/hyperlexical/unbind_sense_screen_v2.py @@ -0,0 +1,295 @@ +"""Sense screen procedure v2. + +Three states are recorded before the class. A whole-expression co-lemma +is lexicalization. It is not a compositional no, and it is not high. +""" + +from __future__ import annotations + +import re + +from hyperlexical.unbind_sense_screen_v1 import ( + Pointer, + Synset, + load_exceptions, + load_wordnet, + parse_data_line, +) + +RULE_VERSION = "RUNE.UNBIND_SENSE_SCREEN.v1" +PROCEDURE = "hyperlex.unbind_sense_screen_v1_classification_procedure.v2" +_LIFESPAN = re.compile(r"\([0-9]{4}-[0-9]{4}\)") +_COMPARATIVE = re.compile(r"^used to form the comparative\b") +_SUPERLATIVE = re.compile(r"^used to form the superlative\b") +_LEXICAL_SYMBOLS = frozenset({"!", "+", "\\", "^", "*", "&", "<", "$"}) +_RELATION_SYMBOLS = frozenset({"+", "\\"}) +_PREDICATE_TYPES = frozenset({"r", "a", "s"}) +_BUCKETS = { + "REFERENTIAL": "REJECT", + "LEXICALIZED_NONCOMPOSITIONAL": "HIGH", + "LEXICALIZED_COMPOSITIONAL": "SECONDARY", + "ORDINARY_COMPOSITIONAL": "SECONDARY", + "AMBIGUOUS": "QUARANTINE", +} +_FAMILY_ORDER = ( + "referential_designation", + "whole_expression_lexicalization", + "compositional_semantic_relation", + "productive_grammatical_frame", + "noncompositional_semantic_mapping", + "insufficient_record_evidence", + "conflicting_record_evidence", +) +_SOURCE_ORDER = ( + "synset.instance_hypernym", + "synset.gloss.lifespan", + "synset.ss_type.predicate", + "synset.lemmas.unrelated_single_word", + "synset.lexical_pointer", + "synset.gloss.grammatical_operator", + "synset.lexical_pointer.derivation_or_pertainym_to_constituent", + "synset.lemmas.productive_alternation", +) + +__all__ = [ + "Pointer", + "Synset", + "classify", + "load_exceptions", + "load_wordnet", + "parse_data_line", +] + + +def classify(surface: str, gloss: str, synset: Synset, exceptions: dict[str, set[str]], targets: dict[tuple[str, str], tuple[str, ...]]) -> dict: + """Apply procedure v2 once. Compositional NO is not produced.""" + text = gloss or "" + tokens = _tokens(surface) + referential, referential_hits = _referential(text, synset) + lexicalized, lexicalized_hits = _lexicalized(tokens, synset, exceptions, text) + compositional, compositional_hits = _compositional(tokens, synset, exceptions, targets, text) + if compositional in {"NO", "CONFLICT"}: + raise RuntimeError("procedure v2 has no compositional NO signal") + hits = referential_hits + lexicalized_hits + compositional_hits + if referential == "CONFLICT" or (referential != "YES" and lexicalized == "CONFLICT"): + return _decision( + "AMBIGUOUS", + "conflicting_record_evidence", + "CONTRADICTORY", + referential, + lexicalized, + compositional, + hits, + None, + ) + if referential == "YES": + return _decision( + "REFERENTIAL", + "referential_designation", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + referential_hits[0][0], + ) + if lexicalized == "YES" and compositional == "NO": + return _decision( + "LEXICALIZED_NONCOMPOSITIONAL", + "noncompositional_semantic_mapping", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + None, + ) + if lexicalized == "YES" and compositional == "YES": + source, family = compositional_hits[0] + return _decision( + "LEXICALIZED_COMPOSITIONAL", + family, + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + source, + ) + if lexicalized == "NO" and compositional == "YES": + return _decision( + "ORDINARY_COMPOSITIONAL", + "productive_grammatical_frame", + "DETERMINATE", + referential, + lexicalized, + compositional, + hits, + "synset.gloss.grammatical_operator", + ) + return _decision( + "AMBIGUOUS", + "insufficient_record_evidence", + "INSUFFICIENT", + referential, + lexicalized, + compositional, + hits, + None, + ) + + +def _decision(sense_class, primary, confidence, referential, lexicalized, compositional, hits, deciding): + families = [] + fired = [] + for source, family in hits: + if source not in fired: + fired.append(source) + if family and family not in families: + families.append(family) + sources = [name for name in _SOURCE_ORDER if name in fired and name != deciding] + if deciding: + sources.insert(0, deciding) + if not sources: + sources = ["none"] + supporting = [name for name in _FAMILY_ORDER if name in families and name != primary] + return { + "rule": RULE_VERSION, + "procedure": PROCEDURE, + "referential_state": referential, + "lexicalized_state": lexicalized, + "compositional_state": compositional, + "sense_class": sense_class, + "bucket": _BUCKETS[sense_class], + "primary_evidence_code": primary, + "supporting_evidence_codes": supporting, + "evidence_sources": sources, + "confidence_status": confidence, + } + + +def _referential(gloss: str, synset: Synset): + if any(pointer.symbol == "@i" for pointer in synset.pointers): + return "YES", [("synset.instance_hypernym", "referential_designation")] + if _LIFESPAN.search(gloss): + return "YES", [("synset.gloss.lifespan", "referential_designation")] + hits = [] + if synset.ss_type in _PREDICATE_TYPES: + hits.append(("synset.ss_type.predicate", None)) + if _operator(gloss): + hits.append(("synset.gloss.grammatical_operator", "productive_grammatical_frame")) + if hits: + return "NO", hits + return "UNKNOWN", [] + + +def _lexicalized(tokens, synset: Synset, exceptions, gloss: str): + hits = [] + if _unrelated(tokens, synset, exceptions): + hits.append(("synset.lemmas.unrelated_single_word", "whole_expression_lexicalization")) + if _lexical_pointer(tokens, synset): + hits.append(("synset.lexical_pointer", "whole_expression_lexicalization")) + operator = [("synset.gloss.grammatical_operator", "productive_grammatical_frame")] if _operator(gloss) else [] + if hits and operator: + return "CONFLICT", hits + operator + if hits: + return "YES", hits + if operator: + return "NO", operator + return "UNKNOWN", [] + + +def _compositional(tokens, synset: Synset, exceptions, targets, gloss: str): + hits = [] + if _constituent_relation(tokens, synset, exceptions, targets): + hits.append(("synset.lexical_pointer.derivation_or_pertainym_to_constituent", "compositional_semantic_relation")) + if _alternation(tokens, synset): + hits.append(("synset.lemmas.productive_alternation", "productive_grammatical_frame")) + if _operator(gloss): + hits.append(("synset.gloss.grammatical_operator", "productive_grammatical_frame")) + if hits: + return "YES", hits + return "UNKNOWN", [] + + +def _tokens(text: str) -> tuple[str, ...]: + return tuple(_norm(text).split()) + + +def _norm(text: str) -> str: + return " ".join(text.replace("_", " ").replace("-", " ").casefold().split()) + + +def _operator(gloss: str) -> bool: + return bool(_COMPARATIVE.match(gloss or "") or _SUPERLATIVE.match(gloss or "")) + + +def _surface_index(tokens: tuple[str, ...], synset: Synset) -> int: + wanted = " ".join(tokens) + for index, lemma in enumerate(synset.lemmas, start=1): + if _norm(lemma) == wanted: + return index + return 0 + + +def _unrelated(tokens: tuple[str, ...], synset: Synset, exceptions: dict[str, set[str]]) -> bool: + owned = set(tokens) + wanted = " ".join(tokens) + for lemma in synset.lemmas: + if _norm(lemma) == wanted: + continue + parts = _tokens(lemma) + if len(parts) != 1: + continue + word = parts[0] + if word in owned or _linked(word, owned, exceptions): + continue + return True + return False + + +def _linked(word: str, tokens: set[str], exceptions: dict[str, set[str]]) -> bool: + related = exceptions.get(word, set()) + for token in tokens: + if token in related or word in exceptions.get(token, set()): + return True + return False + + +def _lexical_pointer(tokens: tuple[str, ...], synset: Synset) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + return any(pointer.source == source and pointer.symbol in _LEXICAL_SYMBOLS for pointer in synset.pointers) + + +def _alternation(tokens: tuple[str, ...], synset: Synset) -> bool: + groups = [_tokens(lemma) for lemma in synset.lemmas if len(_tokens(lemma)) >= 2] + for left in range(len(groups)): + for right in range(left + 1, len(groups)): + one = groups[left] + other = groups[right] + if len(one) != len(other): + continue + if sum(token != sibling for token, sibling in zip(one, other)) != 1: + continue + if one == tokens or other == tokens: + return True + return False + + +def _constituent_relation(tokens, synset: Synset, exceptions, targets) -> bool: + source = _surface_index(tokens, synset) + if source == 0: + return False + owned = set(tokens) + for pointer in synset.pointers: + if pointer.source != source or pointer.symbol not in _RELATION_SYMBOLS or pointer.target < 1: + continue + lemmas = targets.get((pointer.pos, pointer.offset), ()) + if pointer.target > len(lemmas): + continue + parts = _tokens(lemmas[pointer.target - 1]) + if len(parts) == 1 and (parts[0] in owned or _linked(parts[0], owned, exceptions)): + return True + return False diff --git a/tests/shadow/test_magpie_candidate_evaluation.py b/tests/shadow/test_magpie_candidate_evaluation.py new file mode 100644 index 00000000..ac91218e --- /dev/null +++ b/tests/shadow/test_magpie_candidate_evaluation.py @@ -0,0 +1,233 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.magpie_candidate_evaluation import ( + ALIGNED_IDIOMATIC, + ALIGNED_LITERAL, + AMBIGUOUS, + CONFLICT, + EXACT, + MIXED, + NONE, + NORMALIZED, + UNKNOWN, + build_index, + evaluate_row, + exact_key, + normalized_key, + semantic_noncompositional, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "magpie_candidate_evaluation.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) +VERSION = "test-version" +ARTIFACT = "b" * 64 + + +def _instance(identifier, idiom, label="i", variant_type="identical", confidence=1, **extra): + row = { + "confidence": confidence, + "id": identifier, + "idiom": idiom, + "label": label, + "variant_type": variant_type, + } + row.update(extra) + return row + + +def _row(surface, synset, instances): + return evaluate_row(surface, synset, build_index(instances), VERSION, ARTIFACT) + + +def test_module_has_no_phrase_exception_or_probe_surface(): + assert rule_surface_violations(MODULE.read_text(encoding="utf-8"), PROBES) == [] + + +def test_evaluator_does_not_accept_operator_labels_or_gloss(): + names = set(inspect.signature(evaluate_row).parameters) + assert "operator_bucket" not in names + assert "gloss" not in names + assert "operator_reason" not in names + + +def test_exact_match_is_case_and_separator_only(): + found = _row( + "At The End", + "noun:1", + [_instance(1, "at the end", "i"), _instance(2, "at the end", "l")], + ) + assert found["surface_match"] == EXACT + assert found["matched_magpie_expression"] == "at the end" + assert found["idiomatic_instance_count"] == 1 + assert found["literal_instance_count"] == 1 + assert found["sense_alignment"] == UNKNOWN + assert found["semantic_noncompositional"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + assert found["source_instance_ids"] == [1, 2] + + +def test_punctuation_difference_is_normalized_and_not_yes(): + found = _row("end of the day", "noun:1", [_instance(4, "end of the day.")]) + assert exact_key("end of the day") != exact_key("end of the day.") + assert found["surface_match"] == NORMALIZED + assert found["semantic_noncompositional"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + + +def test_curly_apostrophe_is_normalized_and_pronouns_are_not_rewritten(): + curly = "flip one\u2019s lid" + found = _row("flip one's lid", "noun:1", [_instance(5, curly, "i")]) + assert found["surface_match"] == NORMALIZED + assert found["semantic_noncompositional"] == UNKNOWN + other = _row("flip his lid", "noun:1", [_instance(6, "flip one's lid", "i")]) + assert other["surface_match"] == NONE + assert other["primary_evidence_code"] == "magpie_no_match" + assert normalized_key("someone's chair") != normalized_key("one's chair") + + +def test_inflection_variant_metadata_does_not_invent_a_match(): + found = _row( + "threw in the towel", + "verb:1", + [_instance(7, "throw in the towel", "i", variant_type="inflection")], + ) + assert found["surface_match"] == NONE + assert found["semantic_noncompositional"] == UNKNOWN + + +def test_colliding_normalized_types_are_ambiguous_and_unattributed(): + found = _row( + "a b", + "noun:1", + [_instance(8, "a-b"), _instance(9, "a b.")], + ) + assert found["surface_match"] == AMBIGUOUS + assert found["matched_magpie_expression"] is None + assert found["literal_instance_count"] is None + assert found["source_instance_ids"] == [] + assert found["ambiguous_candidates"] == ["a b.", "a-b"] + assert found["sense_alignment"] == UNKNOWN + assert found["primary_evidence_code"] == "magpie_surface_only" + + +def test_unique_exact_wins_over_another_normalized_type(): + found = _row( + "a b", + "noun:1", + [_instance(10, "a b", "l"), _instance(11, "a-b", "i")], + ) + assert found["surface_match"] == EXACT + assert found["matched_magpie_expression"] == "a b" + assert found["literal_instance_count"] == 1 + assert found["idiomatic_instance_count"] == 0 + + +def test_all_idiomatic_labels_without_a_sense_identifier_stay_unknown(): + found = _row( + "bear fruit", + "verb:9", + [_instance(12, "bear fruit", "i"), _instance(13, "bear fruit", "i")], + ) + assert found["idiomatic_instance_count"] == 2 + assert found["literal_instance_count"] == 0 + assert found["sense_alignment"] == UNKNOWN + assert found["alignment_basis"] == "no_bound_sense_identifier" + assert found["semantic_noncompositional"] == UNKNOWN + + +def test_gloss_text_cannot_change_alignment(): + instances = [_instance(14, "bear fruit", "i")] + index = build_index(instances) + first = evaluate_row("bear fruit", "verb:9", index, VERSION, ARTIFACT) + second = evaluate_row("bear fruit", "verb:9", index, VERSION, ARTIFACT) + assert first == second + assert "gloss" not in first + + +def test_equal_sense_identifier_can_align_without_using_labels_as_rules(): + instances = [ + _instance(15, "bear fruit", "i", synset="verb:9"), + _instance(16, "bear fruit", "i", synset="verb:9"), + ] + found = _row("bear fruit", "verb:9", instances) + assert found["sense_alignment"] == ALIGNED_IDIOMATIC + assert found["semantic_noncompositional"] == "YES" + assert found["primary_evidence_code"] == "magpie_aligned_idiomatic" + literal = _row( + "bear fruit", + "verb:9", + [_instance(17, "bear fruit", "l", synset="verb:9")], + ) + assert literal["sense_alignment"] == ALIGNED_LITERAL + assert literal["semantic_noncompositional"] == "NO" + mixed = _row( + "bear fruit", + "verb:9", + [ + _instance(18, "bear fruit", "i", synset="verb:9"), + _instance(19, "bear fruit", "l", synset="verb:9"), + ], + ) + assert mixed["sense_alignment"] == MIXED + assert mixed["semantic_noncompositional"] == UNKNOWN + assert mixed["primary_evidence_code"] == "magpie_mixed_usage" + + +def test_a_different_sense_identifier_does_not_transfer(): + found = _row( + "bear fruit", + "verb:9", + [_instance(20, "bear fruit", "i", synset="verb:8")], + ) + assert found["sense_alignment"] == UNKNOWN + assert found["alignment_basis"] == "sense_identifier_mismatch" + assert found["semantic_noncompositional"] == UNKNOWN + conflict = _row( + "bear fruit", + "verb:9", + [ + _instance(21, "bear fruit", "i", synset="verb:9"), + _instance(22, "bear fruit", "i", synset="verb:8"), + ], + ) + assert conflict["sense_alignment"] == CONFLICT + assert conflict["semantic_noncompositional"] == UNKNOWN + assert conflict["primary_evidence_code"] == "magpie_sense_conflict" + + +def test_other_and_unclear_labels_stay_unresolved(): + found = _row( + "bear fruit", + "verb:9", + [ + _instance(23, "bear fruit", "o"), + _instance(24, "bear fruit", "?"), + _instance(25, "bear fruit", "f"), + ], + ) + assert found["unresolved_instance_count"] == 3 + assert found["idiomatic_instance_count"] == 0 + assert semantic_noncompositional(UNKNOWN) == UNKNOWN + assert semantic_noncompositional(ALIGNED_IDIOMATIC) == "YES" + assert semantic_noncompositional(ALIGNED_LITERAL) == "NO" + assert semantic_noncompositional(MIXED) == UNKNOWN + assert semantic_noncompositional(CONFLICT) == UNKNOWN diff --git a/tests/shadow/test_model_based_wsd_candidate_v1.py b/tests/shadow/test_model_based_wsd_candidate_v1.py new file mode 100644 index 00000000..811c96e9 --- /dev/null +++ b/tests/shadow/test_model_based_wsd_candidate_v1.py @@ -0,0 +1,243 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.model_based_wsd_candidate_v1 import ( + candidate_gloss_text, + candidate_policy, + coverage_gate, + format_probability, + gloss_lemma, + overlay_status, + project_row_status, + quoted_context, + resolve_model_scores, + summarize_confidence, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "model_based_wsd_candidate_v1.py" +REPLAY = ROOT / "scripts" / "shadow" / "hyperlexical" / "model_based_wsd_candidate_v1_replay.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) +KEYS = { + "noun:1": ["alpha%1:00:00::"], + "noun:2": ["beta%1:00:00::"], +} + + +def test_module_freezes_one_model_and_hides_labels_from_the_decision(): + source = MODULE.read_text(encoding="utf-8") + replay = REPLAY.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] + assert "sentence_transformers" not in source + assert "MiniLM" not in source + assert "sentence_transformers" not in replay + assert "MiniLM" not in replay + names = set(inspect.signature(resolve_model_scores).parameters) + assert "operator_bucket" not in names + assert "residual_score" not in names + policy = candidate_policy() + assert policy["state_at_freeze"] == "SPEC_FROZEN" + assert policy["applied_to_hyperlex_at_freeze"] is False + assert policy["selected_source"] == "none" + assert policy["runtime_integration"] is False + assert policy["residual_replay_authorized"] is False + assert policy["model_name"] == "kanishka/GlossBERT" + assert policy["model_revision"] == "0cc3b83af5496e27ebcc95ef0cf37ea0a9281a7a" + assert policy["positive_class_index"] == 1 + assert policy["license"] == "MIT" + assert "operator_labels" in policy["forbidden_inputs"] + assert "residual_embeddings" in policy["forbidden_inputs"] + assert "residual_scores" in policy["forbidden_inputs"] + assert policy["readiness_gate"]["baseline_high_ready"] == 4 + assert policy["readiness_gate"]["baseline_secondary_ready"] == 4 + assert policy["readiness_gate"]["baseline_total_ready"] == 20 + assert policy["readiness_gate"]["comparison"] == "strictly_greater" + assert policy["no_gold_constituent_senses"] is True + assert policy["pwn_mapping"]["cross_version_map"] is False + assert policy["pwn_mapping"]["manual_migration"] is False + + +def test_context_quotes_only_the_indexed_content_token(): + assert quoted_context("alpha in beta", 0, "alpha") == '"alpha" in beta' + assert quoted_context("alpha in beta", 1, "beta") == 'alpha in "beta"' + assert gloss_lemma(["bank", "other"], "banks", {"banks": {"bank"}, "bank": {"banks"}}) == "bank" + assert candidate_gloss_text("bank", "sloping land") == "bank: sloping land" + assert format_probability(0.5) == "0.500000" + + +def test_abstention_ties_and_invalid_senses_do_not_select(): + resolved = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.800000", "noun:2": "0.200000"}, + ) + assert resolved["model_resolution_status"] == "RESOLVED" + assert resolved["selected_synset"] == "noun:1" + assert resolved["selected_sense_key"] == "alpha%1:00:00::" + assert resolved["primary_evidence_code"] == "model_margin" + boundary = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.500000", "noun:2": "0.400000"}, + ) + assert boundary["model_resolution_status"] == "RESOLVED" + assert boundary["model_margin"] == "0.100000" + near = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.550000", "noun:2": "0.500000"}, + ) + assert near["model_resolution_status"] == "AMBIGUOUS" + assert near["selected_synset"] is None + assert near["primary_evidence_code"] == "model_abstention" + low = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.400000", "noun:2": "0.100000"}, + ) + assert low["model_resolution_status"] == "AMBIGUOUS" + assert low["selected_synset"] is None + tie = resolve_model_scores( + ["noun:2", "noun:1"], + KEYS, + {"noun:1": "0.800000", "noun:2": "0.800000"}, + ) + assert tie["model_resolution_status"] == "AMBIGUOUS" + assert tie["selected_synset"] is None + assert tie["primary_evidence_code"] == "model_score_tie" + outsider = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.900000", "noun:9": "0.100000"}, + ) + assert outsider["model_resolution_status"] == "INVALID" + assert outsider["selected_synset"] is None + assert outsider["primary_evidence_code"] == "invalid_model_output" + many_keys = resolve_model_scores( + ["noun:1", "noun:2"], + {"noun:1": ["alpha%1:00:00::", "alpha%1:00:01::"], "noun:2": ["beta%1:00:00::"]}, + {"noun:1": "0.900000", "noun:2": "0.100000"}, + ) + assert many_keys["model_resolution_status"] == "AMBIGUOUS" + assert many_keys["selected_sense_key"] is None + assert many_keys["selected_synset"] is None + assert many_keys["primary_evidence_code"] == "pwn30_sense_key_not_unique" + flooded = resolve_model_scores( + ["noun:1", "noun:2"], + KEYS, + {"noun:1": "0.990000", "noun:2": "0.010000"}, + overflow=True, + ) + assert flooded["model_resolution_status"] == "AMBIGUOUS" + assert flooded["selected_synset"] is None + assert flooded["primary_evidence_code"] == "context_overflow" + failed = resolve_model_scores(["noun:1", "noun:2"], KEYS, None, error="forward failed") + assert failed["model_resolution_status"] == "ERROR" + assert failed["selected_synset"] is None + + +def test_tier3_cannot_override_an_earlier_tier_and_readiness_needs_two_constituents(): + assert overlay_status("EXACT", "STRUCTURAL_EXACT", None) == "EXACT" + assert overlay_status("RESOLVED", "EXTENDED_LESK_V1", None) == "LESK_RESOLVED" + assert overlay_status("UNRESOLVED", "NONE", None) == "UNRESOLVED" + assert overlay_status("AMBIGUOUS", "EXTENDED_LESK_V1", "RESOLVED") == "MODEL_RESOLVED" + assert overlay_status("AMBIGUOUS", "EXTENDED_LESK_V1", "AMBIGUOUS") == "AMBIGUOUS" + assert project_row_status(["EXACT", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert project_row_status(["LESK_RESOLVED", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert project_row_status(["EXACT", "AMBIGUOUS"]) == "UNKNOWN" + assert project_row_status(["EXACT"]) == "UNKNOWN" + try: + overlay_status("EXACT", "STRUCTURAL_EXACT", "RESOLVED") + except RuntimeError as exc: + assert "overrode" in str(exc) + else: + raise AssertionError("exact override was accepted") + + +def test_coverage_gate_is_strict_and_equality_is_not_promising(): + promising = coverage_gate( + high_ready=5, + secondary_ready=5, + total_ready=21, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert promising["candidate_status"] == "CANDIDATE_PROMISING" + assert promising["next_transition_authorized"] is False + baseline = coverage_gate( + high_ready=4, + secondary_ready=4, + total_ready=20, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert baseline["candidate_status"] == "CANDIDATE_INSUFFICIENT" + one_sided = coverage_gate( + high_ready=5, + secondary_ready=4, + total_ready=21, + invalid_output_count=0, + error_count=0, + determinism="IDENTICAL", + ) + assert one_sided["candidate_status"] == "CANDIDATE_INSUFFICIENT" + rejected = coverage_gate( + high_ready=10, + secondary_ready=10, + total_ready=40, + invalid_output_count=1, + error_count=0, + determinism="IDENTICAL", + ) + assert rejected["candidate_status"] == "CANDIDATE_REJECTED" + drifted = coverage_gate( + high_ready=10, + secondary_ready=10, + total_ready=40, + invalid_output_count=0, + error_count=0, + determinism="MISMATCH", + ) + assert drifted["candidate_status"] == "NOT_DETERMINISTIC" + summary = summarize_confidence( + [ + { + "candidate_pwn30_synsets": ["noun:1", "noun:2"], + "model_confidence": "0.750000", + "model_margin": "0.300000", + "model_resolution_status": "RESOLVED", + }, + { + "candidate_pwn30_synsets": ["noun:1", "noun:2", "noun:3"], + "model_confidence": "0.400000", + "model_margin": "0.050000", + "model_resolution_status": "AMBIGUOUS", + }, + ] + ) + assert summary["resolved"] == 1 + assert summary["abstained"] == 1 + assert summary["confidence_bins"]["0.70_to_0.80"] == 1 + assert summary["confidence_bins"]["below_0.50"] == 1 + assert summary["margin_bins"]["0.25_to_0.50"] == 1 + assert summary["by_candidate_count"]["2"]["resolved"] == 1 + assert "operator_bucket" not in summary diff --git a/tests/shadow/test_residual_model_resolved_replay_v1.py b/tests/shadow/test_residual_model_resolved_replay_v1.py new file mode 100644 index 00000000..cd6e9a07 --- /dev/null +++ b/tests/shadow/test_residual_model_resolved_replay_v1.py @@ -0,0 +1,233 @@ +import inspect +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.residual_model_resolved_replay_v1 import ( + AMBIGUOUS, + BOOTSTRAP_RESAMPLES, + BOOTSTRAP_SEED, + EXACT, + INVERTED, + NOT_COMPUTABLE, + NO_SEPARATION, + RESOLVED, + SUPPORTED, + UNRESOLVED, + abstention_reason, + analysis_plan, + bootstrap_intervals, + direction_result, + extreme_driven, + full_distribution, + integrate_constituent, + pair_comparison, + projection_token, + replay_decision, + row_projection, +) +from hyperlexical.semantic_compositionality_residual import percentile +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "residual_model_resolved_replay_v1.py" +REPLAY = ROOT / "scripts" / "shadow" / "hyperlexical" / "residual_model_resolved_replay_v1_replay.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _resolver(**overrides): + row = { + "candidate_synsets": ["noun:00000001", "noun:00000002"], + "constituent_index": 0, + "constituent_pos": "noun", + "constituent_surface": "alpha", + "parent_row_id": "row-1", + "parent_surface": "alpha beta", + "parent_synset": "noun:00000009", + "primary_evidence_code": "extended_lesk_tie", + "resolution_method": "EXTENDED_LESK_V1", + "resolution_status": AMBIGUOUS, + "selected_synset": None, + } + row.update(overrides) + return row + + +def _model(**overrides): + row = { + "model_resolution_status": RESOLVED, + "primary_evidence_code": "model_margin", + "prior_resolution_status": AMBIGUOUS, + "selected_sense_key": "alpha%1:00:00::", + "selected_synset": "noun:00000001", + } + row.update(overrides) + return row + + +def test_module_hides_labels_and_does_not_encode(): + source = MODULE.read_text(encoding="utf-8") + replay = REPLAY.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] + assert "sentence_transformers" not in source + assert "torch" not in source + assert "operator_bucket" not in source + assert "semantic_noncompositional = YES" not in source + assert "semantic_noncompositional = NO" not in source + assert "operator_bucket" not in inspect.signature(integrate_constituent).parameters + plan = analysis_plan() + assert plan["bootstrap_seed"] == BOOTSTRAP_SEED == 0 + assert plan["bootstrap_resamples"] == BOOTSTRAP_RESAMPLES == 10000 + assert plan["emits_yes_no"] is False + assert plan["semantic_noncompositionality_threshold"] is None + assert plan["threshold_eligible"] is False + assert plan["json_schema_document"] is None + assert replay.index("write_json(RECEIPT_PATH") < replay.index("buckets = load_operator_buckets()") + assert "semantic_noncompositional = YES" not in replay + assert "semantic_noncompositional = NO" not in replay + + +def test_integration_copies_frozen_tiers_and_refuses_overrides(): + exact = integrate_constituent( + _resolver( + primary_evidence_code="structural_exact", + resolution_method="STRUCTURAL_EXACT", + resolution_status=EXACT, + selected_synset="noun:00000003", + ), + None, + ) + assert exact["resolution_tier"] == "TIER1_STRUCTURAL" + assert exact["resolution_status"] == EXACT + assert exact["selected_pwn30_synset"] == "noun:00000003" + assert exact["selected_sense_key_if_available"] is None + with pytest.raises(RuntimeError): + integrate_constituent( + _resolver(resolution_method="UNIQUE_LEMMA", resolution_status=EXACT, selected_synset="noun:1"), + _model(), + ) + with pytest.raises(RuntimeError): + integrate_constituent( + _resolver(resolution_method="EXTENDED_LESK_V1", resolution_status=RESOLVED, selected_synset="noun:1"), + _model(), + ) + copied = integrate_constituent(_resolver(), _model()) + assert copied["resolution_tier"] == "TIER3_GLOSSBERT" + assert copied["resolution_status"] == RESOLVED + assert copied["selected_pwn30_synset"] == "noun:00000001" + assert copied["resolution_provenance"] == "model_margin" + with pytest.raises(RuntimeError): + integrate_constituent(_resolver(), _model(selected_synset="noun:99999999")) + abstained = integrate_constituent( + _resolver(), + _model(model_resolution_status=AMBIGUOUS, selected_sense_key=None, selected_synset=None), + ) + assert abstained["resolution_status"] == AMBIGUOUS + assert abstained["selected_pwn30_synset"] is None + unresolved = integrate_constituent( + _resolver( + primary_evidence_code="no_candidate", + resolution_method="NONE", + resolution_status=UNRESOLVED, + selected_synset=None, + candidate_synsets=[], + ), + None, + ) + assert unresolved["resolution_tier"] == UNRESOLVED + assert projection_token("TIER2_EXTENDED_LESK", RESOLVED) == "LESK_RESOLVED" + assert projection_token("TIER3_GLOSSBERT", RESOLVED) == "MODEL_RESOLVED" + assert row_projection(["EXACT"]) == "UNKNOWN" + assert row_projection(["EXACT", "LESK_RESOLVED", "MODEL_RESOLVED"]) == "RESIDUAL_READY" + assert abstention_reason(["EXACT"]) == "fewer_than_two_content_constituents" + assert abstention_reason([AMBIGUOUS, UNRESOLVED]) == "ambiguous_content_constituent" + assert abstention_reason([EXACT, RESOLVED]) is None + + +def test_decision_gate_and_direction_are_preregistered(): + high = ["0.9000000000", "0.5000000000", "0.2000000000"] + secondary = ["0.4000000000", "0.4000000000", "0.1000000000"] + comparison = pair_comparison(high, secondary) + assert comparison["status"] == "DESCRIPTIVE" + assert direction_result(comparison) == SUPPORTED + assert Decimal_gt(comparison["median_difference"]) + assert Decimal_gt(comparison["rank_biserial"]) + assert extreme_driven(high, secondary) is True + promising = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + ) + assert promising["candidate_status"] == "CANDIDATE_PROMISING" + assert promising["threshold_eligible"] is False + assert promising["next_transition_authorized"] is False + concentrated = replay_decision( + readiness_reproduced=True, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=True, + extremes=False, + pos_split=False, + ) + assert concentrated["candidate_status"] == "CANDIDATE_INSUFFICIENT" + assert replay_decision( + readiness_reproduced=False, + determinism="IDENTICAL", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + )["candidate_status"] == NOT_COMPUTABLE + assert replay_decision( + readiness_reproduced=True, + determinism="DIFFERENT", + direction=SUPPORTED, + tier3_concentrated=False, + extremes=False, + pos_split=False, + )["candidate_status"] == "NOT_DETERMINISTIC" + inverted = pair_comparison(secondary, high) + assert direction_result(inverted) == INVERTED + flat = pair_comparison(["0.2000000000", "0.2000000000"], ["0.2000000000"]) + assert direction_result(flat) == NO_SEPARATION + assert pair_comparison([], ["0.1"])["status"] == NOT_COMPUTABLE + + +def Decimal_gt(text: str) -> bool: + from decimal import Decimal + + return Decimal(text) > 0 + + +def test_distribution_matches_residual_percentile_and_bootstrap_is_seeded(): + scores = ["0.1000000000", "0.2000000000", "0.4000000000", "0.8000000000"] + report = full_distribution(scores) + assert report["p25"] == percentile(sorted(scores, key=lambda item: __import__("decimal").Decimal(item)), 25) + assert report["count"] == 4 + assert report["std"] is not None + first = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=0, resamples=30) + repeat = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=0, resamples=30) + other = bootstrap_intervals(["0.2", "0.4", "0.9"], ["0.1", "0.3"], seed=1, resamples=30) + assert first == repeat + assert first != other + assert first["seed"] == 0 + assert first["resamples"] == 30 diff --git a/tests/shadow/test_semantic_compositionality_residual.py b/tests/shadow/test_semantic_compositionality_residual.py new file mode 100644 index 00000000..367315f8 --- /dev/null +++ b/tests/shadow/test_semantic_compositionality_residual.py @@ -0,0 +1,251 @@ +import inspect +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.semantic_compositionality_residual import ( + AMBIGUOUS, + EXACT, + UNIQUE, + UNRESOLVED, + candidate_policy, + distribution, + evaluation_status, + exact_synset_ids, + extract_constituents, + lexical_synset_ids, + percentile, + representation_text, + residual_score, + resolve_constituent, + resolved_synset, + score_record, + select_lemma, + vector_hash, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +MODULE = ROOT / "scripts" / "shadow" / "hyperlexical" / "semantic_compositionality_residual.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _extraction(surface: str = "alpha beta") -> dict: + return extract_constituents(surface) + + +def test_module_has_no_phrase_exception_or_probe_surface(): + assert rule_surface_violations(MODULE.read_text(encoding="utf-8"), PROBES) == [] + + +def test_resolution_does_not_accept_gloss_or_operator_labels(): + blocked = ( + exact_synset_ids, + lexical_synset_ids, + resolve_constituent, + resolved_synset, + score_record, + residual_score, + ) + for function in blocked: + names = set(inspect.signature(function).parameters) + assert "gloss" not in names + assert "operator_bucket" not in names + assert "operator_label" not in names + policy = candidate_policy() + assert "gloss_similarity" in policy["constituent_sense_resolution"]["forbidden"] + assert policy["semantic_noncompositionality_threshold"] is None + assert policy["emits_yes_no"] is False + assert not hasattr(sys.modules["hyperlexical.semantic_compositionality_residual"], "gloss_similarity") + + +def test_structural_tokens_leave_content_words(): + towel = extract_constituents("throw in the towel") + assert towel["surface_tokens"] == ["throw", "in", "the", "towel"] + assert towel["content_constituents"] == ["throw", "towel"] + assert towel["ignored_structural_tokens"] == ["in", "the"] + assert towel["constituent_extraction_status"] == "EXTRACTED" + shots = extract_constituents("call the shots") + assert shots["content_constituents"] == ["call", "shots"] + assert shots["ignored_structural_tokens"] == ["the"] + luck = extract_constituents("as luck would have it") + assert luck["content_constituents"] == ["luck"] + assert luck["constituent_extraction_status"] == "UNKNOWN" + folded = extract_constituents("Throw In The Towel") + assert folded["content_constituents"] == ["Throw", "Towel"] + assert folded["ignored_structural_tokens"] == ["In", "The"] + hyphen = extract_constituents("well-known person") + assert hyphen["content_constituents"] == ["well-known", "person"] + owned = extract_constituents("one\u2019s own goal") + assert owned["ignored_structural_tokens"] == ["one\u2019s"] + assert owned["content_constituents"] == ["own", "goal"] + + +def test_exact_pointer_beats_polysemy_and_zero_target_does_not_count(): + pointers = [ + ("+", 1, "noun:1", "towel"), + ("+", 0, "noun:9", "towel"), + ("@", 1, "noun:8", "towel"), + ("\\", 2, "noun:1", "towels"), + ] + exact = exact_synset_ids( + pointers, + "towel", + {"towel": {"towels"}, "towels": {"towel"}}, + ) + assert exact == ["noun:1", "noun:1"] + assert resolve_constituent(exact, ["noun:1", "noun:2", "verb:3"]) == EXACT + assert resolved_synset(exact, ["noun:2"]) == "noun:1" + many = exact_synset_ids( + [("+", 1, "noun:1", "shot"), ("+", 1, "noun:2", "shot")], + "shots", + {"shots": {"shot"}}, + ) + assert resolve_constituent(many, ["noun:1"]) == AMBIGUOUS + assert resolved_synset(many, ["noun:1"]) is None + + +def test_lexical_resolution_abstains_when_several_synsets_match(): + index = {"towel": ["noun:4"], "shot": ["noun:5", "noun:6"]} + one = lexical_synset_ids(index, "towel", {}) + assert resolve_constituent([], one) == UNIQUE + assert resolved_synset([], one) == "noun:4" + many = lexical_synset_ids(index, "shots", {"shots": {"shot"}}) + assert resolve_constituent([], many) == AMBIGUOUS + assert resolve_constituent([], []) == UNRESOLVED + assert select_lemma(["towel", "bath_towel"], "towels", {"towels": {"towel"}}) == "towel" + + +def test_residual_is_continuous_and_deterministic(): + same = residual_score([1.0, 0.0], [[1.0, 0.0], [1.0, 0.0]]) + assert same is not None + assert same[0] == "0.0000000000" + orthogonal = residual_score([1.0, 0.0], [[0.0, 1.0], [0.0, 1.0]]) + assert orthogonal is not None + assert orthogonal[0] == "1.0000000000" + assert residual_score([0.0, 0.0], [[1.0, 0.0], [0.0, 1.0]]) is None + assert vector_hash([1.0, 0.0]) == vector_hash([1.0, 0.0]) + assert vector_hash([1.0, 0.0]) != vector_hash([0.0, 1.0]) + assert len(vector_hash([1.0])) == 64 + text = representation_text("bath_towel", "noun", " a towel ") + assert text == "bath towel (noun): a towel" + + +def test_ambiguous_constituent_is_unknown_and_is_not_encoded(): + extraction = _extraction("call the shots") + record = score_record( + row_id="row", + surface="call the shots", + pos="verb", + synset="verb:00000001", + extraction=extraction, + resolutions=["UNIQUE", "AMBIGUOUS"], + resolved_synsets=["verb:2", None], + resolved_lemmas=["call", None], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + ) + assert record["score_status"] == "UNKNOWN" + assert record["primary_abstention_reason"] == "ambiguous_content_constituent" + assert record["residual_score"] is None + assert record["whole_vector_hash"] is None + assert record["constituent_resolution_status"] == ["UNIQUE", "AMBIGUOUS"] + short = score_record( + row_id="row", + surface="as luck would have it", + pos="noun", + synset="noun:00000002", + extraction=extract_constituents("as luck would have it"), + resolutions=[], + resolved_synsets=[], + resolved_lemmas=[], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + ) + assert short["primary_abstention_reason"] == "fewer_than_two_content_constituents" + assert short["resolved_constituent_synsets"] == [] + + +def test_scored_row_keeps_a_residual_and_status_ignores_agreement(): + extraction = _extraction() + record = score_record( + row_id="row", + surface="alpha beta", + pos="noun", + synset="noun:00000003", + extraction=extraction, + resolutions=["EXACT", "UNIQUE"], + resolved_synsets=["noun:1", "noun:2"], + resolved_lemmas=["alpha", "beta"], + whole_representation="alpha beta (noun): a whole", + constituent_representations=["alpha (noun): one", "beta (noun): two"], + whole_vector=[1.0, 0.0], + constituent_vectors=[[1.0, 0.0], [1.0, 0.0]], + candidate_spec_sha256="spec", + ) + assert record["score_status"] == "SCORED" + assert record["residual_score"] == "0.0000000000" + assert record["primary_abstention_reason"] is None + assert record["composition_operator"] == "normalized_mean_v1" + assert record["whole_vector_hash"] + assert len(record["constituent_vector_hashes"]) == 2 + overflow = score_record( + row_id="row", + surface="alpha beta", + pos="noun", + synset="noun:00000003", + extraction=extraction, + resolutions=["EXACT", "UNIQUE"], + resolved_synsets=["noun:1", "noun:2"], + resolved_lemmas=["alpha", "beta"], + whole_representation=None, + constituent_representations=None, + whole_vector=None, + constituent_vectors=None, + candidate_spec_sha256="spec", + sequence_overflow=True, + ) + assert overflow["primary_abstention_reason"] == "representation_exceeds_max_sequence_length" + assert evaluation_status(0) == ( + "CANDIDATE_INSUFFICIENT", + "NEXT_CANDIDATE_SOURCE_EVALUATION_AUTHORIZATION", + ) + assert evaluation_status(3) == ( + "CANDIDATE_DISTRIBUTION_FROZEN", + "RESIDUAL_THRESHOLD_FREEZE_AUTHORIZATION", + ) + + +def test_percentile_uses_preregistered_linear_interpolation(): + values = ["0.0000000000", "1.0000000000", "2.0000000000", "3.0000000000"] + assert percentile(values, 25) == "0.7500000000" + assert percentile(values, 50) == "1.5000000000" + assert percentile(values, 75) == "2.2500000000" + empty = distribution([]) + assert empty["status"] == "NOT_COMPUTABLE" + assert empty["median"] is None + filled = distribution(values) + assert filled["status"] == "DESCRIPTIVE" + assert filled["min"] == "0.0000000000" + assert filled["max"] == "3.0000000000" + assert filled["mean"] == "1.5000000000" diff --git a/tests/shadow/test_unbind_screen_v4.py b/tests/shadow/test_unbind_screen_v4.py new file mode 100644 index 00000000..f408094e --- /dev/null +++ b/tests/shadow/test_unbind_screen_v4.py @@ -0,0 +1,236 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v3 import screen +from hyperlexical.unbind_screen_v4 import ( + EmptyLexicon, + ScreenV4Error, + apply_v4, + assess, + measurement_allowed, + normalize_lexical, + rule_surface_violations, +) + +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", +) +V3_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v3.py" +V4_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v4.py" + + +class MapLex: + def __init__(self, nouns, adjectives=()): + self._nouns = nouns + self._adjectives = set(adjectives) + + def noun_lex(self, lemma): + return self._nouns.get(lemma) + + def has_adjective(self, lemma): + return lemma in self._adjectives + + +class BoomLex: + def noun_lex(self, lemma): + raise AssertionError(lemma) + + def has_adjective(self, lemma): + raise AssertionError(lemma) + + +def test_normalization_folds_hyphen_and_keeps_apostrophe(): + assert normalize_lexical("Seventy-Five") == normalize_lexical("seventy five") + assert normalize_lexical("one-hundred") == normalize_lexical("one hundred") + assert "'" in normalize_lexical("one's birthday") + + +def test_v3_number_grammar_does_not_fold_a_hyphen(): + bucket, _rule, _phase = screen("forty-two fifty", "adj", ["forty-two", "fifty"], "a count") + assert bucket == "SECONDARY" + + +def test_v3_spaced_number_still_rejects(): + bucket, rule, _phase = screen("forty two", "adj", ["forty", "two"], "a count") + assert bucket == "REJECT" + assert rule == "productive_numeric_expression" + + +def test_v4_folds_hyphenated_numbers_only_from_secondary(): + moved = apply_v4("SECONDARY", "forty-two fifty", "a count", "adj", EmptyLexicon()) + assert moved["v4_bucket"] == "REJECT" + assert moved["primary_evidence"] == "productive_number" + assert moved["inspected"] is True + held = apply_v4("HIGH_VALUE", "forty-two fifty", "a count", "adj", BoomLex()) + assert held["v4_bucket"] == "HIGH" + assert held["primary_evidence"] is None + assert held["inspected"] is False + rejected = apply_v4("REJECT", "four times", "by a factor of four", "adv", BoomLex()) + assert rejected["v4_bucket"] == "REJECT" + assert rejected["inspected"] is False + + +def test_multiplier_is_a_productive_number(): + moved = apply_v4("SECONDARY", "four times", "by a factor of four", "adv", EmptyLexicon()) + assert moved["v4_bucket"] == "REJECT" + assert moved["primary_evidence"] == "productive_number" + assert moved["supporting_evidence"] == [] + + +def test_comparative_particle_stays_secondary_and_shifted_particle_promotes(): + stayed = apply_v4("SECONDARY", "better off", "in a more fortunate condition", "adj", EmptyLexicon()) + assert stayed["v4_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + promoted = apply_v4("SECONDARY", "zorp up", "having a strong distaste", "adj", EmptyLexicon()) + assert promoted["v4_bucket"] == "HIGH" + assert promoted["primary_evidence"] == "noncompositional_phrasal_binding" + + +def test_literal_particle_with_stem_overlap_stays(): + stayed = apply_v4("SECONDARY", "flare out", "become flared and widen", "verb", EmptyLexicon()) + assert stayed["v4_bucket"] == "SECONDARY" + + +def test_patch_b_frames_are_classes(): + cases = [ + ("strike the ceiling", "get very angry", "verb", "nonliteral_semantic_shift"), + ("like a flash", "without delay", "adv", "conventionalized_idiom"), + ("sure as sunrise", "absolutely certain", "adj", "conventionalized_idiom"), + ("for all practical senses", "in every practical way", "adv", "conventionalized_idiom"), + ("give it a whirl", "whirl", "verb", "conventionalized_idiom"), + ("spin on a coin", "have a small turning radius", "verb", "nonliteral_semantic_shift"), + ("in the civic gaze", "of great interest to the civic world", "adj", "nonliteral_semantic_shift"), + ("stone cold", "without heat", "adj", "fixed_lexicalized_expression"), + ("cooked for finished", "destroyed", "adj", "fixed_lexicalized_expression"), + ("blip a zorp", "show nothing", "verb", "nonliteral_semantic_shift"), + ] + for surface, gloss, pos, evidence in cases: + moved = apply_v4("SECONDARY", surface, gloss, pos, EmptyLexicon()) + assert moved["v4_bucket"] == "HIGH", surface + assert moved["primary_evidence"] == evidence, surface + assert moved["primary_evidence"] not in moved["supporting_evidence"] + + +def test_patch_a_uses_gloss_lexfile_not_the_surface_string(): + person = apply_v4( + "SECONDARY", + "alice brooke carter", + "Scottish physiologist and sibling", + "noun", + MapLex({"scottish": "noun.person", "physiologist": "noun.person"}, {"scottish"}), + ) + assert person["primary_evidence"] == "multi_token_person_name" + role = apply_v4( + "SECONDARY", + "sneak thief", + "a thief who steals", + "noun", + MapLex({"thief": "noun.person"}), + ) + assert role["v4_bucket"] == "SECONDARY" + species = apply_v4( + "SECONDARY", + "ribbon lemur", + "Indian macaque with a tuft", + "noun", + MapLex({"indian": "noun.person", "macaque": "noun.animal"}, {"indian"}), + ) + assert species["primary_evidence"] == "species_or_common_name_referent" + organization = apply_v4( + "SECONDARY", + "warden of the seal", + "a small gang of fighters", + "noun", + MapLex({"small": "noun.cognition", "gang": "noun.group"}, {"small"}), + ) + assert organization["primary_evidence"] == "organization_from_gloss" + medical = apply_v4( + "SECONDARY", + "separation of the choroid", + "visual impairment resulting from the retina becoming separated", + "noun", + MapLex( + { + "impairment": "noun.event", + "retina": "noun.body", + "choroid": "noun.body", + "eye": "noun.body", + }, + {"visual"}, + ), + ) + assert medical["primary_evidence"] == "medical_technical_expression" + + +def test_scorer_source_has_no_probe_surface_or_phrase_literal(): + for path in (V3_PATH, V4_PATH): + source = path.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + constants = [ + node.value + for node in ast.walk(tree) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + ] + for phrase in PROBE_SURFACES: + assert phrase not in constants + assert phrase not in source + + +def test_gate_passes_only_lawful_secondary_moves(): + rows = [ + _replay("alpha beta", "HIGH", "HIGH", "HIGH"), + _replay("gamma delta", "REJECT", "REJECT", "REJECT"), + _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "conventionalized_idiom"), + _replay("eta theta", "SECONDARY", "REJECT", "REJECT", "productive_number"), + _replay("iota kappa", "SECONDARY", "SECONDARY", "SECONDARY"), + ] + report = assess(rows, phrase_specific_rule_fired=False, expected_rows=5) + assert report["regression"] == "REGRESSION_VERIFIED" + assert report["failures"] == [] + assert measurement_allowed(report) is True + broken = list(rows) + broken[0] = _replay("alpha beta", "HIGH", "SECONDARY", "HIGH") + failed = assess(broken, phrase_specific_rule_fired=False, expected_rows=5) + assert failed["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(failed) is False + wrong_patch = list(rows) + wrong_patch[2] = _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "productive_number") + assert assess(wrong_patch, phrase_specific_rule_fired=False, expected_rows=5)["regression"] == "REGRESSION_FAILED" + phrase = assess(rows, phrase_specific_rule_fired=True, expected_rows=5) + assert phrase["regression"] == "REGRESSION_FAILED" + assert phrase["phrase_specific_rule_fired"] is True + + +def test_unknown_bucket_is_refused(): + with pytest.raises(ScreenV4Error): + apply_v4("MAYBE", "alpha beta", "gloss", "noun", EmptyLexicon()) + + +def _replay(surface, v3, v4, operator, primary=None): + return { + "surface": surface, + "v3_bucket": v3, + "v4_bucket": v4, + "operator_bucket": operator, + "primary_evidence": primary, + "supporting_evidence": [], + } diff --git a/tests/shadow/test_unbind_screen_v5.py b/tests/shadow/test_unbind_screen_v5.py new file mode 100644 index 00000000..aae38a84 --- /dev/null +++ b/tests/shadow/test_unbind_screen_v5.py @@ -0,0 +1,266 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v5 import ( + EmptyLexicon, + Entry, + Sense, + apply_v5, + assess, + measurement_allowed, +) +from hyperlexical.unbind_screen_v4 import rule_surface_violations + +PROBE_SURFACES = ( + "hit the roof", + "get it on", + "like a shot", + "fed up", + "taken for granted", + "turn on a dime", + "in the public eye", + "bonnet monkey", + "john scott haldane", + "bearer of the sword", + "detachment of the retina", + "three times", + "one hundred seventy-five", + "on the go", + "flat out", + "in the way", + "to a t", + "slip of the tongue", + "run low", + ".22 caliber", + "phi correlation", + "blue-eyed african daisy", + "monoamine oxidase inhibitor", + "air force research laboratory", + "martin luther king jr's birthday", +) +V5_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v5.py" + + +class BoomLex: + def entry(self, surface): + raise AssertionError(surface) + + def senses(self, token): + raise AssertionError(token) + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def test_outer_buckets_pass_through_without_inspection(): + high = apply_v5("HIGH_VALUE", "alpha beta", "gloss", "noun", BoomLex()) + assert high["v5_bucket"] == "HIGH" + assert high["v4_bucket"] == "HIGH" + assert high["primary_evidence"] is None + assert high["inspected"] is False + rejected = apply_v5("REJECT", "gamma delta", "gloss", "noun", BoomLex()) + assert rejected["v5_bucket"] == "REJECT" + assert rejected["inspected"] is False + assert rejected["supporting_evidence"] == [] + + +def test_proper_name_and_pertainym_designate(): + named = MapLex( + {"north example laboratory": _entry("noun", "North_Example_Laboratory", "noun.artifact", ("North_Example_Laboratory",))}, + {}, + ) + moved = apply_v5("SECONDARY", "north example laboratory", "a workplace", "noun", named) + assert moved["v5_bucket"] == "REJECT" + assert moved["primary_evidence"] == "referential_terminological_dominance" + pert = MapLex( + {"sample bore": _entry("adj", "sample_bore", "adj.pert", ("sample_bore",))}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "bore": (_sense("bore", "noun", "noun.attribute", "a grade of excellence"),)}, + ) + designated = apply_v5("SECONDARY", "sample bore", "of or relating to a measured width", "adj", pert) + assert designated["v5_bucket"] == "REJECT" + assert designated["supporting_evidence"] == [] + + +def test_exocentric_life_form_rejects_and_endocentric_life_form_stays(): + exo = MapLex( + {"sample bloom": _entry("noun", "sample_bloom", "noun.plant", ("sample_bloom",), (("flower",),))}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "bloom": (_sense("bloom", "noun", "noun.plant", "a flower of a plant"),)}, + ) + # constituent gloss repeats flower, so this row is recoverable and must not use that overlap + # to avoid designation. Designation is decided from the synset, before recoverability. + assert apply_v5("SECONDARY", "sample bloom", "a perennial herb", "noun", exo)["v5_bucket"] == "REJECT" + endo = MapLex( + {"sample tree": _entry("noun", "sample_tree", "noun.plant", ("sample_tree",), (("tree",),))}, + {"sample": (_sense("sample", "adj", "adj.all", "illustrative"),), + "tree": (_sense("tree", "noun", "noun.plant", "a tall woody plant"),)}, + ) + stayed = apply_v5("SECONDARY", "sample tree", "a tall woody plant of the sample kind", "noun", endo) + assert stayed["v5_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + + +def test_compound_category_rejects_unless_an_ordinary_synonym_is_present(): + term = MapLex( + {"phi index": _entry( + "noun", "phi_index", "noun.cognition", ("phi_index",), (("nonparametric_statistic",),) + )}, + {"phi": (_sense("phi", "noun", "noun.communication", "a letter of an alphabet"),), + "index": (_sense("index", "noun", "noun.relation", "a numerical scale"),)}, + ) + assert apply_v5("SECONDARY", "phi index", "a statistic of agreement", "noun", term)["primary_evidence"] == "referential_terminological_dominance" + goods = MapLex( + {"sample goods": _entry( + "noun", "sample_goods", "noun.artifact", ("sample_goods", "haberdashery"), (("soft_goods",),) + )}, + {"sample": (_sense("sample", "noun", "noun.artifact", "an illustrative item"),), + "goods": (_sense("goods", "noun", "noun.artifact", "articles of commerce"),)}, + ) + stayed = apply_v5("SECONDARY", "sample goods", "articles of commerce", "noun", goods) + assert stayed["v5_bucket"] == "SECONDARY" + + +def test_adverb_idiom_promotes_and_recoverable_adverb_stays(): + idiom = MapLex( + {"blip out": _entry("adv", "blip_out", "adv.all", ("blip_out", "brusquely"))}, + {"blip": (_sense("blip", "noun", "noun.event", "a small mark on a screen"),), + "out": (_sense("out", "adv", "adv.all", "away from the inside"),)}, + ) + moved = apply_v5("SECONDARY", "blip out", "in a blunt manner", "adv", idiom) + assert moved["v5_bucket"] == "HIGH" + assert moved["primary_evidence"] == "lexicalized_noncompositional" + plain = MapLex( + {"as common": _entry("adv", "as_common", "adv.all", ("as_common", "commonly"))}, + {"common": (_sense("common", "adj", "adj.all", "occurring in the common manner"),)}, + ) + stayed = apply_v5("SECONDARY", "as common", "in the common manner", "adv", plain) + assert stayed["v5_bucket"] == "SECONDARY" + + +def test_orthography_body_and_verb_shift_are_noncompositional(): + ortho = MapLex( + {"to a q": _entry("adv", "to_a_q", "adv.all", ("to_a_q", "to_the_letter"))}, + {}, + ) + assert apply_v5("SECONDARY", "to a q", "in every respect", "adv", ortho)["v5_bucket"] == "HIGH" + body = MapLex( + {"slip of the digit": _entry("noun", "slip_of_the_digit", "noun.communication", ("slip_of_the_digit",))}, + {"slip": (_sense("slip", "noun", "noun.act", "a socially awkward act"),), + "digit": (_sense("digit", "noun", "noun.body", "a finger or toe"),)}, + ) + assert apply_v5("SECONDARY", "slip of the digit", "an accidental mistake in counting", "noun", body)["v5_bucket"] == "HIGH" + shifted = MapLex( + {"dwindle low": _entry("verb", "dwindle_low", "verb.consumption", ("dwindle_low",))}, + {"dwindle": (_sense("dwindle", "verb", "verb.motion", "move fast on foot"),), + "low": (_sense("low", "adj", "adj.all", "not high"),)}, + ) + assert apply_v5("SECONDARY", "dwindle low", "to be spent or finished", "verb", shifted)["v5_bucket"] == "HIGH" + same = MapLex( + {"nudge at": _entry("verb", "nudge_at", "verb.contact", ("nudge_at", "prod"))}, + {"nudge": (_sense("nudge", "verb", "verb.contact", "poke or thrust abruptly"),)}, + ) + assert apply_v5("SECONDARY", "nudge at", "to push against gently", "verb", same)["v5_bucket"] == "SECONDARY" + + +def test_metalinguistic_reduplication_and_missing_lemma_stay(): + discourse = MapLex( + {"by the quip": _entry("adv", "by_the_quip", "adv.all", ("by_the_quip", "incidentally"))}, + {"quip": (_sense("quip", "noun", "noun.communication", "a witty remark"),)}, + ) + assert apply_v5("SECONDARY", "by the quip", "introducing a different topic", "adv", discourse)["v5_bucket"] == "SECONDARY" + repeated = MapLex( + {"plink by plink": _entry("adv", "plink_by_plink", "adv.all", ("plink_by_plink", "gradually"))}, + {"plink": (_sense("plink", "noun", "noun.event", "a short sound"),)}, + ) + assert apply_v5("SECONDARY", "plink by plink", "in a gradual manner", "adv", repeated)["v5_bucket"] == "SECONDARY" + loan = MapLex( + {"al zorp": _entry("adj", "al_zorp", "adj.all", ("al_zorp",))}, + {"al": (_sense("al", "noun", "noun.substance", "a metallic element"),)}, + ) + assert apply_v5("SECONDARY", "al zorp", "of pasta cooked firm", "adj", loan)["v5_bucket"] == "SECONDARY" + + +def test_unknown_surface_stays_secondary(): + stayed = apply_v5("SECONDARY", "brand new coinage", "a fresh phrase", "noun", EmptyLexicon()) + assert stayed["v5_bucket"] == "SECONDARY" + assert stayed["primary_evidence"] is None + assert stayed["inspected"] is True + + +def test_scorer_source_has_no_probe_surface_or_phrase_literal(): + source = V5_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + constants = [ + node.value + for node in ast.walk(tree) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + ] + for phrase in PROBE_SURFACES: + assert phrase not in source + assert phrase not in constants + + +def test_gate_passes_only_lawful_secondary_moves(): + rows = [ + _replay("alpha beta", "HIGH", "HIGH", "HIGH"), + _replay("gamma delta", "REJECT", "REJECT", "REJECT"), + _replay("epsilon zeta", "SECONDARY", "HIGH", "HIGH", "lexicalized_noncompositional"), + _replay("eta theta", "SECONDARY", "REJECT", "REJECT", "referential_terminological_dominance"), + _replay("iota kappa", "SECONDARY", "SECONDARY", "SECONDARY"), + ] + report = assess(rows, phrase_specific_rule_fired=False, expected_rows=5) + assert report["regression"] == "REGRESSION_VERIFIED" + assert report["failures"] == [] + assert report["operator_conflict_on_move"] == 0 + assert measurement_allowed(report) is True + broken = list(rows) + broken[0] = _replay("alpha beta", "HIGH", "SECONDARY", "HIGH") + failed = assess(broken, phrase_specific_rule_fired=False, expected_rows=5) + assert failed["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(failed) is False + conflict = list(rows) + conflict[2] = _replay("epsilon zeta", "SECONDARY", "HIGH", "SECONDARY", "lexicalized_noncompositional") + assert assess(conflict, phrase_specific_rule_fired=False, expected_rows=5)["operator_conflict_on_move"] == 1 + phrase = assess(rows, phrase_specific_rule_fired=True, expected_rows=5) + assert phrase["regression"] == "REGRESSION_FAILED" + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v5("MAYBE", "alpha beta", "gloss", "noun", EmptyLexicon()) + + +def _replay(surface, prior, nxt, operator, primary=None): + return { + "surface": surface, + "v4_bucket": prior, + "v5_bucket": nxt, + "operator_bucket": operator, + "primary_evidence": primary, + "supporting_evidence": [], + } diff --git a/tests/shadow/test_unbind_screen_v6.py b/tests/shadow/test_unbind_screen_v6.py new file mode 100644 index 00000000..0c88c6df --- /dev/null +++ b/tests/shadow/test_unbind_screen_v6.py @@ -0,0 +1,237 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_screen_v5 import Entry, Sense +from hyperlexical.unbind_screen_v6 import ( + COMPOSITIONAL_EVIDENCE, + NONREFERENTIAL_EVIDENCE, + apply_v6, + assess, + measurement_allowed, +) + +PROBE_SURFACES = ( + "keep out", + "on the job", + "hit the roof", + "pop the question", + "all of a sudden", + ".22 caliber", +) +V6_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v6.py" + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def _compositional_lex(): + return MapLex( + {"fully shut": _entry("verb", "Fully_Shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + + +def test_compositional_high_falls_to_secondary_and_stops(): + decision = apply_v6("HIGH", "fully shut", "completely prevent entering", "verb", _compositional_lex()) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == COMPOSITIONAL_EVIDENCE + assert decision["supporting_evidence"] == [] + + +def test_idiomatic_mapping_blocks_post_hoc_rationalization(): + lexicon = MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut", "combust"))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + decision = apply_v6("HIGH", "fully shut", "completely prevent entering", "verb", lexicon) + assert decision["v6_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_single_constituent_gloss_is_not_ordinary_syntax(): + lexicon = MapLex( + {"quite sudden": _entry("adv", "quite_sudden", "adv.all", ("quite_sudden",))}, + { + "quite": (_sense("quite", "adv", "adv.all", "to a degree"),), + "sudden": (_sense("sudden", "adj", "adj.all", "happening without warning"),), + }, + ) + decision = apply_v6("HIGH", "quite sudden", "without warning", "adv", lexicon) + assert decision["v6_bucket"] == "HIGH" + + +def test_state_pertainym_falls_to_secondary(): + lexicon = MapLex( + {"on duty": _entry("adj", "on_duty", "adj.pert", ("on_duty",))}, + {"duty": (_sense("duty", "noun", "noun.act", "work that is a paid activity"),)}, + ) + decision = apply_v6("REJECT", "on duty", "actively engaged in paid work", "adj", lexicon) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == NONREFERENTIAL_EVIDENCE + + +def test_relational_pertainym_stays_a_designation(): + lexicon = MapLex( + {"gun bore": _entry("adj", "gun_bore", "adj.pert", ("gun_bore",))}, + {"bore": (_sense("bore", "noun", "noun.attribute", "a degree of excellence"),)}, + ) + decision = apply_v6( + "REJECT", + "gun bore", + "of or relating to the bore of a gun", + "adj", + lexicon, + ) + assert decision["v6_bucket"] == "REJECT" + assert decision["primary_evidence"] is None + + +def test_sentence_like_naming_gloss_stays_reject(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",), (("terrorist_group",),))}, + { + "alpha": (_sense("alpha", "noun", "noun.communication", "the first letter"),), + "force": (_sense("force", "noun", "noun.group", "a group of people"),), + }, + ) + decision = apply_v6( + "REJECT", + "alpha force", + "a violent group that seeks a separate state for its members", + "noun", + lexicon, + ) + assert decision["v6_bucket"] == "REJECT" + + +def test_provisional_secondary_can_still_reject(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + {"force": (_sense("force", "noun", "noun.group", "a group of people"),)}, + ) + decision = apply_v6("SECONDARY", "alpha force", "a named group", "noun", lexicon) + assert decision["v6_bucket"] == "REJECT" + assert decision["primary_evidence"] == "referential_terminological_dominance" + + +def test_recoverable_secondary_is_not_promoted(): + decision = apply_v6( + "SECONDARY", + "fully shut", + "completely prevent entering", + "verb", + MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ), + ) + assert decision["v6_bucket"] == "SECONDARY" + assert decision["primary_evidence"] is None + + +def test_previously_correct_is_not_bucket_immutability(): + held = assess( + [ + { + "surface": "still right", + "v5_bucket": "HIGH", + "v6_bucket": "HIGH", + "operator_bucket": "HIGH", + "primary_evidence": None, + "supporting_evidence": [], + }, + { + "surface": "corrected", + "v5_bucket": "REJECT", + "v6_bucket": "SECONDARY", + "operator_bucket": "SECONDARY", + "primary_evidence": NONREFERENTIAL_EVIDENCE, + "supporting_evidence": [], + }, + ], + phrase_specific_rule_fired=False, + expected_rows=2, + ) + assert held["previously_correct_lost"] == 0 + assert held["failures"] == [] + broken = assess( + [ + { + "surface": "was right", + "v5_bucket": "HIGH", + "v6_bucket": "SECONDARY", + "operator_bucket": "HIGH", + "primary_evidence": COMPOSITIONAL_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert broken["previously_correct_lost"] == 1 + assert broken["regression"] == "REGRESSION_FAILED" + assert measurement_allowed(broken) is False + + +def test_direct_outer_swap_fails_the_gate(): + report = assess( + [ + { + "surface": "swapped", + "v5_bucket": "HIGH", + "v6_bucket": "REJECT", + "operator_bucket": "REJECT", + "primary_evidence": "referential_terminological_dominance", + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert report["direct_swaps"] == 1 + assert "HIGH and REJECT swapped directly" in report["failures"] + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v6("MAYBE", "fully shut", "gloss", "verb", _compositional_lex()) + + +def test_probe_surfaces_are_not_rules(): + source = V6_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + tree = ast.parse(source) + assert any(isinstance(node, ast.FunctionDef) and node.name == "apply_v6" for node in ast.walk(tree)) diff --git a/tests/shadow/test_unbind_screen_v7.py b/tests/shadow/test_unbind_screen_v7.py new file mode 100644 index 00000000..23a3ab00 --- /dev/null +++ b/tests/shadow/test_unbind_screen_v7.py @@ -0,0 +1,267 @@ +import ast +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_screen_v5 import Entry, Sense +from hyperlexical.unbind_screen_v6 import COMPOSITIONAL_EVIDENCE, apply_v6 +from hyperlexical.unbind_screen_v7 import ( + ORDINARY_EVIDENCE, + apply_v7, + assess, + ordinary_compositional_derivation, +) + +PROBE_SURFACES = ( + "keep out", + "to a lesser extent", + "to the letter", + "with child", + "dressed to the nines", + "union jack", + "atomic number 98", + "law of definite proportions", +) +V7_PATH = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_screen_v7.py" + + +class MapLex: + def __init__(self, entries, senses): + self._entries = entries + self._senses = senses + + def entry(self, surface): + return self._entries.get(surface.casefold()) + + def senses(self, token): + return tuple(self._senses.get(token, ())) + + +def _sense(token, pos, lex, gloss): + return Sense(token, pos, lex, gloss) + + +def _entry(pos, lemma, lex, lemmas, hypernyms=()): + return Entry(pos, lemma, lex, tuple(lemmas), tuple(tuple(item) for item in hypernyms)) + + +def _comparative_lex(): + return MapLex( + {"to a greater degree": _entry("adv", "to_a_greater_degree", "adv.all", ("to_a_greater_degree",))}, + { + "greater": (_sense("greater", "adj", "adj.all", "of greater size"),), + "degree": (_sense("degree", "noun", "noun.attribute", "a position on a scale"),), + }, + ) + + +def _syntactic_lex(lemmas=("fully_shut",)): + return MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", lemmas)}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering"),), + }, + ) + + +def _phrasal_lex(): + return MapLex( + {"move out": _entry("verb", "move_out", "verb.motion", ("move_out",))}, + { + "move": (_sense("move", "verb", "verb.motion", "change position"),), + "out": (_sense("out", "adv", "adv.all", "away outside"),), + }, + ) + + +def test_ordinary_comparative_composition_may_fire(): + decision = apply_v7( + "HIGH", + "to a greater degree", + "used to form the comparative of some adjectives and adverbs", + "adv", + _comparative_lex(), + ) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + assert decision["supporting_evidence"] == [] + assert ordinary_compositional_derivation( + "to a greater degree", + "used to form the comparative of some adjectives and adverbs", + _comparative_lex(), + ) + + +def test_ordinary_syntactic_composition_may_fire(): + decision = apply_v7("HIGH", "fully shut", "completely prevent entering", "verb", _syntactic_lex()) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + + +def test_ordinary_phrasal_composition_may_fire(): + decision = apply_v7("HIGH", "move out", "change position away outside", "verb", _phrasal_lex()) + assert decision["v7_bucket"] == "SECONDARY" + assert decision["primary_evidence"] == ORDINARY_EVIDENCE + + +def test_conventionalized_idiom_must_not_fire(): + decision = apply_v7( + "HIGH", + "fully shut", + "completely prevent entering", + "verb", + _syntactic_lex(("fully_shut", "combust")), + ) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_post_hoc_metaphor_must_not_fire(): + decision = apply_v7( + "HIGH", + "fully shut", + "metaphorically completely prevent entering", + "verb", + _syntactic_lex(), + ) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + + +def test_gloss_resemblance_must_not_fire(): + lexicon = MapLex( + {"fully shut": _entry("verb", "fully_shut", "verb.contact", ("fully_shut",))}, + { + "fully": (_sense("fully", "adv", "adv.all", "completely done"),), + "shut": (_sense("shut", "verb", "verb.contact", "prevent entering now"),), + }, + ) + assert apply_v6("HIGH", "fully shut", "completely prevent", "verb", lexicon)["v6_bucket"] == "SECONDARY" + assert apply_v6("HIGH", "fully shut", "completely prevent", "verb", lexicon)["primary_evidence"] == COMPOSITIONAL_EVIDENCE + decision = apply_v7("HIGH", "fully shut", "completely prevent", "verb", lexicon) + assert decision["v7_bucket"] == "HIGH" + assert decision["primary_evidence"] is None + assert decision["primary_evidence"] != COMPOSITIONAL_EVIDENCE + + +def test_named_phrase_is_irrelevant_to_the_high_challenge(): + lexicon = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + { + "alpha": (_sense("alpha", "noun", "noun.communication", "the first letter"),), + "force": (_sense("force", "noun", "noun.group", "a group of people"),), + }, + ) + assert ordinary_compositional_derivation("alpha force", "a named group", lexicon) is False + high = apply_v7("HIGH", "alpha force", "a named group", "noun", lexicon) + assert high["v7_bucket"] == "HIGH" + assert high["primary_evidence"] is None + rejected = apply_v7("REJECT", "alpha force", "a named group", "noun", lexicon) + inherited = apply_v6("REJECT", "alpha force", "a named group", "noun", lexicon) + assert rejected["v7_bucket"] == inherited["v6_bucket"] == "REJECT" + assert rejected["primary_evidence"] != ORDINARY_EVIDENCE + + +def test_demotion_stops_at_secondary(): + lexicon = _comparative_lex() + gloss = "used to form the comparative of some adjectives and adverbs" + demoted = apply_v7("HIGH", "to a greater degree", gloss, "adv", lexicon) + assert demoted["v7_bucket"] == "SECONDARY" + stopped = apply_v7("SECONDARY", "to a greater degree", gloss, "adv", lexicon) + assert stopped["v7_bucket"] == "SECONDARY" + assert stopped["primary_evidence"] is None + + +def test_reject_behavior_matches_v6_and_secondary_is_not_reopened(): + pertainym = MapLex( + {"on duty": _entry("adj", "on_duty", "adj.pert", ("on_duty",))}, + {"duty": (_sense("duty", "noun", "noun.act", "work that is a paid activity"),)}, + ) + rejected = apply_v7("REJECT", "on duty", "actively engaged in paid work", "adj", pertainym) + inherited = apply_v6("REJECT", "on duty", "actively engaged in paid work", "adj", pertainym) + assert rejected["v7_bucket"] == inherited["v6_bucket"] == "SECONDARY" + assert rejected["primary_evidence"] == inherited["primary_evidence"] == "nonreferential_lexical_use" + naming = MapLex( + {"alpha force": _entry("noun", "Alpha_Force", "noun.group", ("Alpha_Force",))}, + {"force": (_sense("force", "noun", "noun.group", "a group of people"),)}, + ) + assert apply_v6("SECONDARY", "alpha force", "a named group", "noun", naming)["v6_bucket"] == "REJECT" + held = apply_v7("SECONDARY", "alpha force", "a named group", "noun", naming) + assert held["v7_bucket"] == "SECONDARY" + assert held["primary_evidence"] is None + + +def test_previously_correct_is_not_bucket_immutability(): + allowed = assess( + [ + { + "surface": "may retreat", + "v6_bucket": "HIGH", + "v7_bucket": "SECONDARY", + "operator_bucket": "SECONDARY", + "primary_evidence": ORDINARY_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert allowed["previously_correct_lost"] == 0 + assert allowed["correct_high_lost"] == 0 + assert allowed["regression"] == "REGRESSION_VERIFIED" + assert allowed["measurement_eligible"] is False + broken = assess( + [ + { + "surface": "was right", + "v6_bucket": "HIGH", + "v7_bucket": "SECONDARY", + "operator_bucket": "HIGH", + "primary_evidence": ORDINARY_EVIDENCE, + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert broken["previously_correct_lost"] == 1 + assert broken["correct_high_lost"] == 1 + assert broken["regression"] == "REGRESSION_FAILED" + + +def test_direct_outer_swap_fails_the_gate(): + report = assess( + [ + { + "surface": "swapped", + "v6_bucket": "HIGH", + "v7_bucket": "REJECT", + "operator_bucket": "REJECT", + "primary_evidence": "referential_terminological_dominance", + "supporting_evidence": [], + } + ], + phrase_specific_rule_fired=False, + expected_rows=1, + ) + assert report["direct_swaps"] == 1 + assert "HIGH and REJECT swapped directly" in report["failures"] + + +def test_unknown_bucket_is_refused(): + with pytest.raises(Exception): + apply_v7("MAYBE", "move out", "gloss", "verb", _phrasal_lex()) + + +def test_probe_surfaces_are_not_rules(): + source = V7_PATH.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBE_SURFACES) == [] + assert "compositional_recoverability" not in source + tree = ast.parse(source) + assert any(isinstance(node, ast.FunctionDef) and node.name == "apply_v7" for node in ast.walk(tree)) diff --git a/tests/shadow/test_unbind_sense_screen_v1.py b/tests/shadow/test_unbind_sense_screen_v1.py new file mode 100644 index 00000000..dd1e1fb9 --- /dev/null +++ b/tests/shadow/test_unbind_sense_screen_v1.py @@ -0,0 +1,268 @@ +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_sense_screen_v1 import Pointer, Synset, classify, parse_data_line + +SCORER = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_sense_screen_v1.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _synset(ss_type, lemmas, pointers=()): + return Synset("00000000", ss_type, tuple(lemmas), tuple(pointers)) + + +def _pointer(symbol, source=0, target=0, offset="00000011", pos="n"): + return Pointer(symbol, offset, pos, source, target) + + +def _class(surface, gloss, synset, exceptions=None, targets=None): + return classify(surface, gloss, synset, exceptions or {}, targets or {}) + + +def test_instance_hypernym_is_referential(): + found = _class("alpha beta", "a particular named thing", _synset("n", ("alpha_beta",), (_pointer("@i"),))) + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + assert found["primary_evidence_code"] == "referential_designation" + assert found["evidence_source"] == "synset.instance_hypernym" + assert found["confidence_status"] == "DETERMINATE" + + +def test_lifespan_is_referential_and_instance_wins_when_both_fire(): + life = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",))) + assert life["evidence_source"] == "synset.gloss.lifespan" + both = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",), (_pointer("@i"),))) + assert both["sense_class"] == "REFERENTIAL" + assert both["evidence_source"] == "synset.instance_hypernym" + + +def test_year_range_without_parentheses_is_not_a_lifespan(): + found = _class("alpha beta", "a span from 1870-1949", _synset("n", ("alpha_beta",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_geographic_allusion_and_capitalization_are_not_referential(): + gloss = "a sudden turning point in a life (similar to a story on the road from one city to another)" + found = _class( + "path to example", + gloss, + _synset("n", ("Path_to_Example",), (_pointer("@", source=0),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["confidence_status"] == "INSUFFICIENT" + + +def test_technical_gloss_without_a_record_signal_stays_ambiguous(): + found = _class("sample statute", "a law about a measurement of a species", _synset("n", ("sample_statute",))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_unrelated_single_word_colemma_is_noncompositional(): + found = _class("alpha beta", "a stored predicate", _synset("v", ("alpha_beta", "exclude"))) + assert found["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL" + assert found["bucket"] == "HIGH" + assert found["evidence_source"] == "synset.lemmas.unrelated_single_word" + + +def test_constituent_and_its_exception_form_are_not_unrelated(): + bare = _class("alpha beta", "a predicate", _synset("v", ("alpha_beta", "alpha"))) + assert bare["sense_class"] == "AMBIGUOUS" + inflected = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "alphas")), + {"alphas": {"alpha"}, "alpha": {"alphas"}}, + ) + assert inflected["sense_class"] == "AMBIGUOUS" + + +def test_lexical_pointer_without_a_constituent_relation_is_ambiguous(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta",), (_pointer("!", source=1, target=1),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["evidence_source"] == "none" + + +def test_semantic_pointer_is_not_a_lexical_unit(): + found = _class( + "alpha beta", + "a predicate", + _synset("n", ("alpha_beta",), (_pointer("!", source=0, target=0),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_pointer_on_the_other_lemma_does_not_count(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "gamma_delta"), (_pointer("+", source=2, target=1),)), + targets={("n", "00000011"): ("alpha",)}, + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_derivation_back_to_a_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("v", ("alpha_beta",), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["evidence_source"] == "synset.lexical_pointer.derivation_or_pertainym_to_constituent" + + +def test_pertainym_to_an_inflected_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("r", ("alpha_beta",), (_pointer("\\", source=1, target=1, pos="a"),)), + exceptions={"betas": {"beta"}, "beta": {"betas"}}, + targets={("a", "00000011"): ("betas",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["confidence_status"] == "DETERMINATE" + + +def test_unrelated_equivalent_overrides_a_constituent_pointer(): + found = _class( + "alpha beta", + "a stored predicate", + _synset("v", ("alpha_beta", "exclude"), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["sense_class"] == "LEXICALIZED_NONCOMPOSITIONAL" + + +def test_one_token_alternation_is_ordinary_composition(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["evidence_source"] == "synset.lemmas.productive_alternation" + + +def test_alternation_that_excludes_the_surface_does_not_fire(): + found = _class( + "alpha beta gamma", + "a predicate", + _synset("n", ("alpha_beta_gamma", "as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_comparative_and_superlative_formulas_are_ordinary_composition(): + comparative = _class("alpha beta", "used to form the comparative of some words", _synset("r", ("alpha_beta",))) + superlative = _class("alpha beta", "used to form the superlative of some words", _synset("r", ("alpha_beta",))) + buried = _class("alpha beta", "a phrase used to form the comparative later", _synset("r", ("alpha_beta",))) + assert comparative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert comparative["evidence_source"] == "synset.gloss.grammatical_operator" + assert superlative["evidence_source"] == "synset.gloss.grammatical_operator" + assert buried["sense_class"] == "AMBIGUOUS" + + +def test_membership_alone_is_not_secondary(): + found = _class("alpha beta", "a listed phrase", _synset("n", ("alpha_beta",), (_pointer("@"),))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_adjective_without_a_unit_signal_is_ambiguous(): + found = _class("alpha beta", "a listed modifier", _synset("a", ("alpha_beta",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_referential_yes_conflicts_with_a_productive_frame(): + found = _class( + "as wide as needed", + "a particular (1870-1949)", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed"), (_pointer("@i"),)), + ) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_unit_yes_and_no_conflict(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed", "exclude")), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_hyphen_and_underscore_count_as_the_same_lemma(): + found = _class( + "full of the moon", + "a listed name", + _synset("n", ("full-of-the-moon",)), + ) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_high_ambiguity_is_left_in_place(): + rows = [ + _class(f"item {index}", "a listed phrase", _synset("n", (f"item_{index}",))) + for index in range(5) + ] + assert [row["sense_class"] for row in rows] == ["AMBIGUOUS"] * 5 + assert [row["bucket"] for row in rows] == ["QUARANTINE"] * 5 + + +def test_data_line_parser_keeps_hex_word_numbers_and_stops_before_frames(): + line = "00013172 29 v 01 bungle 0 003 @ 00010435 v 0000 + 00074790 n 0104 + 09879744 n 0101 01 + 02 00 | spoil by behaving clumsily" + synset, gloss = parse_data_line(line) + assert gloss == "spoil by behaving clumsily" + assert synset.lemmas == ("bungle",) + assert len(synset.pointers) == 3 + assert synset.pointers[1].symbol == "+" + assert synset.pointers[1].source == 1 + assert synset.pointers[1].target == 4 + wide = ( + "00002137 03 n 02 abstraction 0 abstract_entity 0 010 " + "@ 00001740 n 0000 + 00692329 v 0101 ~ 00023100 n 0000 ~ 00024264 n 0000 " + "~ 00031264 n 0000 ~ 00031921 n 0000 ~ 00033020 n 0000 ~ 00033615 n 0000 " + "~ 05810143 n 0000 ~ 07999699 n 0000 | a general concept" + ) + parsed_wide, _gloss = parse_data_line(wide) + assert len(parsed_wide.pointers) == 10 + assert parsed_wide.pointers[1].symbol == "+" + assert parsed_wide.pointers[1].source == 1 + assert parsed_wide.pointers[1].target == 1 + lifespan = "00000001 11 n 01 alpha_beta 0 001 @ 00000002 n 0000 | a made example (1900-1910); extra" + parsed, first = parse_data_line(lifespan) + found = _class("alpha beta", first, parsed) + assert found["sense_class"] == "REFERENTIAL" + assert found["evidence_source"] == "synset.gloss.lifespan" + + +def test_probe_surfaces_are_not_rules(): + source = SCORER.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == [] diff --git a/tests/shadow/test_unbind_sense_screen_v2.py b/tests/shadow/test_unbind_sense_screen_v2.py new file mode 100644 index 00000000..816e8a3c --- /dev/null +++ b/tests/shadow/test_unbind_sense_screen_v2.py @@ -0,0 +1,304 @@ +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts" / "shadow")) + +from hyperlexical.unbind_screen_v4 import rule_surface_violations +from hyperlexical.unbind_sense_screen_v2 import Pointer, Synset, classify, parse_data_line + +SCORER = ROOT / "scripts" / "shadow" / "hyperlexical" / "unbind_sense_screen_v2.py" +PROBES = ( + "road to damascus", + "as far as possible", + "independent state of papua new guinea", + "full phase of the moon", + "union jack", + "atomic number 98", + "law of definite proportions", + "round the bend", + "throw in the towel", + "flip one's lid", + "luck through", + "now and then", +) + + +def _synset(ss_type, lemmas, pointers=()): + return Synset("00000000", ss_type, tuple(lemmas), tuple(pointers)) + + +def _pointer(symbol, source=0, target=0, offset="00000011", pos="n"): + return Pointer(symbol, offset, pos, source, target) + + +def _class(surface, gloss, synset, exceptions=None, targets=None): + return classify(surface, gloss, synset, exceptions or {}, targets or {}) + + +def test_instance_hypernym_is_referential_and_wins_over_a_colemma(): + found = _class( + "alpha beta", + "a particular named thing", + _synset("n", ("alpha_beta", "exclude"), (_pointer("@i"),)), + ) + assert found["referential_state"] == "YES" + assert found["lexicalized_state"] == "YES" + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + assert found["primary_evidence_code"] == "referential_designation" + assert found["evidence_sources"][0] == "synset.instance_hypernym" + assert "whole_expression_lexicalization" in found["supporting_evidence_codes"] + assert found["confidence_status"] == "DETERMINATE" + + +def test_lifespan_is_referential_until_an_instance_pointer_is_also_present(): + life = _class("alpha beta", "a person (1870-1949)", _synset("n", ("alpha_beta",))) + assert life["evidence_sources"][0] == "synset.gloss.lifespan" + both = _class( + "alpha beta", + "a person (1870-1949)", + _synset("n", ("alpha_beta",), (_pointer("@i"),)), + ) + assert both["evidence_sources"][0] == "synset.instance_hypernym" + + +def test_year_range_without_parentheses_is_not_referential(): + found = _class("alpha beta", "a span from 1870-1949", _synset("n", ("alpha_beta",))) + assert found["referential_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + + +def test_allusion_and_capitalization_are_not_referential(): + gloss = "a sudden turning point in a life (similar to a story on the road from one city to another)" + found = _class("path to example", gloss, _synset("n", ("Path_to_Example",), (_pointer("@", source=0),))) + assert found["referential_state"] == "UNKNOWN" + assert found["lexicalized_state"] == "UNKNOWN" + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_technical_gloss_stays_ambiguous(): + found = _class("sample statute", "a law about a measurement of a species", _synset("n", ("sample_statute",))) + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_colemma_is_lexicalized_yes_and_not_high(): + found = _class("alpha beta", "a stored predicate", _synset("v", ("alpha_beta", "exclude"))) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "UNKNOWN" + assert found["compositional_state"] != "NO" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["primary_evidence_code"] == "insufficient_record_evidence" + assert found["supporting_evidence_codes"] == ["whole_expression_lexicalization"] + assert found["evidence_sources"] == ["synset.lemmas.unrelated_single_word"] + assert found["confidence_status"] == "INSUFFICIENT" + + +def test_constituent_and_its_inflection_are_not_an_unrelated_colemma(): + bare = _class("alpha beta", "a predicate", _synset("v", ("alpha_beta", "alpha"))) + inflected = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "alphas")), + {"alphas": {"alpha"}, "alpha": {"alphas"}}, + ) + assert bare["lexicalized_state"] == "UNKNOWN" + assert inflected["lexicalized_state"] == "UNKNOWN" + + +def test_lexical_pointer_without_a_constituent_relation_stays_ambiguous(): + found = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta",), (_pointer("!", source=1, target=1),)), + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + + +def test_semantic_pointer_and_a_pointer_on_another_lemma_do_not_lexicalize(): + semantic = _class( + "alpha beta", + "a predicate", + _synset("n", ("alpha_beta",), (_pointer("!", source=0, target=0),)), + ) + other = _class( + "alpha beta", + "a predicate", + _synset("v", ("alpha_beta", "gamma_delta"), (_pointer("+", source=2, target=1),)), + targets={("n", "00000011"): ("alpha",)}, + ) + assert semantic["lexicalized_state"] == "UNKNOWN" + assert other["lexicalized_state"] == "UNKNOWN" + assert other["compositional_state"] == "UNKNOWN" + + +def test_constituent_derivation_is_lexicalized_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("v", ("alpha_beta",), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + assert found["primary_evidence_code"] == "compositional_semantic_relation" + assert found["supporting_evidence_codes"] == ["whole_expression_lexicalization"] + assert found["evidence_sources"][0] == "synset.lexical_pointer.derivation_or_pertainym_to_constituent" + + +def test_pertainym_to_an_inflected_constituent_is_compositional(): + found = _class( + "alpha beta", + "a recoverable predicate", + _synset("r", ("alpha_beta",), (_pointer("\\", source=1, target=1, pos="a"),)), + exceptions={"betas": {"beta"}, "beta": {"betas"}}, + targets={("a", "00000011"): ("betas",)}, + ) + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["referential_state"] == "NO" + + +def test_colemma_plus_constituent_is_secondary_not_high(): + found = _class( + "alpha beta", + "a stored predicate", + _synset("v", ("alpha_beta", "exclude"), (_pointer("+", source=1, target=1, pos="v"),)), + targets={("v", "00000011"): ("beta",)}, + ) + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["bucket"] == "SECONDARY" + + +def test_colemma_plus_alternation_is_lexicalized_compositional(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed", "exclude")), + ) + assert found["referential_state"] == "NO" + assert found["lexicalized_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "LEXICALIZED_COMPOSITIONAL" + assert found["primary_evidence_code"] == "productive_grammatical_frame" + assert "whole_expression_lexicalization" in found["supporting_evidence_codes"] + + +def test_alternation_alone_is_compositional_yes_and_ambiguous(): + found = _class( + "as wide as needed", + "to a feasible extent", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["referential_state"] == "NO" + assert found["lexicalized_state"] == "UNKNOWN" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["supporting_evidence_codes"] == ["productive_grammatical_frame"] + + +def test_alternation_that_excludes_the_surface_does_not_fire(): + found = _class( + "alpha beta gamma", + "a predicate", + _synset("n", ("alpha_beta_gamma", "as_wide_as_needed", "as_deep_as_needed")), + ) + assert found["compositional_state"] == "UNKNOWN" + assert found["sense_class"] == "AMBIGUOUS" + + +def test_operator_gloss_is_ordinary_composition(): + comparative = _class("alpha beta", "used to form the comparative of some words", _synset("r", ("alpha_beta",))) + superlative = _class("alpha beta", "used to form the superlative of some words", _synset("r", ("alpha_beta",))) + buried = _class("alpha beta", "a phrase used to form the comparative later", _synset("r", ("alpha_beta",))) + assert comparative["lexicalized_state"] == "NO" + assert comparative["compositional_state"] == "YES" + assert comparative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert comparative["bucket"] == "SECONDARY" + assert comparative["primary_evidence_code"] == "productive_grammatical_frame" + assert comparative["evidence_sources"][0] == "synset.gloss.grammatical_operator" + assert superlative["sense_class"] == "ORDINARY_COMPOSITIONAL" + assert buried["sense_class"] == "AMBIGUOUS" + assert buried["lexicalized_state"] == "UNKNOWN" + + +def test_operator_gloss_plus_colemma_conflicts_and_is_not_high(): + found = _class( + "alpha beta", + "used to form the comparative of some words", + _synset("r", ("alpha_beta", "exclude")), + ) + assert found["lexicalized_state"] == "CONFLICT" + assert found["sense_class"] == "AMBIGUOUS" + assert found["bucket"] == "QUARANTINE" + assert found["primary_evidence_code"] == "conflicting_record_evidence" + assert found["confidence_status"] == "CONTRADICTORY" + + +def test_membership_alone_is_not_secondary_or_high(): + noun = _class("alpha beta", "a listed phrase", _synset("n", ("alpha_beta",), (_pointer("@"),))) + adjective = _class("alpha beta", "a listed modifier", _synset("a", ("alpha_beta",))) + assert noun["referential_state"] == "UNKNOWN" + assert noun["sense_class"] == "AMBIGUOUS" + assert noun["evidence_sources"] == ["none"] + assert adjective["referential_state"] == "NO" + assert adjective["sense_class"] == "AMBIGUOUS" + + +def test_instance_pointer_on_an_alternation_stays_referential(): + found = _class( + "as wide as needed", + "a particular (1870-1949)", + _synset("r", ("as_wide_as_needed", "as_deep_as_needed"), (_pointer("@i"),)), + ) + assert found["referential_state"] == "YES" + assert found["compositional_state"] == "YES" + assert found["sense_class"] == "REFERENTIAL" + assert found["bucket"] == "REJECT" + + +def test_hyphen_and_underscore_count_as_the_same_lemma(): + found = _class("full of the moon", "a listed name", _synset("n", ("full-of-the-moon",))) + assert found["sense_class"] == "AMBIGUOUS" + + +def test_colemma_rows_are_not_repaired_into_high(): + rows = [ + _class(f"item {index}", "a listed phrase", _synset("n", (f"item_{index}", "exclude"))) + for index in range(5) + ] + assert [row["sense_class"] for row in rows] == ["AMBIGUOUS"] * 5 + assert [row["bucket"] for row in rows] == ["QUARANTINE"] * 5 + assert [row["compositional_state"] for row in rows] == ["UNKNOWN"] * 5 + + +def test_decimal_pointer_count_stops_before_frames(): + line = "00013172 29 v 01 bungle 0 003 @ 00010435 v 0000 + 00074790 n 0104 + 09879744 n 0101 01 + 02 00 | spoil by behaving clumsily" + synset, gloss = parse_data_line(line) + assert len(synset.pointers) == 3 + wide = ( + "00002137 03 n 02 abstraction 0 abstract_entity 0 010 " + "@ 00001740 n 0000 + 00692329 v 0101 ~ 00023100 n 0000 ~ 00024264 n 0000 " + "~ 00031264 n 0000 ~ 00031921 n 0000 ~ 00033020 n 0000 ~ 00033615 n 0000 " + "~ 05810143 n 0000 ~ 07999699 n 0000 | a general concept" + ) + parsed, _gloss = parse_data_line(wide) + assert len(parsed.pointers) == 10 + found = _class("alpha beta", gloss, synset) + assert found["compositional_state"] == "UNKNOWN" + + +def test_probe_surfaces_are_not_rules(): + source = SCORER.read_text(encoding="utf-8") + assert rule_surface_violations(source, PROBES) == []